Compare commits

..

5 Commits

Author SHA1 Message Date
lizzie faa771646c fix 2026-07-03 17:44:22 +00:00
lizzie de59fa1ab4 fix 2026-07-03 17:44:12 +00:00
lizzie b79295c528 windows sucks 2026-07-03 17:44:02 +00:00
lizzie c47bae4c4e license 2026-07-03 17:44:02 +00:00
lizzie ca0eebcfe8 [core] move event creation to attached Core::System, remove unused atomics/prevent false share on Core::Timing
Signed-off-by: lizzie <lizzie@eden-emu.dev>
2026-07-03 17:44:02 +00:00
71 changed files with 366 additions and 2061 deletions
-5
View File
@@ -59,11 +59,6 @@ endif()
if (PLATFORM_PS4 OR PLATFORM_MANAGARM) if (PLATFORM_PS4 OR PLATFORM_MANAGARM)
# Doesn't support VA-API, don't go thru the embarrassment of trying to enable it # Doesn't support VA-API, don't go thru the embarrassment of trying to enable it
list(APPEND FFmpeg_HWACCEL_FLAGS --disable-vaapi) list(APPEND FFmpeg_HWACCEL_FLAGS --disable-vaapi)
elseif (ANDROID)
list(APPEND FFmpeg_HWACCEL_FLAGS
--enable-mediacodec
--enable-jni
)
elseif (UNIX AND NOT DEFINED FFmpeg_IS_CROSS_COMPILING AND NOT ANDROID) elseif (UNIX AND NOT DEFINED FFmpeg_IS_CROSS_COMPILING AND NOT ANDROID)
find_package(PkgConfig REQUIRED) find_package(PkgConfig REQUIRED)
pkg_check_modules(LIBVA libva) pkg_check_modules(LIBVA libva)
+1 -1
View File
@@ -27,7 +27,7 @@ if (ARCHITECTURE_arm64)
target_link_libraries(yuzu-android PRIVATE adrenotools) target_link_libraries(yuzu-android PRIVATE adrenotools)
endif() endif()
target_link_libraries(yuzu-android PRIVATE ${FFmpeg_LIBRARIES} OpenSSL::SSL cpp-jwt::cpp-jwt) target_link_libraries(yuzu-android PRIVATE OpenSSL::SSL cpp-jwt::cpp-jwt)
if (ENABLE_UPDATE_CHECKER) if (ENABLE_UPDATE_CHECKER)
target_compile_definitions(yuzu-android PUBLIC ENABLE_UPDATE_CHECKER) target_compile_definitions(yuzu-android PUBLIC ENABLE_UPDATE_CHECKER)
endif() endif()
-11
View File
@@ -36,10 +36,6 @@
#include <frontend_common/content_manager.h> #include <frontend_common/content_manager.h>
#include <jni.h> #include <jni.h>
extern "C" {
#include <libavcodec/jni.h>
}
#include "common/android/multiplayer/multiplayer.h" #include "common/android/multiplayer/multiplayer.h"
#include "common/android/android_common.h" #include "common/android/android_common.h"
#include "common/android/id_cache.h" #include "common/android/id_cache.h"
@@ -684,13 +680,6 @@ const char* fallback_cpu_detection() {
} // namespace } // namespace
extern "C" {
jint InitFFmpegOnLoad(JavaVM* vm) {
av_jni_set_java_vm(vm, nullptr);
return 0;
}
}
extern "C" { extern "C" {
void Java_org_yuzu_yuzu_1emu_NativeLibrary_surfaceChanged(JNIEnv* env, jobject instance, void Java_org_yuzu_yuzu_1emu_NativeLibrary_surfaceChanged(JNIEnv* env, jobject instance,
+5 -3
View File
@@ -22,9 +22,11 @@ using namespace std::literals;
constexpr auto INCREMENT_TIME{5ms}; constexpr auto INCREMENT_TIME{5ms};
DeviceSession::DeviceSession(Core::System& system_) DeviceSession::DeviceSession(Core::System& system_)
: system{system_}, thread_event{Core::Timing::CreateEvent( : system{system_}
"AudioOutSampleTick", , thread_event{system_.CreateTimingEvent("AudioOutSampleTick", [this](s64 time, std::chrono::nanoseconds) {
[this](s64 time, std::chrono::nanoseconds) { return ThreadFunc(); })} {} return ThreadFunc();
})}
{}
DeviceSession::~DeviceSession() { DeviceSession::~DeviceSession() {
Finalize(); Finalize();
-3
View File
@@ -427,11 +427,8 @@ namespace Common::Android {
extern "C" { extern "C" {
#endif #endif
jint InitFFmpegOnLoad(JavaVM *vm);
jint JNI_OnLoad(JavaVM *vm, void *reserved) { jint JNI_OnLoad(JavaVM *vm, void *reserved) {
s_java_vm = vm; s_java_vm = vm;
InitFFmpegOnLoad(vm);
JNIEnv *env; JNIEnv *env;
if (vm->GetEnv(reinterpret_cast<void **>(&env), JNI_VERSION) != JNI_OK) if (vm->GetEnv(reinterpret_cast<void **>(&env), JNI_VERSION) != JNI_OK)
+1 -1
View File
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project // SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later // SPDX-License-Identifier: GPL-3.0-or-later
#pragma once #pragma once
+4
View File
@@ -969,4 +969,8 @@ void System::ApplySettings() {
} }
} }
std::shared_ptr<Core::Timing::EventType> System::CreateTimingEvent(std::string name, Core::Timing::TimedCallback&& callback) {
return std::make_shared<Core::Timing::EventType>(std::move(callback), std::move(name));
}
} // namespace Core } // namespace Core
+4
View File
@@ -19,6 +19,7 @@
#include "core/file_sys/vfs/vfs_types.h" #include "core/file_sys/vfs/vfs_types.h"
#include "core/hle/service/os/event.h" #include "core/hle/service/os/event.h"
#include "core/hle/service/kernel_helpers.h" #include "core/hle/service/kernel_helpers.h"
#include "core/core_timing.h"
namespace Core::Frontend { namespace Core::Frontend {
class EmuWindow; class EmuWindow;
@@ -438,6 +439,9 @@ public:
/// Applies any changes to settings to this core instance. /// Applies any changes to settings to this core instance.
void ApplySettings(); void ApplySettings();
std::shared_ptr<Core::Timing::EventType> CreateTimingEvent(std::string name, Core::Timing::TimedCallback&& callback);
private:
struct Impl; struct Impl;
std::unique_ptr<Impl> impl; std::unique_ptr<Impl> impl;
}; };
+14 -45
View File
@@ -23,10 +23,6 @@ namespace Core::Timing {
constexpr s64 MAX_SLICE_LENGTH = 10000; constexpr s64 MAX_SLICE_LENGTH = 10000;
std::shared_ptr<EventType> CreateEvent(std::string name, TimedCallback&& callback) {
return std::make_shared<EventType>(std::move(callback), std::move(name));
}
struct CoreTiming::Event { struct CoreTiming::Event {
s64 time; s64 time;
u64 fifo_order; u64 fifo_order;
@@ -36,11 +32,10 @@ struct CoreTiming::Event {
// Sort by time, unless the times are the same, in which case sort by // Sort by time, unless the times are the same, in which case sort by
// the order added to the queue // the order added to the queue
friend bool operator>(const Event& left, const Event& right) { friend bool operator>(const Event& left, const Event& right) noexcept {
return std::tie(left.time, left.fifo_order) > std::tie(right.time, right.fifo_order); return std::tie(left.time, left.fifo_order) > std::tie(right.time, right.fifo_order);
} }
friend bool operator<(const Event& left, const Event& right) noexcept {
friend bool operator<(const Event& left, const Event& right) {
return std::tie(left.time, left.fifo_order) < std::tie(right.time, right.fifo_order); return std::tie(left.time, left.fifo_order) < std::tie(right.time, right.fifo_order);
} }
}; };
@@ -60,8 +55,6 @@ void CoreTiming::Initialize(std::function<void()>&& on_thread_init_) {
Common::SetCurrentThreadName("HostTiming"); Common::SetCurrentThreadName("HostTiming");
Common::SetCurrentThreadPriority(Common::ThreadPriority::High); Common::SetCurrentThreadPriority(Common::ThreadPriority::High);
on_thread_init(); on_thread_init();
has_started = true;
// base frequency in MHz: 1ns (10^-9) = 1GHz (10^9) // base frequency in MHz: 1ns (10^-9) = 1GHz (10^9)
while (!stop_token.stop_requested()) { while (!stop_token.stop_requested()) {
while (!paused && !stop_token.stop_requested()) { while (!paused && !stop_token.stop_requested()) {
@@ -77,8 +70,8 @@ void CoreTiming::Initialize(std::function<void()>&& on_thread_init_) {
// continue. // continue.
wait_set = true; wait_set = true;
event.Wait(); event.Wait();
wait_set = false;
} }
wait_set = false;
} }
paused_set = true; paused_set = true;
pause_event.Wait(); pause_event.Wait();
@@ -144,39 +137,28 @@ void CoreTiming::ScheduleEvent(std::chrono::nanoseconds ns_into_future,
event.Set(); event.Set();
} }
void CoreTiming::ScheduleLoopingEvent(std::chrono::nanoseconds start_time, void CoreTiming::ScheduleLoopingEvent(std::chrono::nanoseconds start_time, std::chrono::nanoseconds resched_time, const std::shared_ptr<EventType>& event_type, bool absolute_time) {
std::chrono::nanoseconds resched_time,
const std::shared_ptr<EventType>& event_type,
bool absolute_time) {
{ {
std::scoped_lock scope{basic_lock}; std::scoped_lock scope{basic_lock};
const auto next_time{absolute_time ? start_time : GetGlobalTimeNs() + start_time}; const auto next_time{absolute_time ? start_time : GetGlobalTimeNs() + start_time};
auto h = event_queue.emplace(Event{next_time.count(), event_fifo_id++, event_type, resched_time.count()});
auto h{event_queue.emplace(
Event{next_time.count(), event_fifo_id++, event_type, resched_time.count()})};
(*h).handle = h; (*h).handle = h;
} }
event.Set(); event.Set();
} }
void CoreTiming::UnscheduleEvent(const std::shared_ptr<EventType>& event_type, void CoreTiming::UnscheduleEvent(const std::shared_ptr<EventType>& event_type, UnscheduleEventType type) {
UnscheduleEventType type) {
{ {
std::scoped_lock lk{basic_lock}; std::scoped_lock lk{basic_lock};
std::vector<heap_t::handle_type> to_remove; std::vector<heap_t::handle_type> to_remove;
for (auto itr = event_queue.begin(); itr != event_queue.end(); itr++) { for (auto it = event_queue.begin(); it != event_queue.end(); it++) {
const Event& e = *itr; auto const& e = *it;
if (e.type.lock().get() == event_type.get()) { if (e.type.lock().get() == event_type.get()) {
to_remove.push_back(itr->handle); to_remove.push_back(it->handle);
} }
} }
for (auto& h : to_remove)
for (auto& h : to_remove) {
event_queue.erase(h); event_queue.erase(h);
}
event_type->sequence_number++; event_type->sequence_number++;
} }
@@ -187,16 +169,15 @@ void CoreTiming::UnscheduleEvent(const std::shared_ptr<EventType>& event_type,
} }
static u64 GetNextTickCount(u64 next_ticks) { static u64 GetNextTickCount(u64 next_ticks) {
if (Settings::values.use_custom_cpu_ticks.GetValue()) { if (Settings::values.use_custom_cpu_ticks.GetValue())
return Settings::values.cpu_ticks.GetValue(); return Settings::values.cpu_ticks.GetValue();
}
return next_ticks; return next_ticks;
} }
void CoreTiming::AddTicks(u64 ticks_to_add) { void CoreTiming::AddTicks(u64 ticks_to_add) {
const u64 ticks = GetNextTickCount(ticks_to_add); const u64 ticks = GetNextTickCount(ticks_to_add);
cpu_ticks += ticks; cpu_ticks += ticks;
downcount -= static_cast<s64>(ticks); downcount -= s64(ticks);
} }
void CoreTiming::Idle() { void CoreTiming::Idle() {
@@ -270,19 +251,14 @@ std::optional<s64> CoreTiming::Advance() {
next_time = pause_end_time + next_schedule_time; next_time = pause_end_time + next_schedule_time;
} }
event_queue.update(evt.handle, Event{next_time, event_fifo_id++, evt.type, event_queue.update(evt.handle, Event{next_time, event_fifo_id++, evt.type, next_schedule_time, evt.handle});
next_schedule_time, evt.handle});
} }
} }
global_timer = GetGlobalTimeNs().count(); global_timer = GetGlobalTimeNs().count();
} }
if (!event_queue.empty()) { return event_queue.empty() ? std::optional<s64>{} : event_queue.top().time;
return event_queue.top().time;
} else {
return std::nullopt;
}
} }
void CoreTiming::Reset() { void CoreTiming::Reset() {
@@ -293,7 +269,6 @@ void CoreTiming::Reset() {
timer_thread.request_stop(); timer_thread.request_stop();
timer_thread.join(); timer_thread.join();
} }
has_started = false;
} }
/// @brief Returns current time in nanoseconds. /// @brief Returns current time in nanoseconds.
@@ -310,10 +285,4 @@ std::chrono::microseconds CoreTiming::GetGlobalTimeUs() const noexcept {
: std::chrono::microseconds{Common::WallClock::CPUTickToUS(cpu_ticks)}; : std::chrono::microseconds{Common::WallClock::CPUTickToUS(cpu_ticks)};
} }
#ifdef _WIN32
void CoreTiming::SetTimerResolutionNs(std::chrono::nanoseconds ns) {
timer_resolution_ns = ns.count();
}
#endif
} // namespace Core::Timing } // namespace Core::Timing
+11 -33
View File
@@ -29,8 +29,7 @@ using TimedCallback = std::function<std::optional<std::chrono::nanoseconds>(
/// Contains the characteristics of a particular event. /// Contains the characteristics of a particular event.
struct EventType { struct EventType {
explicit EventType(TimedCallback&& callback_, std::string&& name_) explicit EventType(TimedCallback&& callback_, std::string&& name_) : callback{std::move(callback_)}, name{std::move(name_)}, sequence_number{0} {}
: callback{std::move(callback_)}, name{std::move(name_)}, sequence_number{0} {}
/// The event's callback function. /// The event's callback function.
TimedCallback callback; TimedCallback callback;
@@ -90,11 +89,6 @@ public:
/// Checks if core timing is running. /// Checks if core timing is running.
bool IsRunning() const; bool IsRunning() const;
/// Checks if the timer thread has started.
bool HasStarted() const {
return has_started;
}
/// Checks if there are any pending time events. /// Checks if there are any pending time events.
bool HasPendingEvents() const; bool HasPendingEvents() const;
@@ -134,45 +128,29 @@ public:
/// Checks for events manually and returns time in nanoseconds for next event, threadsafe. /// Checks for events manually and returns time in nanoseconds for next event, threadsafe.
std::optional<s64> Advance(); std::optional<s64> Advance();
#ifdef _WIN32
void SetTimerResolutionNs(std::chrono::nanoseconds ns);
#endif
struct Event; struct Event;
void Reset(); void Reset();
using heap_t = boost::heap::fibonacci_heap<CoreTiming::Event, boost::heap::compare<std::greater<>>>; using heap_t = boost::heap::fibonacci_heap<CoreTiming::Event, boost::heap::compare<std::greater<>>>;
Common::Event event{};
Common::Event pause_event{};
alignas(64) mutable std::mutex basic_lock;
alignas(64) std::mutex advance_lock;
alignas(64) std::atomic<bool> paused{};
alignas(64) std::atomic<bool> paused_set{};
alignas(64) std::atomic<bool> wait_set{};
std::function<void()> on_thread_init{};
heap_t event_queue; heap_t event_queue;
std::jthread timer_thread;
s64 global_timer = 0; s64 global_timer = 0;
#ifdef _WIN32
s64 timer_resolution_ns;
#endif
u64 event_fifo_id = 0; u64 event_fifo_id = 0;
s64 pause_end_time{}; s64 pause_end_time{};
/// Cycle timing /// Cycle timing
u64 cpu_ticks{}; u64 cpu_ticks{};
s64 downcount{}; s64 downcount{};
Common::Event event{};
Common::Event pause_event{};
std::function<void()> on_thread_init{};
std::jthread timer_thread;
mutable std::mutex basic_lock;
std::mutex advance_lock;
std::atomic<bool> paused{};
std::atomic<bool> paused_set{};
std::atomic<bool> wait_set{};
std::atomic<bool> has_started{};
bool is_multicore{}; bool is_multicore{};
}; };
/// Creates a core timing event with the given name and callback.
///
/// @param name The name of the core timing event to create.
/// @param callback The callback to execute for the event.
///
/// @returns An EventType instance representing the created event.
///
std::shared_ptr<EventType> CreateEvent(std::string name, TimedCallback&& callback);
} // namespace Core::Timing } // namespace Core::Timing
+5 -6
View File
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project // SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later // SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2022 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2022 yuzu Emulator Project
@@ -13,11 +13,10 @@ namespace Kernel {
void KHardwareTimer::Initialize() { void KHardwareTimer::Initialize() {
// Create the timing callback to register with CoreTiming. // Create the timing callback to register with CoreTiming.
m_event_type = Core::Timing::CreateEvent("KHardwareTimer::Callback", m_event_type = m_kernel.System().CreateTimingEvent("KHardwareTimer::Callback", [this](s64, std::chrono::nanoseconds) {
[this](s64, std::chrono::nanoseconds) { this->DoTask();
this->DoTask(); return std::nullopt;
return std::nullopt; });
});
} }
void KHardwareTimer::Finalize() { void KHardwareTimer::Finalize() {
+1 -1
View File
@@ -255,7 +255,7 @@ struct KernelCore::Impl {
} }
void InitializePreemption(KernelCore& kernel) { void InitializePreemption(KernelCore& kernel) {
preemption_event = Core::Timing::CreateEvent("PreemptionCallback", [this, &kernel](s64 time, std::chrono::nanoseconds) -> std::optional<std::chrono::nanoseconds> { preemption_event = system.CreateTimingEvent("PreemptionCallback", [this, &kernel](s64 time, std::chrono::nanoseconds) -> std::optional<std::chrono::nanoseconds> {
{ {
KScopedSchedulerLock lock(kernel); KScopedSchedulerLock lock(kernel);
global_scheduler_context->PreemptThreads(kernel); global_scheduler_context->PreemptThreads(kernel);
@@ -25,16 +25,11 @@ AlarmWorker::~AlarmWorker() {
void AlarmWorker::Initialize(std::shared_ptr<Service::PSC::Time::ServiceManager> time_m) { void AlarmWorker::Initialize(std::shared_ptr<Service::PSC::Time::ServiceManager> time_m) {
m_time_m = std::move(time_m); m_time_m = std::move(time_m);
m_timer_event = m_ctx.CreateEvent("Glue:AlarmWorker:TimerEvent"); m_timer_event = m_ctx.CreateEvent("Glue:AlarmWorker:TimerEvent");
m_timer_timing_event = Core::Timing::CreateEvent( m_timer_timing_event = m_system.CreateTimingEvent("Glue:AlarmWorker::AlarmTimer", [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
"Glue:AlarmWorker::AlarmTimer", m_timer_event->Signal(m_system.Kernel());
[this](s64 time, return std::nullopt;
std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> { });
m_timer_event->Signal(m_system.Kernel());
return std::nullopt;
});
AttachToClosestAlarmEvent(); AttachToClosestAlarmEvent();
} }
+10 -19
View File
@@ -22,28 +22,19 @@ namespace Service::Glue::Time {
TimeWorker::TimeWorker(Core::System& system, StandardSteadyClockResource& steady_clock_resource, TimeWorker::TimeWorker(Core::System& system, StandardSteadyClockResource& steady_clock_resource,
FileTimestampWorker& file_timestamp_worker) FileTimestampWorker& file_timestamp_worker)
: m_system{system}, m_ctx{m_system, "Glue:TimeWorker"}, m_event{m_ctx.CreateEvent( : m_system{system}, m_ctx{m_system, "Glue:TimeWorker"}, m_event{m_ctx.CreateEvent("Glue:TimeWorker:Event")},
"Glue:TimeWorker:Event")},
m_steady_clock_resource{steady_clock_resource}, m_steady_clock_resource{steady_clock_resource},
m_file_timestamp_worker{file_timestamp_worker}, m_timer_steady_clock{m_ctx.CreateEvent( m_file_timestamp_worker{file_timestamp_worker}, m_timer_steady_clock{m_ctx.CreateEvent("Glue:TimeWorker:SteadyClockTimerEvent")},
"Glue:TimeWorker:SteadyClockTimerEvent")},
m_timer_file_system{m_ctx.CreateEvent("Glue:TimeWorker:FileTimeTimerEvent")}, m_timer_file_system{m_ctx.CreateEvent("Glue:TimeWorker:FileTimeTimerEvent")},
m_alarm_worker{m_system, m_steady_clock_resource}, m_pm_state_change_handler{m_alarm_worker} { m_alarm_worker{m_system, m_steady_clock_resource}, m_pm_state_change_handler{m_alarm_worker} {
m_timer_steady_clock_timing_event = Core::Timing::CreateEvent( m_timer_steady_clock_timing_event = m_system.CreateTimingEvent("Time::SteadyClockEvent", [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
"Time::SteadyClockEvent", m_timer_steady_clock->Signal(m_system.Kernel());
[this](s64 time, return std::nullopt;
std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> { });
m_timer_steady_clock->Signal(m_system.Kernel()); m_timer_file_system_timing_event = m_system.CreateTimingEvent("Time::SteadyClockEvent", [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
return std::nullopt; m_timer_file_system->Signal(m_system.Kernel());
}); return std::nullopt;
});
m_timer_file_system_timing_event = Core::Timing::CreateEvent(
"Time::SteadyClockEvent",
[this](s64 time,
std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
m_timer_file_system->Signal(m_system.Kernel());
return std::nullopt;
});
} }
TimeWorker::~TimeWorker() { TimeWorker::~TimeWorker() {
+6 -11
View File
@@ -51,17 +51,12 @@ Hidbus::Hidbus(Core::System& system_)
RegisterHandlers(functions); RegisterHandlers(functions);
// Register update callbacks // Register update callbacks
hidbus_update_event = Core::Timing::CreateEvent( hidbus_update_event = system_.CreateTimingEvent("Hidbus::UpdateCallback", [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
"Hidbus::UpdateCallback", const auto guard = LockService();
[this](s64 time, UpdateHidbus(ns_late);
std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> { return std::nullopt;
const auto guard = LockService(); });
UpdateHidbus(ns_late); system_.CoreTiming().ScheduleLoopingEvent(hidbus_update_ns, hidbus_update_ns, hidbus_update_event);
return std::nullopt;
});
system_.CoreTiming().ScheduleLoopingEvent(hidbus_update_ns, hidbus_update_ns,
hidbus_update_event);
} }
Hidbus::~Hidbus() { Hidbus::~Hidbus() {
+8 -16
View File
@@ -23,25 +23,17 @@ Conductor::Conductor(Core::System& system, Container& container, DisplayList& di
}); });
if (system.IsMulticore()) { if (system.IsMulticore()) {
m_event = Core::Timing::CreateEvent( m_event = system.CreateTimingEvent("ScreenComposition", [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
"ScreenComposition", m_signal.Set();
[this](s64 time, return std::chrono::nanoseconds(this->GetNextTicks());
std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> { });
m_signal.Set();
return std::chrono::nanoseconds(this->GetNextTicks());
});
system.CoreTiming().ScheduleLoopingEvent(FrameNs, FrameNs, m_event); system.CoreTiming().ScheduleLoopingEvent(FrameNs, FrameNs, m_event);
m_thread = std::jthread([this](std::stop_token token) { this->VsyncThread(token); }); m_thread = std::jthread([this](std::stop_token token) { this->VsyncThread(token); });
} else { } else {
m_event = Core::Timing::CreateEvent( m_event = system.CreateTimingEvent("ScreenComposition", [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
"ScreenComposition", this->ProcessVsync();
[this](s64 time, return std::chrono::nanoseconds(this->GetNextTicks());
std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> { });
this->ProcessVsync();
return std::chrono::nanoseconds(this->GetNextTicks());
});
system.CoreTiming().ScheduleLoopingEvent(FrameNs, FrameNs, m_event); system.CoreTiming().ScheduleLoopingEvent(FrameNs, FrameNs, m_event);
} }
} }
+4 -6
View File
@@ -231,12 +231,10 @@ CheatEngine::~CheatEngine() {
} }
void CheatEngine::Initialize() { void CheatEngine::Initialize() {
event = Core::Timing::CreateEvent( event = system.CreateTimingEvent("CheatEngine::FrameCallback::" + Common::HexToString(metadata.main_nso_build_id), [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
"CheatEngine::FrameCallback::" + Common::HexToString(metadata.main_nso_build_id), FrameCallback(ns_late);
[this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> { return std::nullopt;
FrameCallback(ns_late); });
return std::nullopt;
});
core_timing.ScheduleLoopingEvent(CHEAT_ENGINE_NS, CHEAT_ENGINE_NS, event); core_timing.ScheduleLoopingEvent(CHEAT_ENGINE_NS, CHEAT_ENGINE_NS, event);
metadata.process_id = system.ApplicationProcess()->GetProcessId(); metadata.process_id = system.ApplicationProcess()->GetProcessId();
+5 -7
View File
@@ -52,14 +52,12 @@ void MemoryWriteWidth(Core::Memory::Memory& memory, u32 width, VAddr addr, u64 v
} // Anonymous namespace } // Anonymous namespace
Freezer::Freezer(Core::Timing::CoreTiming& core_timing_, Core::Memory::Memory& memory_) Freezer::Freezer(Core::System& system_, Core::Timing::CoreTiming& core_timing_, Core::Memory::Memory& memory_)
: core_timing{core_timing_}, memory{memory_} { : core_timing{core_timing_}, memory{memory_} {
event = Core::Timing::CreateEvent("MemoryFreezer::FrameCallback", event = system_.CreateTimingEvent("MemoryFreezer::FrameCallback", [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
[this](s64 time, std::chrono::nanoseconds ns_late) FrameCallback(ns_late);
-> std::optional<std::chrono::nanoseconds> { return std::nullopt;
FrameCallback(ns_late); });
return std::nullopt;
});
core_timing.ScheduleEvent(memory_freezer_ns, event); core_timing.ScheduleEvent(memory_freezer_ns, event);
} }
+10 -5
View File
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2019 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2019 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
@@ -11,14 +14,16 @@
#include <vector> #include <vector>
#include "common/common_types.h" #include "common/common_types.h"
namespace Core::Timing { namespace Core {
class System;
namespace Timing {
class CoreTiming; class CoreTiming;
struct EventType; struct EventType;
} // namespace Core::Timing } // namespace Core::Timing
namespace Memory {
namespace Core::Memory {
class Memory; class Memory;
} } //namespace Core::Memory
} //namespace Core
namespace Tools { namespace Tools {
@@ -38,7 +43,7 @@ public:
u64 value; u64 value;
}; };
explicit Freezer(Core::Timing::CoreTiming& core_timing_, Core::Memory::Memory& memory_); explicit Freezer(Core::System& system_, Core::Timing::CoreTiming& core_timing_, Core::Memory::Memory& memory_);
~Freezer(); ~Freezer();
// Enables or disables the entire memory freezer. // Enables or disables the entire memory freezer.
+20 -34
View File
@@ -56,33 +56,22 @@ ResourceManager::ResourceManager(Core::System& system_,
applet_resource = std::make_shared<AppletResource>(system); applet_resource = std::make_shared<AppletResource>(system);
// Register update callbacks // Register update callbacks
npad_update_event = Core::Timing::CreateEvent("HID::UpdatePadCallback", npad_update_event = system.CreateTimingEvent("HID::UpdatePadCallback", [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
[this](s64 time, std::chrono::nanoseconds ns_late) UpdateNpad(ns_late);
-> std::optional<std::chrono::nanoseconds> { return std::nullopt;
UpdateNpad(ns_late); });
return std::nullopt; default_update_event = system.CreateTimingEvent("HID::UpdateDefaultCallback", [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
}); UpdateControllers(ns_late);
default_update_event = Core::Timing::CreateEvent( return std::nullopt;
"HID::UpdateDefaultCallback", });
[this](s64 time, mouse_keyboard_update_event = system.CreateTimingEvent("HID::UpdateMouseKeyboardCallback", [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> { UpdateMouseKeyboard(ns_late);
UpdateControllers(ns_late); return std::nullopt;
return std::nullopt; });
}); motion_update_event = system.CreateTimingEvent("HID::UpdateMotionCallback", [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
mouse_keyboard_update_event = Core::Timing::CreateEvent( UpdateMotion(ns_late);
"HID::UpdateMouseKeyboardCallback", return std::nullopt;
[this](s64 time, });
std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
UpdateMouseKeyboard(ns_late);
return std::nullopt;
});
motion_update_event = Core::Timing::CreateEvent(
"HID::UpdateMotionCallback",
[this](s64 time,
std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
UpdateMotion(ns_late);
return std::nullopt;
});
} }
ResourceManager::~ResourceManager() { ResourceManager::~ResourceManager() {
@@ -267,13 +256,10 @@ void ResourceManager::InitializeTouchScreenSampler() {
touch_screen = std::make_shared<TouchScreen>(touch_resource); touch_screen = std::make_shared<TouchScreen>(touch_resource);
gesture = std::make_shared<Gesture>(touch_resource); gesture = std::make_shared<Gesture>(touch_resource);
touch_update_event = Core::Timing::CreateEvent( touch_update_event = system.CreateTimingEvent("HID::TouchUpdateCallback", [this](s64 time, std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> {
"HID::TouchUpdateCallback", touch_resource->OnTouchUpdate(time);
[this](s64 time, return std::nullopt;
std::chrono::nanoseconds ns_late) -> std::optional<std::chrono::nanoseconds> { });
touch_resource->OnTouchUpdate(time);
return std::nullopt;
});
touch_resource->SetTouchDriver(touch_driver); touch_resource->SetTouchDriver(touch_driver);
touch_resource->SetAppletResource(applet_resource, &shared_mutex); touch_resource->SetAppletResource(applet_resource, &shared_mutex);
+1 -6
View File
@@ -266,12 +266,7 @@ void Init(QWidget* root) {
Common::GetMemInfo().TotalPhysicalMemory / f64{1_GiB}); Common::GetMemInfo().TotalPhysicalMemory / f64{1_GiB});
LOG_INFO(Frontend, "Host Swap: {:.2f} GiB", Common::GetMemInfo().TotalSwapMemory / f64{1_GiB}); LOG_INFO(Frontend, "Host Swap: {:.2f} GiB", Common::GetMemInfo().TotalSwapMemory / f64{1_GiB});
#ifdef _WIN32 #ifdef _WIN32
LOG_INFO(Frontend, "Host Timer Resolution: {:.4f} ms", LOG_INFO(Frontend, "Host Timer Resolution: {:.4f} ms", std::chrono::duration_cast<std::chrono::duration<f64, std::milli>>(Common::Windows::SetCurrentTimerResolutionToMaximum()).count());
std::chrono::duration_cast<std::chrono::duration<f64, std::milli>>(
Common::Windows::SetCurrentTimerResolutionToMaximum())
.count());
QtCommon::system->CoreTiming().SetTimerResolutionNs(
Common::Windows::GetCurrentTimerResolution());
#endif #endif
// Remove cached contents generated during the previous session // Remove cached contents generated during the previous session
@@ -403,9 +403,6 @@ void SetupCapabilities(const Profile& profile, const Info& info, EmitContext& ct
if (info.uses_sampled_1d) { if (info.uses_sampled_1d) {
ctx.AddCapability(spv::Capability::Sampled1D); ctx.AddCapability(spv::Capability::Sampled1D);
} }
if (info.uses_image_1d) {
ctx.AddCapability(spv::Capability::Image1D);
}
if (info.uses_sparse_residency) { if (info.uses_sparse_residency) {
ctx.AddCapability(spv::Capability::SparseResidency); ctx.AddCapability(spv::Capability::SparseResidency);
} }
@@ -435,7 +432,7 @@ void SetupCapabilities(const Profile& profile, const Info& info, EmitContext& ct
} }
if ((info.uses_subgroup_vote || info.uses_subgroup_invocation_id || if ((info.uses_subgroup_vote || info.uses_subgroup_invocation_id ||
info.uses_subgroup_shuffles) && info.uses_subgroup_shuffles) &&
profile.support_vote && profile.SupportsSubgroupStage(ctx.stage)) { profile.support_vote) {
ctx.AddCapability(spv::Capability::GroupNonUniformBallot); ctx.AddCapability(spv::Capability::GroupNonUniformBallot);
ctx.AddCapability(spv::Capability::GroupNonUniformShuffle); ctx.AddCapability(spv::Capability::GroupNonUniformShuffle);
if (!profile.warp_size_potentially_larger_than_guest) { if (!profile.warp_size_potentially_larger_than_guest) {
@@ -202,8 +202,7 @@ void EmitGetIndirectBranchVariable(EmitContext&) {
} }
Id EmitGetCbufU8(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset) { Id EmitGetCbufU8(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset) {
if (ctx.profile.support_descriptor_aliasing && ctx.profile.support_int8 && if (ctx.profile.support_descriptor_aliasing && ctx.profile.support_int8) {
ctx.profile.support_uniform_and_storage_buffer_8bit) {
const Id load{GetCbuf(ctx, ctx.U8, &UniformDefinitions::U8, sizeof(u8), binding, offset, const Id load{GetCbuf(ctx, ctx.U8, &UniformDefinitions::U8, sizeof(u8), binding, offset,
ctx.load_const_func_u8)}; ctx.load_const_func_u8)};
return ctx.OpUConvert(ctx.U32[1], load); return ctx.OpUConvert(ctx.U32[1], load);
@@ -220,8 +219,7 @@ Id EmitGetCbufU8(EmitContext& ctx, const IR::Value& binding, const IR::Value& of
} }
Id EmitGetCbufS8(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset) { Id EmitGetCbufS8(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset) {
if (ctx.profile.support_descriptor_aliasing && ctx.profile.support_int8 && if (ctx.profile.support_descriptor_aliasing && ctx.profile.support_int8) {
ctx.profile.support_uniform_and_storage_buffer_8bit) {
const Id load{GetCbuf(ctx, ctx.S8, &UniformDefinitions::S8, sizeof(s8), binding, offset, const Id load{GetCbuf(ctx, ctx.S8, &UniformDefinitions::S8, sizeof(s8), binding, offset,
ctx.load_const_func_u8)}; ctx.load_const_func_u8)};
return ctx.OpSConvert(ctx.U32[1], load); return ctx.OpSConvert(ctx.U32[1], load);
@@ -238,8 +236,7 @@ Id EmitGetCbufS8(EmitContext& ctx, const IR::Value& binding, const IR::Value& of
} }
Id EmitGetCbufU16(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset) { Id EmitGetCbufU16(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset) {
if (ctx.profile.support_descriptor_aliasing && ctx.profile.support_int16 && if (ctx.profile.support_descriptor_aliasing && ctx.profile.support_int16) {
ctx.profile.support_uniform_and_storage_buffer_16bit) {
const Id load{GetCbuf(ctx, ctx.U16, &UniformDefinitions::U16, sizeof(u16), binding, offset, const Id load{GetCbuf(ctx, ctx.U16, &UniformDefinitions::U16, sizeof(u16), binding, offset,
ctx.load_const_func_u16)}; ctx.load_const_func_u16)};
return ctx.OpUConvert(ctx.U32[1], load); return ctx.OpUConvert(ctx.U32[1], load);
@@ -256,8 +253,7 @@ Id EmitGetCbufU16(EmitContext& ctx, const IR::Value& binding, const IR::Value& o
} }
Id EmitGetCbufS16(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset) { Id EmitGetCbufS16(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset) {
if (ctx.profile.support_descriptor_aliasing && ctx.profile.support_int16 && if (ctx.profile.support_descriptor_aliasing && ctx.profile.support_int16) {
ctx.profile.support_uniform_and_storage_buffer_16bit) {
const Id load{GetCbuf(ctx, ctx.S16, &UniformDefinitions::S16, sizeof(s16), binding, offset, const Id load{GetCbuf(ctx, ctx.S16, &UniformDefinitions::S16, sizeof(s16), binding, offset,
ctx.load_const_func_u16)}; ctx.load_const_func_u16)};
return ctx.OpSConvert(ctx.U32[1], load); return ctx.OpSConvert(ctx.U32[1], load);
@@ -260,13 +260,6 @@ bool IsTextureMsaa(EmitContext& ctx, const IR::TextureInstInfo& info) {
return ctx.textures.at(info.descriptor_index).is_multisample; return ctx.textures.at(info.descriptor_index).is_multisample;
} }
bool IsTextureInteger(EmitContext& ctx, const IR::TextureInstInfo& info) {
if (info.type == TextureType::Buffer) {
return false;
}
return ctx.textures.at(info.descriptor_index).is_integer;
}
Id Decorate(EmitContext& ctx, IR::Inst* inst, Id sample) { Id Decorate(EmitContext& ctx, IR::Inst* inst, Id sample) {
const auto info{inst->Flags<IR::TextureInstInfo>()}; const auto info{inst->Flags<IR::TextureInstInfo>()};
if (info.relaxed_precision != 0) { if (info.relaxed_precision != 0) {
@@ -487,14 +480,11 @@ Id EmitBoundImageWrite(EmitContext&) {
Id EmitImageSampleImplicitLod(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords, Id EmitImageSampleImplicitLod(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id bias_lc, const IR::Value& offset) { Id bias_lc, const IR::Value& offset) {
const auto info{inst->Flags<IR::TextureInstInfo>()}; const auto info{inst->Flags<IR::TextureInstInfo>()};
const bool is_integer{IsTextureInteger(ctx, info)};
const Id result_type{is_integer ? ctx.U32[4] : ctx.F32[4]};
Id color;
if (ctx.stage == Stage::Fragment) { if (ctx.stage == Stage::Fragment) {
const ImageOperands operands(ctx, info.has_bias != 0, false, info.has_lod_clamp != 0, const ImageOperands operands(ctx, info.has_bias != 0, false, info.has_lod_clamp != 0,
bias_lc, offset); bias_lc, offset);
color = Emit(&EmitContext::OpImageSparseSampleImplicitLod, return Emit(&EmitContext::OpImageSparseSampleImplicitLod,
&EmitContext::OpImageSampleImplicitLod, ctx, inst, result_type, &EmitContext::OpImageSampleImplicitLod, ctx, inst, ctx.F32[4],
Texture(ctx, info, index), coords, operands.MaskOptional(), operands.Span()); Texture(ctx, info, index), coords, operands.MaskOptional(), operands.Span());
} else { } else {
// We can't use implicit lods on non-fragment stages on SPIR-V. Maxwell hardware behaves as // We can't use implicit lods on non-fragment stages on SPIR-V. Maxwell hardware behaves as
@@ -502,29 +492,26 @@ Id EmitImageSampleImplicitLod(EmitContext& ctx, IR::Inst* inst, const IR::Value&
// derivatives // derivatives
const Id lod{ctx.Const(0.0f)}; const Id lod{ctx.Const(0.0f)};
const ImageOperands operands(ctx, false, true, info.has_lod_clamp != 0, lod, offset); const ImageOperands operands(ctx, false, true, info.has_lod_clamp != 0, lod, offset);
color = Emit(&EmitContext::OpImageSparseSampleExplicitLod, return Emit(&EmitContext::OpImageSparseSampleExplicitLod,
&EmitContext::OpImageSampleExplicitLod, ctx, inst, result_type, &EmitContext::OpImageSampleExplicitLod, ctx, inst, ctx.F32[4],
Texture(ctx, info, index), coords, operands.Mask(), operands.Span()); Texture(ctx, info, index), coords, operands.Mask(), operands.Span());
} }
return is_integer ? ctx.OpBitcast(ctx.F32[4], color) : color;
} }
Id EmitImageSampleExplicitLod(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords, Id EmitImageSampleExplicitLod(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id lod, const IR::Value& offset) { Id lod, const IR::Value& offset) {
const auto info{inst->Flags<IR::TextureInstInfo>()}; const auto info{inst->Flags<IR::TextureInstInfo>()};
const bool is_integer{IsTextureInteger(ctx, info)};
const Id result_type{is_integer ? ctx.U32[4] : ctx.F32[4]};
const ImageOperands operands(ctx, false, true, false, lod, offset); const ImageOperands operands(ctx, false, true, false, lod, offset);
Id result = Emit(&EmitContext::OpImageSparseSampleExplicitLod, Id result = Emit(&EmitContext::OpImageSparseSampleExplicitLod,
&EmitContext::OpImageSampleExplicitLod, ctx, inst, result_type, &EmitContext::OpImageSampleExplicitLod, ctx, inst, ctx.F32[4],
Texture(ctx, info, index), coords, operands.Mask(), operands.Span()); Texture(ctx, info, index), coords, operands.Mask(), operands.Span());
#ifdef __ANDROID__ #ifdef __ANDROID__
if (!is_integer && Settings::values.fix_bloom_effects.GetValue()) { if (Settings::values.fix_bloom_effects.GetValue()) {
result = ctx.OpVectorTimesScalar(ctx.F32[4], result, ctx.Const(0.98f)); result = ctx.OpVectorTimesScalar(ctx.F32[4], result, ctx.Const(0.98f));
} }
#endif #endif
return is_integer ? ctx.OpBitcast(ctx.F32[4], result) : result; return result;
} }
Id EmitImageSampleDrefImplicitLod(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id EmitImageSampleDrefImplicitLod(EmitContext& ctx, IR::Inst* inst, const IR::Value& index,
@@ -560,39 +547,30 @@ Id EmitImageSampleDrefExplicitLod(EmitContext& ctx, IR::Inst* inst, const IR::Va
Id EmitImageGather(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords, Id EmitImageGather(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
const IR::Value& offset, const IR::Value& offset2) { const IR::Value& offset, const IR::Value& offset2) {
const auto info{inst->Flags<IR::TextureInstInfo>()}; const auto info{inst->Flags<IR::TextureInstInfo>()};
const bool is_integer{IsTextureInteger(ctx, info)};
const Id result_type{is_integer ? ctx.U32[4] : ctx.F32[4]};
const ImageOperands operands(ctx, offset, offset2); const ImageOperands operands(ctx, offset, offset2);
if (ctx.profile.need_gather_subpixel_offset) { if (ctx.profile.need_gather_subpixel_offset) {
coords = ImageGatherSubpixelOffset(ctx, info, TextureImage(ctx, info, index), coords); coords = ImageGatherSubpixelOffset(ctx, info, TextureImage(ctx, info, index), coords);
} }
const Id color{Emit(&EmitContext::OpImageSparseGather, &EmitContext::OpImageGather, ctx, inst, return Emit(&EmitContext::OpImageSparseGather, &EmitContext::OpImageGather, ctx, inst,
result_type, Texture(ctx, info, index), coords, ctx.F32[4], Texture(ctx, info, index), coords, ctx.Const(info.gather_component),
ctx.Const(info.gather_component), operands.MaskOptional(), operands.MaskOptional(), operands.Span());
operands.Span())};
return is_integer ? ctx.OpBitcast(ctx.F32[4], color) : color;
} }
Id EmitImageGatherDref(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords, Id EmitImageGatherDref(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
const IR::Value& offset, const IR::Value& offset2, Id dref) { const IR::Value& offset, const IR::Value& offset2, Id dref) {
const auto info{inst->Flags<IR::TextureInstInfo>()}; const auto info{inst->Flags<IR::TextureInstInfo>()};
const bool is_integer{IsTextureInteger(ctx, info)};
const Id result_type{is_integer ? ctx.U32[4] : ctx.F32[4]};
const ImageOperands operands(ctx, offset, offset2); const ImageOperands operands(ctx, offset, offset2);
if (ctx.profile.need_gather_subpixel_offset) { if (ctx.profile.need_gather_subpixel_offset) {
coords = ImageGatherSubpixelOffset(ctx, info, TextureImage(ctx, info, index), coords); coords = ImageGatherSubpixelOffset(ctx, info, TextureImage(ctx, info, index), coords);
} }
const Id color{Emit(&EmitContext::OpImageSparseDrefGather, &EmitContext::OpImageDrefGather, return Emit(&EmitContext::OpImageSparseDrefGather, &EmitContext::OpImageDrefGather, ctx, inst,
ctx, inst, result_type, Texture(ctx, info, index), coords, dref, ctx.F32[4], Texture(ctx, info, index), coords, dref, operands.MaskOptional(),
operands.MaskOptional(), operands.Span())}; operands.Span());
return is_integer ? ctx.OpBitcast(ctx.F32[4], color) : color;
} }
Id EmitImageFetch(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords, Id offset, Id EmitImageFetch(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords, Id offset,
Id lod, Id ms) { Id lod, Id ms) {
const auto info{inst->Flags<IR::TextureInstInfo>()}; const auto info{inst->Flags<IR::TextureInstInfo>()};
const bool is_integer{IsTextureInteger(ctx, info)};
const Id result_type{is_integer ? ctx.U32[4] : ctx.F32[4]};
AddOffsetToCoordinates(ctx, info, coords, offset); AddOffsetToCoordinates(ctx, info, coords, offset);
if (info.type == TextureType::Buffer) { if (info.type == TextureType::Buffer) {
lod = Id{}; lod = Id{};
@@ -602,10 +580,8 @@ Id EmitImageFetch(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id c
lod = Id{}; lod = Id{};
} }
const ImageOperands operands(lod, ms); const ImageOperands operands(lod, ms);
const Id color{Emit(&EmitContext::OpImageSparseFetch, &EmitContext::OpImageFetch, ctx, inst, return Emit(&EmitContext::OpImageSparseFetch, &EmitContext::OpImageFetch, ctx, inst, ctx.F32[4],
result_type, TextureImage(ctx, info, index), coords, TextureImage(ctx, info, index), coords, operands.MaskOptional(), operands.Span());
operands.MaskOptional(), operands.Span())};
return is_integer ? ctx.OpBitcast(ctx.F32[4], color) : color;
} }
Id EmitImageQueryDimensions(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id lod, Id EmitImageQueryDimensions(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id lod,
@@ -650,17 +626,14 @@ Id EmitImageQueryLod(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, I
Id EmitImageGradient(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords, Id EmitImageGradient(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id derivatives, const IR::Value& offset, Id lod_clamp) { Id derivatives, const IR::Value& offset, Id lod_clamp) {
const auto info{inst->Flags<IR::TextureInstInfo>()}; const auto info{inst->Flags<IR::TextureInstInfo>()};
const bool is_integer{IsTextureInteger(ctx, info)};
const Id result_type{is_integer ? ctx.U32[4] : ctx.F32[4]};
const auto operands = info.num_derivatives == 3 const auto operands = info.num_derivatives == 3
? ImageOperands(ctx, info.has_lod_clamp != 0, derivatives, ? ImageOperands(ctx, info.has_lod_clamp != 0, derivatives,
ctx.Def(offset), {}, lod_clamp) ctx.Def(offset), {}, lod_clamp)
: ImageOperands(ctx, info.has_lod_clamp != 0, derivatives, : ImageOperands(ctx, info.has_lod_clamp != 0, derivatives,
info.num_derivatives, offset, lod_clamp); info.num_derivatives, offset, lod_clamp);
const Id color{Emit(&EmitContext::OpImageSparseSampleExplicitLod, return Emit(&EmitContext::OpImageSparseSampleExplicitLod,
&EmitContext::OpImageSampleExplicitLod, ctx, inst, result_type, &EmitContext::OpImageSampleExplicitLod, ctx, inst, ctx.F32[4],
Texture(ctx, info, index), coords, operands.Mask(), operands.Span())}; Texture(ctx, info, index), coords, operands.Mask(), operands.Span());
return is_integer ? ctx.OpBitcast(ctx.F32[4], color) : color;
} }
Id EmitImageRead(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords) { Id EmitImageRead(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords) {
@@ -1,6 +1,3 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
@@ -13,14 +10,7 @@ Id SubgroupScope(EmitContext& ctx) {
return ctx.Const(static_cast<u32>(spv::Scope::Subgroup)); return ctx.Const(static_cast<u32>(spv::Scope::Subgroup));
} }
bool StageSupportsSubgroups(EmitContext& ctx) {
return ctx.profile.SupportsSubgroupStage(ctx.stage);
}
Id GetThreadId(EmitContext& ctx) { Id GetThreadId(EmitContext& ctx) {
if (!StageSupportsSubgroups(ctx)) {
return ctx.u32_zero_value;
}
return ctx.OpLoad(ctx.U32[1], ctx.subgroup_local_invocation_id); return ctx.OpLoad(ctx.U32[1], ctx.subgroup_local_invocation_id);
} }
@@ -78,9 +68,6 @@ Id GetMaxThreadId(EmitContext& ctx, Id thread_id, Id clamp, Id segmentation_mask
} }
Id SelectValue(EmitContext& ctx, Id in_range, Id value, Id src_thread_id) { Id SelectValue(EmitContext& ctx, Id in_range, Id value, Id src_thread_id) {
if (!StageSupportsSubgroups(ctx)) {
return value;
}
return ctx.OpSelect( return ctx.OpSelect(
ctx.U32[1], in_range, ctx.U32[1], in_range,
ctx.OpGroupNonUniformShuffle(ctx.U32[1], SubgroupScope(ctx), value, src_thread_id), value); ctx.OpGroupNonUniformShuffle(ctx.U32[1], SubgroupScope(ctx), value, src_thread_id), value);
@@ -102,9 +89,6 @@ Id EmitLaneId(EmitContext& ctx) {
} }
Id EmitVoteAll(EmitContext& ctx, Id pred) { Id EmitVoteAll(EmitContext& ctx, Id pred) {
if (!StageSupportsSubgroups(ctx)) {
return pred;
}
if (!ctx.profile.warp_size_potentially_larger_than_guest) { if (!ctx.profile.warp_size_potentially_larger_than_guest) {
return ctx.OpGroupNonUniformAll(ctx.U1, SubgroupScope(ctx), pred); return ctx.OpGroupNonUniformAll(ctx.U1, SubgroupScope(ctx), pred);
} }
@@ -118,9 +102,6 @@ Id EmitVoteAll(EmitContext& ctx, Id pred) {
} }
Id EmitVoteAny(EmitContext& ctx, Id pred) { Id EmitVoteAny(EmitContext& ctx, Id pred) {
if (!StageSupportsSubgroups(ctx)) {
return pred;
}
if (!ctx.profile.warp_size_potentially_larger_than_guest) { if (!ctx.profile.warp_size_potentially_larger_than_guest) {
return ctx.OpGroupNonUniformAny(ctx.U1, SubgroupScope(ctx), pred); return ctx.OpGroupNonUniformAny(ctx.U1, SubgroupScope(ctx), pred);
} }
@@ -134,9 +115,6 @@ Id EmitVoteAny(EmitContext& ctx, Id pred) {
} }
Id EmitVoteEqual(EmitContext& ctx, Id pred) { Id EmitVoteEqual(EmitContext& ctx, Id pred) {
if (!StageSupportsSubgroups(ctx)) {
return ctx.true_value;
}
if (!ctx.profile.warp_size_potentially_larger_than_guest) { if (!ctx.profile.warp_size_potentially_larger_than_guest) {
return ctx.OpGroupNonUniformAllEqual(ctx.U1, SubgroupScope(ctx), pred); return ctx.OpGroupNonUniformAllEqual(ctx.U1, SubgroupScope(ctx), pred);
} }
@@ -151,9 +129,6 @@ Id EmitVoteEqual(EmitContext& ctx, Id pred) {
} }
Id EmitSubgroupBallot(EmitContext& ctx, Id pred) { Id EmitSubgroupBallot(EmitContext& ctx, Id pred) {
if (!StageSupportsSubgroups(ctx)) {
return ctx.OpSelect(ctx.U32[1], pred, ctx.Const(1u), ctx.u32_zero_value);
}
const Id ballot{ctx.OpGroupNonUniformBallot(ctx.U32[4], SubgroupScope(ctx), pred)}; const Id ballot{ctx.OpGroupNonUniformBallot(ctx.U32[4], SubgroupScope(ctx), pred)};
if (!ctx.profile.warp_size_potentially_larger_than_guest) { if (!ctx.profile.warp_size_potentially_larger_than_guest) {
return ctx.OpCompositeExtract(ctx.U32[1], ballot, 0U); return ctx.OpCompositeExtract(ctx.U32[1], ballot, 0U);
@@ -162,37 +137,22 @@ Id EmitSubgroupBallot(EmitContext& ctx, Id pred) {
} }
Id EmitSubgroupEqMask(EmitContext& ctx) { Id EmitSubgroupEqMask(EmitContext& ctx) {
if (!StageSupportsSubgroups(ctx)) {
return ctx.Const(1u);
}
return LoadMask(ctx, ctx.subgroup_mask_eq); return LoadMask(ctx, ctx.subgroup_mask_eq);
} }
Id EmitSubgroupLtMask(EmitContext& ctx) { Id EmitSubgroupLtMask(EmitContext& ctx) {
if (!StageSupportsSubgroups(ctx)) {
return ctx.u32_zero_value;
}
return LoadMask(ctx, ctx.subgroup_mask_lt); return LoadMask(ctx, ctx.subgroup_mask_lt);
} }
Id EmitSubgroupLeMask(EmitContext& ctx) { Id EmitSubgroupLeMask(EmitContext& ctx) {
if (!StageSupportsSubgroups(ctx)) {
return ctx.Const(1u);
}
return LoadMask(ctx, ctx.subgroup_mask_le); return LoadMask(ctx, ctx.subgroup_mask_le);
} }
Id EmitSubgroupGtMask(EmitContext& ctx) { Id EmitSubgroupGtMask(EmitContext& ctx) {
if (!StageSupportsSubgroups(ctx)) {
return ctx.u32_zero_value;
}
return LoadMask(ctx, ctx.subgroup_mask_gt); return LoadMask(ctx, ctx.subgroup_mask_gt);
} }
Id EmitSubgroupGeMask(EmitContext& ctx) { Id EmitSubgroupGeMask(EmitContext& ctx) {
if (!StageSupportsSubgroups(ctx)) {
return ctx.Const(1u);
}
return LoadMask(ctx, ctx.subgroup_mask_ge); return LoadMask(ctx, ctx.subgroup_mask_ge);
} }
@@ -262,7 +222,7 @@ Id EmitShuffleButterfly(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id
Id EmitFSwizzleAdd(EmitContext& ctx, Id op_a, Id op_b, Id swizzle) { Id EmitFSwizzleAdd(EmitContext& ctx, Id op_a, Id op_b, Id swizzle) {
const Id three{ctx.Const(3U)}; const Id three{ctx.Const(3U)};
Id mask{GetThreadId(ctx)}; Id mask{ctx.OpLoad(ctx.U32[1], ctx.subgroup_local_invocation_id)};
mask = ctx.OpBitwiseAnd(ctx.U32[1], mask, three); mask = ctx.OpBitwiseAnd(ctx.U32[1], mask, three);
mask = ctx.OpShiftLeftLogical(ctx.U32[1], mask, ctx.Const(1U)); mask = ctx.OpShiftLeftLogical(ctx.U32[1], mask, ctx.Const(1U));
mask = ctx.OpShiftRightLogical(ctx.U32[1], swizzle, mask); mask = ctx.OpShiftRightLogical(ctx.U32[1], swizzle, mask);
@@ -30,7 +30,7 @@ enum class Operation {
Id ImageType(EmitContext& ctx, const TextureDescriptor& desc) { Id ImageType(EmitContext& ctx, const TextureDescriptor& desc) {
const spv::ImageFormat format{spv::ImageFormat::Unknown}; const spv::ImageFormat format{spv::ImageFormat::Unknown};
const Id type{desc.is_integer ? ctx.U32[1] : ctx.F32[1]}; const Id type{ctx.F32[1]};
const bool depth{desc.is_depth}; const bool depth{desc.is_depth};
const bool ms{desc.is_multisample}; const bool ms{desc.is_multisample};
switch (desc.type) { switch (desc.type) {
@@ -1126,7 +1126,7 @@ void EmitContext::DefineConstantBuffers(const Info& info, u32& binding) {
} }
IR::Type types{info.used_constant_buffer_types | info.used_indirect_cbuf_types}; IR::Type types{info.used_constant_buffer_types | info.used_indirect_cbuf_types};
if (True(types & IR::Type::U8)) { if (True(types & IR::Type::U8)) {
if (profile.support_int8 && profile.support_uniform_and_storage_buffer_8bit) { if (profile.support_int8) {
DefineConstBuffers(*this, info, &UniformDefinitions::U8, binding, U8, 'u', sizeof(u8)); DefineConstBuffers(*this, info, &UniformDefinitions::U8, binding, U8, 'u', sizeof(u8));
DefineConstBuffers(*this, info, &UniformDefinitions::S8, binding, S8, 's', sizeof(s8)); DefineConstBuffers(*this, info, &UniformDefinitions::S8, binding, S8, 's', sizeof(s8));
} else { } else {
@@ -1134,7 +1134,7 @@ void EmitContext::DefineConstantBuffers(const Info& info, u32& binding) {
} }
} }
if (True(types & IR::Type::U16)) { if (True(types & IR::Type::U16)) {
if (profile.support_int16 && profile.support_uniform_and_storage_buffer_16bit) { if (profile.support_int16) {
DefineConstBuffers(*this, info, &UniformDefinitions::U16, binding, U16, 'u', DefineConstBuffers(*this, info, &UniformDefinitions::U16, binding, U16, 'u',
sizeof(u16)); sizeof(u16));
DefineConstBuffers(*this, info, &UniformDefinitions::S16, binding, S16, 's', DefineConstBuffers(*this, info, &UniformDefinitions::S16, binding, S16, 's',
@@ -1196,18 +1196,10 @@ void EmitContext::DefineConstantBufferIndirectFunctions(const Info& info) {
IR::Type types{info.used_indirect_cbuf_types}; IR::Type types{info.used_indirect_cbuf_types};
bool supports_aliasing = profile.support_descriptor_aliasing; bool supports_aliasing = profile.support_descriptor_aliasing;
if (supports_aliasing && True(types & IR::Type::U8)) { if (supports_aliasing && True(types & IR::Type::U8)) {
if (profile.support_int8 && profile.support_uniform_and_storage_buffer_8bit) { load_const_func_u8 = make_accessor(U8, &UniformDefinitions::U8);
load_const_func_u8 = make_accessor(U8, &UniformDefinitions::U8);
} else {
types |= IR::Type::U32;
}
} }
if (supports_aliasing && True(types & IR::Type::U16)) { if (supports_aliasing && True(types & IR::Type::U16)) {
if (profile.support_int16 && profile.support_uniform_and_storage_buffer_16bit) { load_const_func_u16 = make_accessor(U16, &UniformDefinitions::U16);
load_const_func_u16 = make_accessor(U16, &UniformDefinitions::U16);
} else {
types |= IR::Type::U32;
}
} }
if (supports_aliasing && True(types & IR::Type::F32)) { if (supports_aliasing && True(types & IR::Type::F32)) {
load_const_func_f32 = make_accessor(F32[1], &UniformDefinitions::F32); load_const_func_f32 = make_accessor(F32[1], &UniformDefinitions::F32);
@@ -1383,7 +1375,6 @@ void EmitContext::DefineTextures(const Info& info, u32& binding, u32& scaling_in
.image_type = image_type, .image_type = image_type,
.count = desc.count, .count = desc.count,
.is_multisample = desc.is_multisample, .is_multisample = desc.is_multisample,
.is_integer = desc.is_integer,
}); });
if (profile.supported_spirv >= 0x00010400) { if (profile.supported_spirv >= 0x00010400) {
interfaces.push_back(id); interfaces.push_back(id);
@@ -1447,7 +1438,7 @@ void EmitContext::DefineInputs(const IR::Program& program) {
if (info.uses_is_helper_invocation) { if (info.uses_is_helper_invocation) {
is_helper_invocation = DefineInput(*this, U1, false, spv::BuiltIn::HelperInvocation); is_helper_invocation = DefineInput(*this, U1, false, spv::BuiltIn::HelperInvocation);
} }
if (info.uses_subgroup_mask && profile.SupportsSubgroupStage(stage)) { if (info.uses_subgroup_mask) {
subgroup_mask_eq = DefineInput(*this, U32[4], false, spv::BuiltIn::SubgroupEqMaskKHR); subgroup_mask_eq = DefineInput(*this, U32[4], false, spv::BuiltIn::SubgroupEqMaskKHR);
subgroup_mask_lt = DefineInput(*this, U32[4], false, spv::BuiltIn::SubgroupLtMaskKHR); subgroup_mask_lt = DefineInput(*this, U32[4], false, spv::BuiltIn::SubgroupLtMaskKHR);
subgroup_mask_le = DefineInput(*this, U32[4], false, spv::BuiltIn::SubgroupLeMaskKHR); subgroup_mask_le = DefineInput(*this, U32[4], false, spv::BuiltIn::SubgroupLeMaskKHR);
@@ -1461,10 +1452,9 @@ void EmitContext::DefineInputs(const IR::Program& program) {
Decorate(subgroup_mask_ge, spv::Decoration::Flat); Decorate(subgroup_mask_ge, spv::Decoration::Flat);
} }
} }
if ((info.uses_fswzadd || info.uses_subgroup_invocation_id || info.uses_subgroup_shuffles || if (info.uses_fswzadd || info.uses_subgroup_invocation_id || info.uses_subgroup_shuffles ||
(profile.warp_size_potentially_larger_than_guest && (profile.warp_size_potentially_larger_than_guest &&
(info.uses_subgroup_vote || info.uses_subgroup_mask))) && (info.uses_subgroup_vote || info.uses_subgroup_mask))) {
profile.SupportsSubgroupStage(stage)) {
AddCapability(spv::Capability::GroupNonUniform); AddCapability(spv::Capability::GroupNonUniform);
subgroup_local_invocation_id = subgroup_local_invocation_id =
DefineInput(*this, U32[1], false, spv::BuiltIn::SubgroupLocalInvocationId); DefineInput(*this, U32[1], false, spv::BuiltIn::SubgroupLocalInvocationId);
@@ -42,7 +42,6 @@ struct TextureDefinition {
Id image_type; Id image_type;
u32 count; u32 count;
bool is_multisample; bool is_multisample;
bool is_integer;
}; };
struct TextureBufferDefinition { struct TextureBufferDefinition {
@@ -1,6 +1,3 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
@@ -46,10 +43,6 @@ union TextureInstInfo {
BitField<25, 2, u32> num_derivatives; BitField<25, 2, u32> num_derivatives;
BitField<27, 3, ImageFormat> image_format; BitField<27, 3, ImageFormat> image_format;
BitField<30, 1, u32> ndv_is_active; BitField<30, 1, u32> ndv_is_active;
/// Only meaningful for the sampled-texture family (ImageSampleImplicitLod, ImageFetch,
/// ImageGather, etc.); unused/zero for storage-image opcodes, which carry their own
/// is_integer via ImageDescriptor/ImageBufferDescriptor instead.
BitField<31, 1, u32> is_integer;
}; };
static_assert(sizeof(TextureInstInfo) <= sizeof(u32)); static_assert(sizeof(TextureInstInfo) <= sizeof(u32));
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project // SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later // SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
@@ -569,8 +569,6 @@ void VisitUsages(Info& info, IR::Inst& inst) {
case IR::Opcode::ImageRead: { case IR::Opcode::ImageRead: {
const auto flags{inst.Flags<IR::TextureInstInfo>()}; const auto flags{inst.Flags<IR::TextureInstInfo>()};
info.uses_typeless_image_reads |= flags.image_format == ImageFormat::Typeless; info.uses_typeless_image_reads |= flags.image_format == ImageFormat::Typeless;
info.uses_image_1d |=
flags.type == TextureType::Color1D || flags.type == TextureType::ColorArray1D;
info.uses_sparse_residency |= info.uses_sparse_residency |=
inst.GetAssociatedPseudoOperation(IR::Opcode::GetSparseFromOp) != nullptr; inst.GetAssociatedPseudoOperation(IR::Opcode::GetSparseFromOp) != nullptr;
break; break;
@@ -579,8 +577,6 @@ void VisitUsages(Info& info, IR::Inst& inst) {
const auto flags{inst.Flags<IR::TextureInstInfo>()}; const auto flags{inst.Flags<IR::TextureInstInfo>()};
info.uses_typeless_image_writes |= flags.image_format == ImageFormat::Typeless; info.uses_typeless_image_writes |= flags.image_format == ImageFormat::Typeless;
info.uses_image_buffers |= flags.type == TextureType::Buffer; info.uses_image_buffers |= flags.type == TextureType::Buffer;
info.uses_image_1d |=
flags.type == TextureType::Color1D || flags.type == TextureType::ColorArray1D;
break; break;
} }
case IR::Opcode::SubgroupEqMask: case IR::Opcode::SubgroupEqMask:
@@ -765,13 +761,9 @@ void VisitUsages(Info& info, IR::Inst& inst) {
case IR::Opcode::ImageAtomicAnd32: case IR::Opcode::ImageAtomicAnd32:
case IR::Opcode::ImageAtomicOr32: case IR::Opcode::ImageAtomicOr32:
case IR::Opcode::ImageAtomicXor32: case IR::Opcode::ImageAtomicXor32:
case IR::Opcode::ImageAtomicExchange32: { case IR::Opcode::ImageAtomicExchange32:
const auto flags{inst.Flags<IR::TextureInstInfo>()};
info.uses_atomic_image_u32 = true; info.uses_atomic_image_u32 = true;
info.uses_image_1d |=
flags.type == TextureType::Color1D || flags.type == TextureType::ColorArray1D;
break; break;
}
default: default:
break; break;
} }
+1 -20
View File
@@ -562,7 +562,6 @@ public:
})}; })};
// TODO: Read this from TIC // TODO: Read this from TIC
texture_descriptors[index].is_multisample |= desc.is_multisample; texture_descriptors[index].is_multisample |= desc.is_multisample;
texture_descriptors[index].is_integer |= desc.is_integer;
return index; return index;
} }
@@ -686,21 +685,6 @@ void TexturePass(Environment& env, IR::Program& program, const HostTranslateInfo
program.info.image_descriptors, program.info.image_descriptors,
}; };
const u32 sampled_dynamic_cap = DynamicSampledTextureCap(program.info, host_info, DynamicSampledTextureArrayCount(to_replace)); const u32 sampled_dynamic_cap = DynamicSampledTextureCap(program.info, host_info, DynamicSampledTextureArrayCount(to_replace));
bool has_last_is_integer{false};
u32 last_cbuf_index{};
u32 last_cbuf_offset{};
bool last_is_integer{false};
const auto is_texture_pixel_format_integer{[&](const ConstBufferAddr& cbuf_addr) {
if (has_last_is_integer && last_cbuf_index == cbuf_addr.index &&
last_cbuf_offset == cbuf_addr.offset) {
return last_is_integer;
}
last_is_integer = IsTexturePixelFormatIntegerCached(env, cbuf_addr);
last_cbuf_index = cbuf_addr.index;
last_cbuf_offset = cbuf_addr.offset;
has_last_is_integer = true;
return last_is_integer;
}};
for (TextureInst& texture_inst : to_replace) { for (TextureInst& texture_inst : to_replace) {
// TODO: Handle arrays // TODO: Handle arrays
IR::Inst* const inst{texture_inst.inst}; IR::Inst* const inst{texture_inst.inst};
@@ -765,7 +749,7 @@ void TexturePass(Environment& env, IR::Program& program, const HostTranslateInfo
} }
const bool is_written{inst->GetOpcode() != IR::Opcode::ImageRead}; const bool is_written{inst->GetOpcode() != IR::Opcode::ImageRead};
const bool is_read{inst->GetOpcode() != IR::Opcode::ImageWrite}; const bool is_read{inst->GetOpcode() != IR::Opcode::ImageWrite};
const bool is_integer{is_texture_pixel_format_integer(cbuf)}; const bool is_integer{IsTexturePixelFormatIntegerCached(env, cbuf)};
if (flags.type == TextureType::Buffer) { if (flags.type == TextureType::Buffer) {
index = descriptors.Add(ImageBufferDescriptor{ index = descriptors.Add(ImageBufferDescriptor{
.format = flags.image_format, .format = flags.image_format,
@@ -807,12 +791,10 @@ void TexturePass(Environment& env, IR::Program& program, const HostTranslateInfo
}); });
} else { } else {
count = std::min(count, sampled_dynamic_cap); count = std::min(count, sampled_dynamic_cap);
const bool is_integer{is_texture_pixel_format_integer(cbuf)};
index = descriptors.Add(TextureDescriptor{ index = descriptors.Add(TextureDescriptor{
.type = flags.type, .type = flags.type,
.is_depth = flags.is_depth != 0, .is_depth = flags.is_depth != 0,
.is_multisample = is_multisample, .is_multisample = is_multisample,
.is_integer = is_integer,
.has_secondary = cbuf.has_secondary, .has_secondary = cbuf.has_secondary,
.cbuf_index = cbuf.index, .cbuf_index = cbuf.index,
.cbuf_offset = cbuf.offset, .cbuf_offset = cbuf.offset,
@@ -823,7 +805,6 @@ void TexturePass(Environment& env, IR::Program& program, const HostTranslateInfo
.count = count, .count = count,
.size_shift = size_shift, .size_shift = size_shift,
}); });
flags.is_integer.Assign(is_integer ? 1 : 0);
} }
break; break;
} }
-9
View File
@@ -10,16 +10,12 @@
namespace Shader { namespace Shader {
enum class Stage : u32;
struct Profile { struct Profile {
u32 supported_spirv{0x00010000}; u32 supported_spirv{0x00010000};
bool unified_descriptor_binding{}; bool unified_descriptor_binding{};
bool support_descriptor_aliasing{}; bool support_descriptor_aliasing{};
bool support_int8{}; bool support_int8{};
bool support_uniform_and_storage_buffer_8bit{};
bool support_int16{}; bool support_int16{};
bool support_uniform_and_storage_buffer_16bit{};
bool support_int64{}; bool support_int64{};
bool support_vertex_instance_id{}; bool support_vertex_instance_id{};
bool support_float_controls{}; bool support_float_controls{};
@@ -34,7 +30,6 @@ struct Profile {
bool support_fp64_signed_zero_nan_preserve{}; bool support_fp64_signed_zero_nan_preserve{};
bool support_explicit_workgroup_layout{}; bool support_explicit_workgroup_layout{};
bool support_vote{}; bool support_vote{};
u32 supported_subgroup_stages{0x7F};
bool support_viewport_index_layer_non_geometry{}; bool support_viewport_index_layer_non_geometry{};
bool support_viewport_mask{}; bool support_viewport_mask{};
bool support_typeless_image_loads{}; bool support_typeless_image_loads{};
@@ -98,10 +93,6 @@ struct Profile {
u64 min_ssbo_alignment{}; u64 min_ssbo_alignment{};
u32 max_user_clip_distances{}; u32 max_user_clip_distances{};
bool SupportsSubgroupStage(Stage stage) const {
return (supported_subgroup_stages & (1u << static_cast<u32>(stage))) != 0;
}
}; };
} // namespace Shader } // namespace Shader
-1
View File
@@ -209,7 +209,6 @@ struct TextureDescriptor {
TextureType type; TextureType type;
bool is_depth; bool is_depth;
bool is_multisample; bool is_multisample;
bool is_integer;
bool has_secondary; bool has_secondary;
u32 cbuf_index; u32 cbuf_index;
u32 cbuf_offset; u32 cbuf_offset;
+15 -10
View File
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: 2016 Dolphin Emulator Project // SPDX-FileCopyrightText: 2016 Dolphin Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
@@ -53,14 +56,15 @@ u64 TestTimerSpeed(Core::Timing::CoreTiming& core_timing) {
} // Anonymous namespace } // Anonymous namespace
TEST_CASE("CoreTiming[BasicOrder]", "[core]") { TEST_CASE("CoreTiming[BasicOrder]", "[core]") {
Core::System system{};
ScopeInit guard; ScopeInit guard;
auto& core_timing = guard.core_timing; auto& core_timing = guard.core_timing;
std::vector<std::shared_ptr<Core::Timing::EventType>> events{ std::vector<std::shared_ptr<Core::Timing::EventType>> events{
Core::Timing::CreateEvent("callbackA", HostCallbackTemplate<0>), system.CreateTimingEvent("callbackA", HostCallbackTemplate<0>),
Core::Timing::CreateEvent("callbackB", HostCallbackTemplate<1>), system.CreateTimingEvent("callbackB", HostCallbackTemplate<1>),
Core::Timing::CreateEvent("callbackC", HostCallbackTemplate<2>), system.CreateTimingEvent("callbackC", HostCallbackTemplate<2>),
Core::Timing::CreateEvent("callbackD", HostCallbackTemplate<3>), system.CreateTimingEvent("callbackD", HostCallbackTemplate<3>),
Core::Timing::CreateEvent("callbackE", HostCallbackTemplate<4>), system.CreateTimingEvent("callbackE", HostCallbackTemplate<4>),
}; };
expected_callback = 0; expected_callback = 0;
@@ -93,14 +97,15 @@ TEST_CASE("CoreTiming[BasicOrder]", "[core]") {
} }
TEST_CASE("CoreTiming[BasicOrderNoPausing]", "[core]") { TEST_CASE("CoreTiming[BasicOrderNoPausing]", "[core]") {
Core::System system{};
ScopeInit guard; ScopeInit guard;
auto& core_timing = guard.core_timing; auto& core_timing = guard.core_timing;
std::vector<std::shared_ptr<Core::Timing::EventType>> events{ std::vector<std::shared_ptr<Core::Timing::EventType>> events{
Core::Timing::CreateEvent("callbackA", HostCallbackTemplate<0>), system.CreateTimingEvent("callbackA", HostCallbackTemplate<0>),
Core::Timing::CreateEvent("callbackB", HostCallbackTemplate<1>), system.CreateTimingEvent("callbackB", HostCallbackTemplate<1>),
Core::Timing::CreateEvent("callbackC", HostCallbackTemplate<2>), system.CreateTimingEvent("callbackC", HostCallbackTemplate<2>),
Core::Timing::CreateEvent("callbackD", HostCallbackTemplate<3>), system.CreateTimingEvent("callbackD", HostCallbackTemplate<3>),
Core::Timing::CreateEvent("callbackE", HostCallbackTemplate<4>), system.CreateTimingEvent("callbackE", HostCallbackTemplate<4>),
}; };
core_timing.SyncPause(true); core_timing.SyncPause(true);
+39 -35
View File
@@ -20,54 +20,58 @@ Decoder::Decoder(Host1x::Host1x& host1x_, s32 id_, const Host1x::NvdecCommon::Nv
Decoder::~Decoder() = default; Decoder::~Decoder() = default;
void Decoder::SetFrameDimensions(s32 width, s32 height) {
if (width <= 0 || height <= 0) {
frame_dimensions.reset();
return;
}
frame_dimensions = FFmpeg::FrameDimensions{width, height};
}
void Decoder::Decode() { void Decoder::Decode() {
if (!initialized) { if (!initialized) {
return; return;
} }
const auto packet_data = ComposeFrame(); const auto packet_data = ComposeFrame();
FFmpeg::FrameOffsets offsets{};
offsets.hidden = vp9_hidden_frame;
offsets.interlaced = IsInterlaced();
if (offsets.interlaced) {
std::tie(offsets.luma, offsets.luma_bottom, std::ignore, std::ignore) =
GetInterlacedOffsets();
} else {
std::tie(offsets.luma, std::ignore) = GetProgressiveOffsets();
}
// Send assembled bitstream to decoder. // Send assembled bitstream to decoder.
if (!decode_api.SendPacket(packet_data, offsets, GetFrameDimensions())) { if (!decode_api.SendPacket(packet_data)) {
return; return;
} }
auto push = [&](u64 luma, std::shared_ptr<FFmpeg::Frame> frame) { // Only receive/store visible frames.
if (UsingDecodeOrder()) { if (vp9_hidden_frame) {
host1x.frame_queue.PushDecodeOrder(id, luma, std::move(frame)); return;
} else { }
host1x.frame_queue.PushPresentOrder(id, luma, std::move(frame));
}
};
while (auto result = decode_api.ReceiveFrame()) { // Receive output frames from decoder.
auto& [frame, frame_offsets] = *result; auto frame = decode_api.ReceiveFrame();
if (!frame) {
continue; if (!frame) {
return;
}
if (IsInterlaced()) {
auto [luma_top, luma_bottom, chroma_top, chroma_bottom] = GetInterlacedOffsets();
auto frame_copy = frame;
if (!frame.get()) {
LOG_ERROR(HW_GPU,
"Nvdec {} failed to decode interlaced frame for top {:#X} bottom 0x{:X}", id,
luma_top, luma_bottom);
} }
if (frame_offsets.interlaced) {
auto frame_copy = frame; if (UsingDecodeOrder()) {
push(frame_offsets.luma, std::move(frame)); host1x.frame_queue.PushDecodeOrder(id, luma_top, std::move(frame));
push(frame_offsets.luma_bottom, std::move(frame_copy)); host1x.frame_queue.PushDecodeOrder(id, luma_bottom, std::move(frame_copy));
} else { } else {
push(frame_offsets.luma, std::move(frame)); host1x.frame_queue.PushPresentOrder(id, luma_top, std::move(frame));
host1x.frame_queue.PushPresentOrder(id, luma_bottom, std::move(frame_copy));
}
} else {
auto [luma_offset, chroma_offset] = GetProgressiveOffsets();
if (!frame.get()) {
LOG_ERROR(HW_GPU, "Nvdec {} failed to decode progressive frame for luma {:#X}", id,
luma_offset);
}
if (UsingDecodeOrder()) {
host1x.frame_queue.PushDecodeOrder(id, luma_offset, std::move(frame));
} else {
host1x.frame_queue.PushPresentOrder(id, luma_offset, std::move(frame));
} }
} }
} }
-8
View File
@@ -10,7 +10,6 @@
#include <mutex> #include <mutex>
#include <optional> #include <optional>
#include <string_view> #include <string_view>
#include <tuple>
#include <ankerl/unordered_dense.h> #include <ankerl/unordered_dense.h>
#include <queue> #include <queue>
@@ -47,19 +46,12 @@ protected:
virtual std::tuple<u64, u64, u64, u64> GetInterlacedOffsets() = 0; virtual std::tuple<u64, u64, u64, u64> GetInterlacedOffsets() = 0;
virtual bool IsInterlaced() = 0; virtual bool IsInterlaced() = 0;
void SetFrameDimensions(s32 width, s32 height);
std::optional<FFmpeg::FrameDimensions> GetFrameDimensions() const {
return frame_dimensions;
}
FFmpeg::DecodeApi decode_api; FFmpeg::DecodeApi decode_api;
Host1x::Host1x& host1x; Host1x::Host1x& host1x;
const Host1x::NvdecCommon::NvdecRegisters& regs; const Host1x::NvdecCommon::NvdecRegisters& regs;
s32 id; s32 id;
bool initialized : 1 = false; bool initialized : 1 = false;
bool vp9_hidden_frame : 1 = false; bool vp9_hidden_frame : 1 = false;
std::optional<FFmpeg::FrameDimensions> frame_dimensions;
}; };
} // namespace Tegra } // namespace Tegra
-4
View File
@@ -52,10 +52,6 @@ bool H264::IsInterlaced() {
std::span<const u8> H264::ComposeFrame() { std::span<const u8> H264::ComposeFrame() {
host1x.gmmu_manager.ReadBlock(regs.picture_info_offset.Address(), &current_context, sizeof(H264DecoderContext)); host1x.gmmu_manager.ReadBlock(regs.picture_info_offset.Address(), &current_context, sizeof(H264DecoderContext));
const auto& params = current_context.h264_parameter_set;
SetFrameDimensions(static_cast<s32>(params.pic_width_in_mbs) * 16,
static_cast<s32>(params.frame_height_in_mbs) * 16);
const s64 frame_number = current_context.h264_parameter_set.frame_number.Value(); const s64 frame_number = current_context.h264_parameter_set.frame_number.Value();
if (!is_first_frame && frame_number != 0) { if (!is_first_frame && frame_number != 0) {
frame_scratch.resize_destructive(current_context.stream_len); frame_scratch.resize_destructive(current_context.stream_len);
-1
View File
@@ -35,7 +35,6 @@ std::tuple<u64, u64, u64, u64> VP8::GetInterlacedOffsets() {
std::span<const u8> VP8::ComposeFrame() { std::span<const u8> VP8::ComposeFrame() {
host1x.gmmu_manager.ReadBlock(regs.picture_info_offset.Address(), &current_context, sizeof(VP8PictureInfo)); host1x.gmmu_manager.ReadBlock(regs.picture_info_offset.Address(), &current_context, sizeof(VP8PictureInfo));
SetFrameDimensions(current_context.frame_width, current_context.frame_height);
const bool is_key_frame = current_context.key_frame == 1u; const bool is_key_frame = current_context.key_frame == 1u;
const auto bitstream_size = size_t(current_context.vld_buffer_size); const auto bitstream_size = size_t(current_context.vld_buffer_size);
-1
View File
@@ -841,7 +841,6 @@ std::span<const u8> VP9::ComposeFrame() {
{ {
Vp9FrameContainer curr_frame = GetCurrentFrame(); Vp9FrameContainer curr_frame = GetCurrentFrame();
current_frame_info = curr_frame.info; current_frame_info = curr_frame.info;
SetFrameDimensions(current_frame_info.frame_size.width, current_frame_info.frame_size.height);
bitstream = std::move(curr_frame.bit_stream); bitstream = std::move(curr_frame.bit_stream);
} }
// The uncompressed header routine sets PrevProb parameters needed for the compressed header // The uncompressed header routine sets PrevProb parameters needed for the compressed header
+10 -159
View File
@@ -4,10 +4,6 @@
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
#include <cstring>
#include <string_view>
#include <vector>
#include "common/assert.h" #include "common/assert.h"
#include "common/logging.h" #include "common/logging.h"
#include "common/scope_exit.h" #include "common/scope_exit.h"
@@ -88,52 +84,6 @@ std::string AVError(int errnum) {
return errbuf; return errbuf;
} }
#if defined(__ANDROID__)
size_t FindNalStartCode(std::span<const u8> data, size_t i) {
const size_t n = data.size();
if (i + 3 < n && data[i] == 0 && data[i + 1] == 0 && data[i + 2] == 0 && data[i + 3] == 1) {
return 4;
}
if (i + 2 < n && data[i] == 0 && data[i + 1] == 0 && data[i + 2] == 1) {
return 3;
}
return 0;
}
std::vector<u8> ExtractH264ParameterSetExtradata(std::span<const u8> packet) {
std::vector<u8> extradata;
const size_t size = packet.size();
size_t i = 0;
while (i < size) {
const size_t sc = FindNalStartCode(packet, i);
if (sc == 0) {
++i;
continue;
}
const size_t nal_start = i + sc;
if (nal_start >= size) {
break;
}
const u8 nal_type = packet[nal_start] & 0x1F;
size_t j = nal_start + 1;
while (j < size && FindNalStartCode(packet, j) == 0) {
++j;
}
if (nal_type == 7 || nal_type == 8) {
constexpr u8 start[4] = {0, 0, 0, 1};
extradata.insert(extradata.end(), start, start + sizeof(start));
extradata.insert(extradata.end(), packet.begin() + nal_start, packet.begin() + j);
} else if (nal_type == 1 || nal_type == 5) {
break;
}
i = j;
}
return extradata;
}
#endif
} }
Packet::Packet(std::span<const u8> data) { Packet::Packet(std::span<const u8> data) {
@@ -168,24 +118,7 @@ Decoder::Decoder(Tegra::Host1x::NvdecCommon::VideoCodec codec) {
return AV_CODEC_ID_NONE; return AV_CODEC_ID_NONE;
} }
}(); }();
m_codec = avcodec_find_decoder(av_codec);
#if defined(__ANDROID__)
if (Settings::values.nvdec_emulation.GetValue() == Settings::NvdecEmulation::Gpu) {
const char* mc_name = nullptr;
switch (av_codec) {
case AV_CODEC_ID_H264: mc_name = "h264_mediacodec"; break;
case AV_CODEC_ID_VP8: mc_name = "vp8_mediacodec"; break;
case AV_CODEC_ID_VP9: mc_name = "vp9_mediacodec"; break;
default: break;
}
if (mc_name) {
m_codec = avcodec_find_decoder_by_name(mc_name);
}
}
#endif
if (!m_codec) {
m_codec = avcodec_find_decoder(av_codec);
}
} }
bool Decoder::SupportsDecodingOnDevice(AVPixelFormat* out_pix_fmt, AVHWDeviceType type) const { bool Decoder::SupportsDecodingOnDevice(AVPixelFormat* out_pix_fmt, AVHWDeviceType type) const {
@@ -281,10 +214,6 @@ DecoderContext::DecoderContext(const Decoder& decoder) : m_decoder{decoder} {
av_opt_set(m_codec_context->priv_data, "tune", "zerolatency", 0); av_opt_set(m_codec_context->priv_data, "tune", "zerolatency", 0);
m_codec_context->thread_count = 0; m_codec_context->thread_count = 0;
m_codec_context->thread_type &= ~FF_THREAD_FRAME; m_codec_context->thread_type &= ~FF_THREAD_FRAME;
#if defined(__ANDROID__)
m_codec_context->flags |= AV_CODEC_FLAG_LOW_DELAY;
m_codec_context->flags2 |= AV_CODEC_FLAG2_FAST;
#endif
} }
DecoderContext::~DecoderContext() { DecoderContext::~DecoderContext() {
@@ -298,25 +227,15 @@ void DecoderContext::InitializeHardwareDecoder(const HardwareContext& context, A
m_codec_context->pix_fmt = hw_pix_fmt; m_codec_context->pix_fmt = hw_pix_fmt;
} }
bool DecoderContext::OpenContext(const Decoder& decoder, std::span<const u8> extradata) { bool DecoderContext::OpenContext(const Decoder& decoder) {
if (!extradata.empty()) {
av_freep(&m_codec_context->extradata);
m_codec_context->extradata = static_cast<u8*>(
av_mallocz(extradata.size() + AV_INPUT_BUFFER_PADDING_SIZE));
if (!m_codec_context->extradata) {
LOG_ERROR(HW_GPU, "Failed to allocate extradata");
return false;
}
std::memcpy(m_codec_context->extradata, extradata.data(), extradata.size());
m_codec_context->extradata_size = static_cast<int>(extradata.size());
}
if (const int ret = avcodec_open2(m_codec_context, decoder.GetCodec(), nullptr); ret < 0) { if (const int ret = avcodec_open2(m_codec_context, decoder.GetCodec(), nullptr); ret < 0) {
LOG_ERROR(HW_GPU, "avcodec_open2 error: {}", AVError(ret)); LOG_ERROR(HW_GPU, "avcodec_open2 error: {}", AVError(ret));
return false; return false;
} }
LOG_INFO(HW_GPU, "Using decoder {}", decoder.GetCodec()->name); if (!m_codec_context->hw_device_ctx) {
LOG_INFO(HW_GPU, "Using FFmpeg CPU decoder");
}
return true; return true;
} }
@@ -362,13 +281,6 @@ void DecodeApi::Reset() {
m_hardware_context.reset(); m_hardware_context.reset();
m_decoder_context.reset(); m_decoder_context.reset();
m_decoder.reset(); m_decoder.reset();
m_opened = false;
m_defer_android_mediacodec_open = false;
m_needs_h264_extradata = false;
m_next_pts = 0;
while (!m_pending_offsets.empty()) {
m_pending_offsets.pop();
}
} }
bool DecodeApi::Initialize(Tegra::Host1x::NvdecCommon::VideoCodec codec) { bool DecodeApi::Initialize(Tegra::Host1x::NvdecCommon::VideoCodec codec) {
@@ -376,90 +288,29 @@ bool DecodeApi::Initialize(Tegra::Host1x::NvdecCommon::VideoCodec codec) {
m_decoder.emplace(codec); m_decoder.emplace(codec);
m_decoder_context.emplace(*m_decoder); m_decoder_context.emplace(*m_decoder);
bool is_mediacodec = false;
#if defined(__ANDROID__)
const std::string_view decoder_name = m_decoder->GetCodec() ? m_decoder->GetCodec()->name : "";
is_mediacodec = decoder_name == "h264_mediacodec" ||
decoder_name == "vp8_mediacodec" ||
decoder_name == "vp9_mediacodec";
#endif
// Enable GPU decoding if requested. // Enable GPU decoding if requested.
if (!is_mediacodec && if (Settings::values.nvdec_emulation.GetValue() == Settings::NvdecEmulation::Gpu) {
Settings::values.nvdec_emulation.GetValue() == Settings::NvdecEmulation::Gpu) {
m_hardware_context.emplace(); m_hardware_context.emplace();
m_hardware_context->InitializeForDecoder(*m_decoder_context, *m_decoder); m_hardware_context->InitializeForDecoder(*m_decoder_context, *m_decoder);
} }
#if defined(__ANDROID__)
m_defer_android_mediacodec_open = is_mediacodec;
m_needs_h264_extradata = decoder_name == "h264_mediacodec";
if (m_defer_android_mediacodec_open) {
return true;
}
#endif
// Open the decoder context. // Open the decoder context.
if (!m_decoder_context->OpenContext(*m_decoder)) { if (!m_decoder_context->OpenContext(*m_decoder)) {
this->Reset(); this->Reset();
return false; return false;
} }
m_opened = true;
return true; return true;
} }
bool DecodeApi::SendPacket(std::span<const u8> packet_data, const FrameOffsets& offsets, bool DecodeApi::SendPacket(std::span<const u8> packet_data) {
std::optional<FrameDimensions> dimensions) {
if (!m_opened) {
std::vector<u8> extradata;
#if defined(__ANDROID__)
if (m_defer_android_mediacodec_open) {
if (!dimensions) {
return true;
}
auto* ctx = m_decoder_context->GetCodecContext();
ctx->width = dimensions->width;
ctx->height = dimensions->height;
ctx->coded_width = dimensions->width;
ctx->coded_height = dimensions->height;
}
if (m_needs_h264_extradata) {
extradata = ExtractH264ParameterSetExtradata(packet_data);
if (extradata.empty()) {
return true;
}
}
#endif
if (!m_decoder_context->OpenContext(*m_decoder, extradata)) {
this->Reset();
return false;
}
m_opened = true;
}
if (!offsets.hidden) {
m_pending_offsets.push(offsets);
}
FFmpeg::Packet packet(packet_data); FFmpeg::Packet packet(packet_data);
packet.GetPacket()->pts = m_next_pts;
packet.GetPacket()->dts = m_next_pts;
++m_next_pts;
return m_decoder_context->SendPacket(packet); return m_decoder_context->SendPacket(packet);
} }
std::optional<DecodeApi::DecodedFrame> DecodeApi::ReceiveFrame() { std::shared_ptr<Frame> DecodeApi::ReceiveFrame() {
auto frame = m_decoder_context->ReceiveFrame(); // Receive raw frame from decoder.
if (!frame) { return m_decoder_context->ReceiveFrame();
return std::nullopt;
}
FrameOffsets offsets{};
if (!m_pending_offsets.empty()) {
offsets = m_pending_offsets.front();
m_pending_offsets.pop();
}
return DecodedFrame{std::move(frame), offsets};
} }
} }
+3 -26
View File
@@ -179,7 +179,7 @@ public:
~DecoderContext(); ~DecoderContext();
void InitializeHardwareDecoder(const HardwareContext& context, AVPixelFormat hw_pix_fmt); void InitializeHardwareDecoder(const HardwareContext& context, AVPixelFormat hw_pix_fmt);
bool OpenContext(const Decoder& decoder, std::span<const u8> extradata = {}); bool OpenContext(const Decoder& decoder);
bool SendPacket(const Packet& packet); bool SendPacket(const Packet& packet);
std::shared_ptr<Frame> ReceiveFrame(); std::shared_ptr<Frame> ReceiveFrame();
@@ -198,18 +198,6 @@ private:
bool m_decode_order{}; bool m_decode_order{};
}; };
struct FrameOffsets {
bool interlaced{};
bool hidden{};
u64 luma{};
u64 luma_bottom{};
};
struct FrameDimensions {
s32 width{};
s32 height{};
};
class DecodeApi { class DecodeApi {
public: public:
YUZU_NON_COPYABLE(DecodeApi); YUZU_NON_COPYABLE(DecodeApi);
@@ -225,24 +213,13 @@ public:
return m_decoder_context->UsingDecodeOrder(); return m_decoder_context->UsingDecodeOrder();
} }
bool SendPacket(std::span<const u8> packet_data, const FrameOffsets& offsets, bool SendPacket(std::span<const u8> packet_data);
std::optional<FrameDimensions> dimensions = std::nullopt); std::shared_ptr<Frame> ReceiveFrame();
struct DecodedFrame {
std::shared_ptr<Frame> frame;
FrameOffsets offsets;
};
std::optional<DecodedFrame> ReceiveFrame();
private: private:
std::optional<FFmpeg::Decoder> m_decoder; std::optional<FFmpeg::Decoder> m_decoder;
std::optional<FFmpeg::DecoderContext> m_decoder_context; std::optional<FFmpeg::DecoderContext> m_decoder_context;
std::optional<FFmpeg::HardwareContext> m_hardware_context; std::optional<FFmpeg::HardwareContext> m_hardware_context;
bool m_opened{};
bool m_defer_android_mediacodec_open{};
bool m_needs_h264_extradata{};
s64 m_next_pts{};
std::queue<FrameOffsets> m_pending_offsets;
}; };
} // namespace FFmpeg } // namespace FFmpeg
-1
View File
@@ -31,7 +31,6 @@ Nvdec::Nvdec(Host1x& host1x_, s32 id_, u32 syncpt)
Nvdec::~Nvdec() { Nvdec::~Nvdec() {
LOG_INFO(HW_GPU, "Destroying nvdec {}", id); LOG_INFO(HW_GPU, "Destroying nvdec {}", id);
host1x.frame_queue.Close(id);
} }
void Nvdec::ProcessMethod(u32 method, u32 argument) { void Nvdec::ProcessMethod(u32 method, u32 argument) {
+1 -8
View File
@@ -118,7 +118,6 @@ void Vic::Execute() noexcept {
output_surface.resize(output_width * output_height); output_surface.resize(output_width * output_height);
if (Settings::values.nvdec_emulation.GetValue() != Settings::NvdecEmulation::Off) { if (Settings::values.nvdec_emulation.GetValue() != Settings::NvdecEmulation::Off) {
bool decoded_frame = false;
for (size_t i = 0; i < config.slot_structs.size(); i++) { for (size_t i = 0; i < config.slot_structs.size(); i++) {
if (auto& slot_config = config.slot_structs[i]; slot_config.config.slot_enable) { if (auto& slot_config = config.slot_structs[i]; slot_config.config.slot_enable) {
auto const luma_offset = regs.surfaces[i][SurfaceIndex::Current].luma.Address(); auto const luma_offset = regs.surfaces[i][SurfaceIndex::Current].luma.Address();
@@ -137,17 +136,11 @@ void Vic::Execute() noexcept {
break; break;
} }
Blend(config, slot_config, config.output_surface_config.out_pixel_format); Blend(config, slot_config, config.output_surface_config.out_pixel_format);
decoded_frame = true;
} else { } else {
LOG_TRACE(HW_GPU, "Vic {} failed to get frame with offset {:#X}", id, luma_offset); LOG_ERROR(HW_GPU, "Vic {} failed to get frame with offset {:#X}", id, luma_offset);
} }
} }
} }
if (decoded_frame) {
has_decoded_frame = true;
} else if (!has_decoded_frame) {
return;
}
} else { } else {
// Fill the frame with black, as otherwise they can have random data and be very glitchy. // Fill the frame with black, as otherwise they can have random data and be very glitchy.
std::fill(output_surface.begin(), output_surface.end(), Pixel{}); std::fill(output_surface.begin(), output_surface.end(), Pixel{});
-1
View File
@@ -628,7 +628,6 @@ private:
s32 id; s32 id;
s32 nvdec_id{-1}; s32 nvdec_id{-1};
bool has_decoded_frame{};
u32 syncpoint; u32 syncpoint;
}; };
+6 -288
View File
@@ -1,6 +1,3 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
@@ -43,11 +40,7 @@ layout(binding = BINDING_INPUT_BUFFER, std430) readonly restrict buffer InputBuf
uvec4 astc_data[]; uvec4 astc_data[];
}; };
#ifdef VULKAN
layout(binding = BINDING_OUTPUT_IMAGE) uniform writeonly restrict image2DArray dest_image;
#else
layout(binding = BINDING_OUTPUT_IMAGE, rgba8) uniform writeonly restrict image2DArray dest_image; layout(binding = BINDING_OUTPUT_IMAGE, rgba8) uniform writeonly restrict image2DArray dest_image;
#endif
const uint GOB_SIZE_X_SHIFT = 6; const uint GOB_SIZE_X_SHIFT = 6;
const uint GOB_SIZE_Y_SHIFT = 3; const uint GOB_SIZE_Y_SHIFT = 3;
@@ -596,163 +589,6 @@ ivec4 BlueContract(int a, int r, int g, int b) {
return ivec4(a, (r + b) >> 1, (g + b) >> 1, b); return ivec4(a, (r + b) >> 1, (g + b) >> 1, b);
} }
bool IsHDRColorEndpointMode(uint cem) {
return cem == 2u || cem == 3u || cem == 7u || cem == 11u || cem == 14u || cem == 15u;
}
// Sign-extends the low nbits of value (a 2's complement field packed into the bottom of an
// otherwise-unsigned integer), per C.2.15's HDR endpoint bitfield unpacking.
int SignExtend(int value, uint nbits) {
int sign_bit = 1 << (nbits - 1u);
return (value ^ sign_bit) - sign_bit;
}
// HDR Endpoint Mode 7 (C.2.15): base RGB + scale factor.
void DecodeHDREndpointMode7(uint v0, uint v1, uint v2, uint v3, out ivec3 e0, out ivec3 e1) {
uint modeval = ((v0 & 0xC0u) >> 6u) | ((v1 & 0x80u) >> 5u) | ((v2 & 0x80u) >> 4u);
uint majcomp;
uint mode;
if ((modeval & 0xCu) != 0xCu) {
majcomp = modeval >> 2u;
mode = modeval & 3u;
} else if (modeval != 0xFu) {
majcomp = modeval & 3u;
mode = 4u;
} else {
majcomp = 0u;
mode = 5u;
}
int red = int(v0 & 0x3Fu);
int green = int(v1 & 0x1Fu);
int blue = int(v2 & 0x1Fu);
int scale = int(v3 & 0x1Fu);
uint x0 = (v1 >> 6u) & 1u;
uint x1 = (v1 >> 5u) & 1u;
uint x2 = (v2 >> 6u) & 1u;
uint x3 = (v2 >> 5u) & 1u;
uint x4 = (v3 >> 7u) & 1u;
uint x5 = (v3 >> 6u) & 1u;
uint x6 = (v3 >> 5u) & 1u;
uint ohm = 1u << mode;
if ((ohm & 0x30u) != 0u) green |= int(x0 << 6u);
if ((ohm & 0x3Au) != 0u) green |= int(x1 << 5u);
if ((ohm & 0x30u) != 0u) blue |= int(x2 << 6u);
if ((ohm & 0x3Au) != 0u) blue |= int(x3 << 5u);
if ((ohm & 0x3Du) != 0u) scale |= int(x6 << 5u);
if ((ohm & 0x2Du) != 0u) scale |= int(x5 << 6u);
if ((ohm & 0x04u) != 0u) scale |= int(x4 << 7u);
if ((ohm & 0x3Bu) != 0u) red |= int(x4 << 6u);
if ((ohm & 0x04u) != 0u) red |= int(x3 << 6u);
if ((ohm & 0x10u) != 0u) red |= int(x5 << 7u);
if ((ohm & 0x0Fu) != 0u) red |= int(x2 << 7u);
if ((ohm & 0x05u) != 0u) red |= int(x1 << 8u);
if ((ohm & 0x0Au) != 0u) red |= int(x0 << 8u);
if ((ohm & 0x05u) != 0u) red |= int(x0 << 9u);
if ((ohm & 0x02u) != 0u) red |= int(x6 << 9u);
if ((ohm & 0x01u) != 0u) red |= int(x3 << 10u);
if ((ohm & 0x02u) != 0u) red |= int(x5 << 10u);
int shamts[6] = int[](1, 1, 2, 3, 4, 5);
int shamt = shamts[mode];
red <<= shamt;
green <<= shamt;
blue <<= shamt;
scale <<= shamt;
if (mode != 5u) {
green = red - green;
blue = red - blue;
}
if (majcomp == 1u) {
int t = red; red = green; green = t;
}
if (majcomp == 2u) {
int t = red; red = blue; blue = t;
}
e1 = ivec3(clamp(red, 0, 0xFFF), clamp(green, 0, 0xFFF), clamp(blue, 0, 0xFFF));
e0 = ivec3(clamp(red - scale, 0, 0xFFF), clamp(green - scale, 0, 0xFFF),
clamp(blue - scale, 0, 0xFFF));
}
// HDR Endpoint Mode 11 (C.2.15): direct RGB pair. Shared by modes 11, 14 and 15, which all
// decode their RGB the same way and only differ in how alpha is filled in.
void DecodeHDREndpointMode11(uint v0, uint v1, uint v2, uint v3, uint v4, uint v5, out ivec3 e0,
out ivec3 e1) {
uint majcomp = ((v4 & 0x80u) >> 7u) | ((v5 & 0x80u) >> 6u);
if (majcomp == 3u) {
e0 = ivec3(int(v0 << 4u), int(v2 << 4u), int((v4 & 0x7Fu) << 5u));
e1 = ivec3(int(v1 << 4u), int(v3 << 4u), int((v5 & 0x7Fu) << 5u));
return;
}
uint mode = ((v1 & 0x80u) >> 7u) | ((v2 & 0x80u) >> 6u) | ((v3 & 0x80u) >> 5u);
int va = int(v0 | ((v1 & 0x40u) << 2u));
int vb0 = int(v2 & 0x3Fu);
int vb1 = int(v3 & 0x3Fu);
int vc = int(v1 & 0x3Fu);
int vd0 = int(v4 & 0x7Fu);
int vd1 = int(v5 & 0x7Fu);
int dbitstab[8] = int[](7, 6, 7, 6, 5, 6, 5, 6);
vd0 = SignExtend(vd0, uint(dbitstab[mode]));
vd1 = SignExtend(vd1, uint(dbitstab[mode]));
uint x0 = (v2 >> 6u) & 1u;
uint x1 = (v3 >> 6u) & 1u;
uint x2 = (v4 >> 6u) & 1u;
uint x3 = (v5 >> 6u) & 1u;
uint x4 = (v4 >> 5u) & 1u;
uint x5 = (v5 >> 5u) & 1u;
uint ohm = 1u << mode;
if ((ohm & 0xA4u) != 0u) va |= int(x0 << 9u);
if ((ohm & 0x08u) != 0u) va |= int(x2 << 9u);
if ((ohm & 0x50u) != 0u) va |= int(x4 << 9u);
if ((ohm & 0x50u) != 0u) va |= int(x5 << 10u);
if ((ohm & 0xA0u) != 0u) va |= int(x1 << 10u);
if ((ohm & 0xC0u) != 0u) va |= int(x2 << 11u);
if ((ohm & 0x04u) != 0u) vc |= int(x1 << 6u);
if ((ohm & 0xE8u) != 0u) vc |= int(x3 << 6u);
if ((ohm & 0x20u) != 0u) vc |= int(x2 << 7u);
if ((ohm & 0x5Bu) != 0u) vb0 |= int(x0 << 6u);
if ((ohm & 0x5Bu) != 0u) vb1 |= int(x1 << 6u);
if ((ohm & 0x12u) != 0u) vb0 |= int(x2 << 7u);
if ((ohm & 0x12u) != 0u) vb1 |= int(x3 << 7u);
int shamt = (int(mode) >> 1) ^ 3;
va <<= shamt;
vb0 <<= shamt;
vb1 <<= shamt;
vc <<= shamt;
vd0 <<= shamt;
vd1 <<= shamt;
int r1 = clamp(va, 0, 0xFFF);
int g1 = clamp(va - vb0, 0, 0xFFF);
int b1 = clamp(va - vb1, 0, 0xFFF);
int r0 = clamp(va - vc, 0, 0xFFF);
int g0 = clamp(va - vb0 - vc - vd0, 0, 0xFFF);
int b0 = clamp(va - vb1 - vc - vd1, 0, 0xFFF);
if (majcomp == 1u) {
int t;
t = r0; r0 = g0; g0 = t;
t = r1; r1 = g1; g1 = t;
} else if (majcomp == 2u) {
int t;
t = r0; r0 = b0; b0 = t;
t = r1; r1 = b1; b1 = t;
}
e0 = ivec3(r0, g0, b0);
e1 = ivec3(r1, g1, b1);
}
void ComputeEndpoints(out uvec4 ep1, out uvec4 ep2, uint color_endpoint_mode, uint color_values[32], void ComputeEndpoints(out uvec4 ep1, out uvec4 ep2, uint color_endpoint_mode, uint color_values[32],
inout uint colvals_index) { inout uint colvals_index) {
#define READ_UINT_VALUES(N) \ #define READ_UINT_VALUES(N) \
@@ -879,88 +715,8 @@ void ComputeEndpoints(out uvec4 ep1, out uvec4 ep2, uint color_endpoint_mode, ui
} }
break; break;
} }
case 2: {
READ_UINT_VALUES(2)
uint y0, y1;
if (V[0].y >= V[0].x) {
y0 = V[0].x << 4u;
y1 = V[0].y << 4u;
} else {
y0 = (V[0].y << 4u) + 8u;
y1 = (V[0].x << 4u) - 8u;
}
ep1 = uvec4(0x780u, y0, y0, y0);
ep2 = uvec4(0x780u, y1, y1, y1);
break;
}
case 3: {
READ_UINT_VALUES(2)
uint y0, d;
if ((V[0].x & 0x80u) != 0u) {
y0 = ((V[0].y & 0xE0u) << 4u) | ((V[0].x & 0x7Fu) << 2u);
d = (V[0].y & 0x1Fu) << 2u;
} else {
y0 = ((V[0].y & 0xF0u) << 4u) | ((V[0].x & 0x7Fu) << 1u);
d = (V[0].y & 0x0Fu) << 1u;
}
const uint y1 = min(y0 + d, 0xFFFu);
ep1 = uvec4(0x780u, y0, y0, y0);
ep2 = uvec4(0x780u, y1, y1, y1);
break;
}
case 7: {
READ_UINT_VALUES(4)
ivec3 e0, e1;
DecodeHDREndpointMode7(V[0].x, V[0].y, V[0].z, V[0].w, e0, e1);
ep1 = uvec4(0x780u, uint(e0.x), uint(e0.y), uint(e0.z));
ep2 = uvec4(0x780u, uint(e1.x), uint(e1.y), uint(e1.z));
break;
}
case 11: {
READ_UINT_VALUES(6)
ivec3 e0, e1;
DecodeHDREndpointMode11(V[0].x, V[0].y, V[0].z, V[0].w, V[1].x, V[1].y, e0, e1);
ep1 = uvec4(0x780u, uint(e0.x), uint(e0.y), uint(e0.z));
ep2 = uvec4(0x780u, uint(e1.x), uint(e1.y), uint(e1.z));
break;
}
case 14: {
READ_UINT_VALUES(8)
ivec3 e0, e1;
DecodeHDREndpointMode11(V[0].x, V[0].y, V[0].z, V[0].w, V[1].x, V[1].y, e0, e1);
// Only HDR mode with LDR (8-bit UNORM)-interpreted alpha; left as-is (0-255).
ep1 = uvec4(V[1].z, uint(e0.x), uint(e0.y), uint(e0.z));
ep2 = uvec4(V[1].w, uint(e1.x), uint(e1.y), uint(e1.z));
break;
}
case 15: {
READ_UINT_VALUES(8)
ivec3 e0, e1;
DecodeHDREndpointMode11(V[0].x, V[0].y, V[0].z, V[0].w, V[1].x, V[1].y, e0, e1);
const uint mode = ((V[1].z >> 7u) & 1u) | ((V[1].w >> 6u) & 2u);
int a6 = int(V[1].z & 0x7Fu);
int a7 = int(V[1].w & 0x7Fu);
int alpha0, alpha1;
if (mode == 3u) {
alpha0 = a6 << 5;
alpha1 = a7 << 5;
} else {
a6 |= (a7 << int(mode + 1u)) & 0x780;
a7 &= int(0x3Fu >> mode);
a7 ^= int(0x20u >> mode);
a7 -= int(0x20u >> mode);
a6 <<= int(4u - mode);
a7 <<= int(4u - mode);
a7 += a6;
alpha0 = a6;
alpha1 = clamp(a7, 0, 0xFFF);
}
ep1 = uvec4(uint(alpha0), uint(e0.x), uint(e0.y), uint(e0.z));
ep2 = uvec4(uint(alpha1), uint(e1.x), uint(e1.y), uint(e1.z));
break;
}
default: { default: {
// Not a valid CEM at all (all 16 values are now handled above). // HDR mode, or more likely a bug computing the color_endpoint_mode
ep1 = uvec4(0xFF, 0xFF, 0, 0); ep1 = uvec4(0xFF, 0xFF, 0, 0);
ep2 = uvec4(0xFF, 0xFF, 0, 0); ep2 = uvec4(0xFF, 0xFF, 0, 0);
break; break;
@@ -1356,51 +1112,13 @@ void DecompressBlock(ivec3 coord) {
if (num_partitions > 1) { if (num_partitions > 1) {
local_partition = Select2DPartition(partition_index, i, j, num_partitions); local_partition = Select2DPartition(partition_index, i, j, num_partitions);
} }
const uint local_cem = color_endpoint_mode[local_partition]; const uvec4 C0 = ReplicateByteTo16(endpoints0[local_partition]);
const uvec4 C1 = ReplicateByteTo16(endpoints1[local_partition]);
const uvec4 weight_vec = GetUnquantizedWeightVector(j, i, size_params, plane_index, dual_plane); const uvec4 weight_vec = GetUnquantizedWeightVector(j, i, size_params, plane_index, dual_plane);
const vec4 Cf =
vec4 p; vec4((C0 * (uvec4(64) - weight_vec) + C1 * weight_vec + uvec4(32)) / 64);
if (IsHDRColorEndpointMode(local_cem)) { const vec4 p = (Cf / 65535.0f);
// Endpoints are raw 12-bit pseudo-logarithmic values; shift left 4 bits to
// become 16-bit before interpolating, per C.2.19.
const uvec4 C0 = endpoints0[local_partition] << 4u;
const uvec4 C1 = endpoints1[local_partition] << 4u;
const uvec4 C = (C0 * (uvec4(64) - weight_vec) + C1 * weight_vec + uvec4(32)) / 64u;
const uvec4 E = (C & uvec4(0xF800u)) >> 11u;
const uvec4 M = C & uvec4(0x7FFu);
const uvec4 Mt_lo = 3u * M;
const uvec4 Mt_mid = 4u * M - 512u;
const uvec4 Mt_hi = 5u * M - 2048u;
const uvec4 Mt =
mix(Mt_lo, mix(Mt_mid, Mt_hi, greaterThanEqual(M, uvec4(1536u))),
greaterThanEqual(M, uvec4(512u)));
const uvec4 Cf = (E << 10u) + (Mt >> 3u);
// +Inf/NaN clamps to the largest finite FP16 value (0x7BFF).
const uvec4 half_bits = mix(Cf, uvec4(0x7BFFu), greaterThanEqual(Cf, uvec4(0x7C00u)));
p = vec4(unpackHalf2x16(half_bits.x).x, unpackHalf2x16(half_bits.y).x,
unpackHalf2x16(half_bits.z).x, unpackHalf2x16(half_bits.w).x);
// Mode 14 keeps an LDR (8-bit UNORM)-interpreted alpha; component 0 here (A,
// see the ep1/ep2 layout used throughout this file).
if (local_cem == 14u) {
const uint a0 = ReplicateByteTo16(uvec4(endpoints0[local_partition].x)).x;
const uint a1 = ReplicateByteTo16(uvec4(endpoints1[local_partition].x)).x;
const uint Ca = (a0 * (64u - weight_vec.x) + a1 * weight_vec.x + 32u) / 64u;
p.x = float(Ca) / 65535.0f;
}
} else {
const uvec4 C0 = ReplicateByteTo16(endpoints0[local_partition]);
const uvec4 C1 = ReplicateByteTo16(endpoints1[local_partition]);
const vec4 Cf =
vec4((C0 * (uvec4(64) - weight_vec) + C1 * weight_vec + uvec4(32)) / 64);
p = Cf / 65535.0f;
}
#ifdef VULKAN
imageStore(dest_image, coord + ivec3(i, j, 0), p.gbar); imageStore(dest_image, coord + ivec3(i, j, 0), p.gbar);
#else
imageStore(dest_image, coord + ivec3(i, j, 0), clamp(p, 0.0f, 1.0f).gbar);
#endif
} }
} }
} }
@@ -1,6 +1,3 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
@@ -8,11 +5,11 @@
#extension GL_ARB_shader_stencil_export : require #extension GL_ARB_shader_stencil_export : require
layout(binding = 0) uniform sampler2D depth_tex; layout(binding = 0) uniform sampler2D depth_tex;
layout(binding = 1) uniform usampler2D stencil_tex; layout(binding = 1) uniform isampler2D stencil_tex;
layout(location = 0) in vec2 texcoord; layout(location = 0) in vec2 texcoord;
void main() { void main() {
gl_FragDepth = textureLod(depth_tex, texcoord, 0).r; gl_FragDepth = textureLod(depth_tex, texcoord, 0).r;
gl_FragStencilRefARB = int(textureLod(stencil_tex, texcoord, 0).r); gl_FragStencilRefARB = textureLod(stencil_tex, texcoord, 0).r;
} }
@@ -39,55 +39,6 @@ constexpr std::array POLYGON_OFFSET_ENABLE_LUT = {
POLYGON, // Patches POLYGON, // Patches
}; };
constexpr std::array TOPOLOGY_CLASS_REPRESENTATIVE_LUT = {
Maxwell::PrimitiveTopology::Points, // Points
Maxwell::PrimitiveTopology::Lines, // Lines
Maxwell::PrimitiveTopology::LineLoop, // LineLoop
Maxwell::PrimitiveTopology::LineStrip, // LineStrip
Maxwell::PrimitiveTopology::Triangles, // Triangles
Maxwell::PrimitiveTopology::Triangles, // TriangleStrip
Maxwell::PrimitiveTopology::Triangles, // TriangleFan
Maxwell::PrimitiveTopology::Triangles, // Quads
Maxwell::PrimitiveTopology::Triangles, // QuadStrip
Maxwell::PrimitiveTopology::Triangles, // Polygon
Maxwell::PrimitiveTopology::LinesAdjacency, // LinesAdjacency
Maxwell::PrimitiveTopology::LinesAdjacency, // LineStripAdjacency
Maxwell::PrimitiveTopology::TrianglesAdjacency, // TrianglesAdjacency
Maxwell::PrimitiveTopology::TrianglesAdjacency, // TriangleStripAdjacency
Maxwell::PrimitiveTopology::Patches, // Patches
};
bool IsDualSourceBlendFactor(Maxwell::Blend::Factor factor) {
using F = Maxwell::Blend::Factor;
switch (factor) {
case F::Source1Color_D3D:
case F::OneMinusSource1Color_D3D:
case F::Source1Alpha_D3D:
case F::OneMinusSource1Alpha_D3D:
case F::Source1Color_GL:
case F::OneMinusSource1Color_GL:
case F::Source1Alpha_GL:
case F::OneMinusSource1Alpha_GL:
return true;
default:
return false;
}
}
bool ComputeAttachment0DualSourceBlend(const Maxwell& regs) {
if (!regs.blend.enable[0]) {
return false;
}
const auto uses_dual_source = [](const auto& blend) {
return IsDualSourceBlendFactor(blend.color_source) ||
IsDualSourceBlendFactor(blend.color_dest) ||
IsDualSourceBlendFactor(blend.alpha_source) ||
IsDualSourceBlendFactor(blend.alpha_dest);
};
return regs.blend_per_target_enabled ? uses_dual_source(regs.blend_per_target[0])
: uses_dual_source(regs.blend);
}
void RefreshXfbState(VideoCommon::TransformFeedbackState& state, const Maxwell& regs) { void RefreshXfbState(VideoCommon::TransformFeedbackState& state, const Maxwell& regs) {
std::ranges::transform(regs.transform_feedback.controls, state.layouts.begin(), std::ranges::transform(regs.transform_feedback.controls, state.layouts.begin(),
[](const auto& layout) { [](const auto& layout) {
@@ -111,7 +62,6 @@ void FixedPipelineState::Refresh(Tegra::Engines::Maxwell3D& maxwell3d, DynamicFe
extended_dynamic_state_2_logic_op.Assign(features.has_extended_dynamic_state_2_logic_op ? 1 : 0); extended_dynamic_state_2_logic_op.Assign(features.has_extended_dynamic_state_2_logic_op ? 1 : 0);
extended_dynamic_state_3_blend.Assign(features.has_extended_dynamic_state_3_blend ? 1 : 0); extended_dynamic_state_3_blend.Assign(features.has_extended_dynamic_state_3_blend ? 1 : 0);
extended_dynamic_state_3_enables.Assign(features.has_extended_dynamic_state_3_enables ? 1 : 0); extended_dynamic_state_3_enables.Assign(features.has_extended_dynamic_state_3_enables ? 1 : 0);
color_write_enable_dynamic.Assign(features.has_color_write_enable ? 1 : 0);
dynamic_vertex_input.Assign(features.has_dynamic_vertex_input ? 1 : 0); dynamic_vertex_input.Assign(features.has_dynamic_vertex_input ? 1 : 0);
xfb_enabled.Assign(regs.transform_feedback_enabled != 0); xfb_enabled.Assign(regs.transform_feedback_enabled != 0);
ndc_minus_one_to_one.Assign(regs.depth_mode == Maxwell::DepthMode::MinusOneToOne ? 1 : 0); ndc_minus_one_to_one.Assign(regs.depth_mode == Maxwell::DepthMode::MinusOneToOne ? 1 : 0);
@@ -121,13 +71,8 @@ void FixedPipelineState::Refresh(Tegra::Engines::Maxwell3D& maxwell3d, DynamicFe
tessellation_clockwise.Assign(regs.tessellation.params.output_primitives.Value() == tessellation_clockwise.Assign(regs.tessellation.params.output_primitives.Value() ==
Maxwell::Tessellation::OutputPrimitives::Triangles_CW); Maxwell::Tessellation::OutputPrimitives::Triangles_CW);
patch_control_points_minus_one.Assign(regs.patch_vertices - 1); patch_control_points_minus_one.Assign(regs.patch_vertices - 1);
const bool can_collapse_topology_class = topology.Assign(topology_);
features.has_extended_dynamic_state && features.has_extended_dynamic_state_2;
topology.Assign(can_collapse_topology_class
? TOPOLOGY_CLASS_REPRESENTATIVE_LUT[static_cast<size_t>(topology_)]
: topology_);
msaa_mode.Assign(regs.anti_alias_samples_mode); msaa_mode.Assign(regs.anti_alias_samples_mode);
attachment0_dual_source_blend.Assign(ComputeAttachment0DualSourceBlend(regs) ? 1 : 0);
raw2 = 0; raw2 = 0;
@@ -242,15 +187,6 @@ void FixedPipelineState::Refresh(Tegra::Engines::Maxwell3D& maxwell3d, DynamicFe
maxwell3d.dirty.flags[Dirty::Blending] = false; maxwell3d.dirty.flags[Dirty::Blending] = false;
for (size_t index = 0; index < attachments.size(); ++index) { for (size_t index = 0; index < attachments.size(); ++index) {
attachments[index].Refresh(regs, index); attachments[index].Refresh(regs, index);
auto& attachment = attachments[index];
if (color_write_enable_dynamic && attachment.mask_r == 0 &&
attachment.mask_g == 0 && attachment.mask_b == 0 &&
attachment.mask_a == 0) {
attachment.mask_r.Assign(1);
attachment.mask_g.Assign(1);
attachment.mask_b.Assign(1);
attachment.mask_a.Assign(1);
}
} }
} }
} }
@@ -31,7 +31,6 @@ struct DynamicFeatures {
bool has_dynamic_state3_logic_op_enable; bool has_dynamic_state3_logic_op_enable;
bool has_dynamic_state3_line_stipple_enable; bool has_dynamic_state3_line_stipple_enable;
bool has_dynamic_vertex_input; bool has_dynamic_vertex_input;
bool has_color_write_enable;
bool has_provoking_vertex; bool has_provoking_vertex;
bool has_provoking_vertex_first_mode; bool has_provoking_vertex_first_mode;
bool has_provoking_vertex_last_mode; bool has_provoking_vertex_last_mode;
@@ -209,9 +208,6 @@ struct FixedPipelineState {
BitField<12, 2, u32> tessellation_spacing; BitField<12, 2, u32> tessellation_spacing;
BitField<14, 1, u32> tessellation_clockwise; BitField<14, 1, u32> tessellation_clockwise;
BitField<15, 5, u32> patch_control_points_minus_one; BitField<15, 5, u32> patch_control_points_minus_one;
BitField<20, 1, u32> color_write_enable_dynamic;
BitField<21, 1, u32> attachment0_dual_source_blend;
BitField<24, 4, Maxwell::PrimitiveTopology> topology; BitField<24, 4, Maxwell::PrimitiveTopology> topology;
BitField<28, 4, Tegra::Texture::MsaaMode> msaa_mode; BitField<28, 4, Tegra::Texture::MsaaMode> msaa_mode;
@@ -223,10 +223,6 @@ FormatInfo SurfaceFormat(const Device& device, FormatType format_type, bool with
SURFACE_FORMAT_ELEM(VK_FORMAT_ETC2_R8G8B8_SRGB_BLOCK, 0, ETC2_RGB_SRGB) \ SURFACE_FORMAT_ELEM(VK_FORMAT_ETC2_R8G8B8_SRGB_BLOCK, 0, ETC2_RGB_SRGB) \
SURFACE_FORMAT_ELEM(VK_FORMAT_ETC2_R8G8B8A8_SRGB_BLOCK, 0, ETC2_RGBA_SRGB) \ SURFACE_FORMAT_ELEM(VK_FORMAT_ETC2_R8G8B8A8_SRGB_BLOCK, 0, ETC2_RGBA_SRGB) \
SURFACE_FORMAT_ELEM(VK_FORMAT_ETC2_R8G8B8A1_SRGB_BLOCK, 0, ETC2_RGB_PTA_SRGB) \ SURFACE_FORMAT_ELEM(VK_FORMAT_ETC2_R8G8B8A1_SRGB_BLOCK, 0, ETC2_RGB_PTA_SRGB) \
SURFACE_FORMAT_ELEM(VK_FORMAT_EAC_R11_UNORM_BLOCK, 0, EAC_R11_UNORM) \
SURFACE_FORMAT_ELEM(VK_FORMAT_EAC_R11_SNORM_BLOCK, 0, EAC_R11_SNORM) \
SURFACE_FORMAT_ELEM(VK_FORMAT_EAC_R11G11_UNORM_BLOCK, 0, EAC_R11G11_UNORM) \
SURFACE_FORMAT_ELEM(VK_FORMAT_EAC_R11G11_SNORM_BLOCK, 0, EAC_R11G11_SNORM) \
/* Depth formats */ \ /* Depth formats */ \
SURFACE_FORMAT_ELEM(VK_FORMAT_D32_SFLOAT, usage_attachable, D32_FLOAT) \ SURFACE_FORMAT_ELEM(VK_FORMAT_D32_SFLOAT, usage_attachable, D32_FLOAT) \
SURFACE_FORMAT_ELEM(VK_FORMAT_D16_UNORM, usage_attachable, D16_UNORM) \ SURFACE_FORMAT_ELEM(VK_FORMAT_D16_UNORM, usage_attachable, D16_UNORM) \
@@ -282,17 +278,7 @@ FormatInfo SurfaceFormat(const Device& device, FormatType format_type, bool with
} }
} else if (!device.IsOptimalEtc2Supported() && VideoCore::Surface::IsPixelFormatETC2(pixel_format)) { } else if (!device.IsOptimalEtc2Supported() && VideoCore::Surface::IsPixelFormatETC2(pixel_format)) {
// Transcode on hardware that doesn't support ETC2 natively // Transcode on hardware that doesn't support ETC2 natively
if (pixel_format == PixelFormat::EAC_R11_SNORM) { tuple.format = is_srgb ? VK_FORMAT_A8B8G8R8_SRGB_PACK32 : VK_FORMAT_A8B8G8R8_UNORM_PACK32;
tuple.format = VK_FORMAT_R8_SNORM;
} else if (pixel_format == PixelFormat::EAC_R11_UNORM) {
tuple.format = VK_FORMAT_R8_UNORM;
} else if (pixel_format == PixelFormat::EAC_R11G11_SNORM) {
tuple.format = VK_FORMAT_R8G8_SNORM;
} else if (pixel_format == PixelFormat::EAC_R11G11_UNORM) {
tuple.format = VK_FORMAT_R8G8_UNORM;
} else {
tuple.format = is_srgb ? VK_FORMAT_A8B8G8R8_SRGB_PACK32 : VK_FORMAT_A8B8G8R8_UNORM_PACK32;
}
} }
bool const attachable = (tuple.usage & usage_attachable) != 0; bool const attachable = (tuple.usage & usage_attachable) != 0;
bool const storage = (tuple.usage & usage_storage) != 0; bool const storage = (tuple.usage & usage_storage) != 0;
@@ -7,7 +7,6 @@
#pragma once #pragma once
#include <cstddef> #include <cstddef>
#include <optional>
#include <boost/container/small_vector.hpp> #include <boost/container/small_vector.hpp>
@@ -16,7 +15,6 @@
#include "shader_recompiler/shader_info.h" #include "shader_recompiler/shader_info.h"
#include "video_core/renderer_vulkan/vk_texture_cache.h" #include "video_core/renderer_vulkan/vk_texture_cache.h"
#include "video_core/renderer_vulkan/vk_update_descriptor.h" #include "video_core/renderer_vulkan/vk_update_descriptor.h"
#include "video_core/surface.h"
#include "video_core/texture_cache/types.h" #include "video_core/texture_cache/types.h"
#include "video_core/vulkan_common/vulkan_device.h" #include "video_core/vulkan_common/vulkan_device.h"
@@ -24,29 +22,6 @@ namespace Vulkan {
using Shader::Backend::SPIRV::NUM_TEXTURE_AND_IMAGE_SCALING_WORDS; using Shader::Backend::SPIRV::NUM_TEXTURE_AND_IMAGE_SCALING_WORDS;
[[nodiscard]] inline std::optional<PixelFormat> PixelFormatFromImageFormat(
Shader::ImageFormat format) {
switch (format) {
case Shader::ImageFormat::Typeless:
return std::nullopt;
case Shader::ImageFormat::R8_UINT:
return PixelFormat::R8_UINT;
case Shader::ImageFormat::R8_SINT:
return PixelFormat::R8_SINT;
case Shader::ImageFormat::R16_UINT:
return PixelFormat::R16_UINT;
case Shader::ImageFormat::R16_SINT:
return PixelFormat::R16_SINT;
case Shader::ImageFormat::R32_UINT:
return PixelFormat::R32_UINT;
case Shader::ImageFormat::R32G32_UINT:
return PixelFormat::R32G32_UINT;
case Shader::ImageFormat::R32G32B32A32_UINT:
return PixelFormat::R32G32B32A32_UINT;
}
return std::nullopt;
}
[[nodiscard]] inline u32 NumDescriptorEntries(const Shader::Info& info) { [[nodiscard]] inline u32 NumDescriptorEntries(const Shader::Info& info) {
return Shader::NumDescriptors(info.constant_buffer_descriptors) + return Shader::NumDescriptors(info.constant_buffer_descriptors) +
Shader::NumDescriptors(info.storage_buffers_descriptors) + Shader::NumDescriptors(info.storage_buffers_descriptors) +
@@ -236,12 +211,8 @@ inline void PushImageDescriptors(TextureCache& texture_cache,
const Sampler& sampler{texture_cache.GetSampler(sampler_id)}; const Sampler& sampler{texture_cache.GetSampler(sampler_id)};
const bool use_fallback_sampler{sampler.HasAddedAnisotropy() && const bool use_fallback_sampler{sampler.HasAddedAnisotropy() &&
!image_view.SupportsAnisotropy()}; !image_view.SupportsAnisotropy()};
VkSampler vk_sampler{use_fallback_sampler ? sampler.HandleWithDefaultAnisotropy() const VkSampler vk_sampler{use_fallback_sampler ? sampler.HandleWithDefaultAnisotropy()
: sampler.Handle()}; : sampler.Handle()};
if (sampler.HasLinearFiltering() &&
VideoCore::Surface::IsPixelFormatInteger(image_view.format)) {
vk_sampler = sampler.HandleWithNearestFilter();
}
guest_descriptor_queue.AddSampledImage(vk_image_view, vk_sampler); guest_descriptor_queue.AddSampledImage(vk_image_view, vk_sampler);
const bool element_rescaled{texture_cache.IsRescaling(image_view)}; const bool element_rescaled{texture_cache.IsRescaling(image_view)};
is_rescaled |= element_rescaled; is_rescaled |= element_rescaled;
@@ -84,7 +84,7 @@ vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allo
} // Anonymous namespace } // Anonymous namespace
Buffer::Buffer(BufferCacheRuntime& runtime, VideoCommon::NullBufferParams null_params) Buffer::Buffer(BufferCacheRuntime& runtime, VideoCommon::NullBufferParams null_params)
: VideoCommon::BufferBase(null_params), scheduler{&runtime.scheduler}, tracker{4096} { : VideoCommon::BufferBase(null_params), tracker{4096} {
if (runtime.device.HasNullDescriptor()) { if (runtime.device.HasNullDescriptor()) {
return; return;
} }
@@ -95,18 +95,12 @@ Buffer::Buffer(BufferCacheRuntime& runtime, VideoCommon::NullBufferParams null_p
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_) Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_)
: VideoCommon::BufferBase(cpu_addr_, size_bytes_), device{&runtime.device}, : VideoCommon::BufferBase(cpu_addr_, size_bytes_), device{&runtime.device},
scheduler{&runtime.scheduler},
buffer{CreateBuffer(*device, runtime.memory_allocator, SizeBytes())}, tracker{SizeBytes()} { buffer{CreateBuffer(*device, runtime.memory_allocator, SizeBytes())}, tracker{SizeBytes()} {
if (runtime.device.HasDebuggingToolAttached()) { if (runtime.device.HasDebuggingToolAttached()) {
buffer.SetObjectNameEXT(fmt::format("Buffer 0x{:x}", CpuAddr()).c_str()); buffer.SetObjectNameEXT(fmt::format("Buffer 0x{:x}", CpuAddr()).c_str());
} }
} }
void Buffer::MarkUsage(u64 offset, u64 size) noexcept {
tracker.Track(offset, size);
last_usage_tick = scheduler->CurrentTick();
}
VkBufferView Buffer::View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format) { VkBufferView Buffer::View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format) {
if (!device) { if (!device) {
// Null buffer supported, return a null descriptor // Null buffer supported, return a null descriptor
@@ -390,9 +384,7 @@ u32 BufferCacheRuntime::GetStorageBufferAlignment() const {
void BufferCacheRuntime::TickFrame(Common::SlotVector<Buffer>& slot_buffers) noexcept { void BufferCacheRuntime::TickFrame(Common::SlotVector<Buffer>& slot_buffers) noexcept {
for (auto it = slot_buffers.begin(); it != slot_buffers.end(); it++) { for (auto it = slot_buffers.begin(); it != slot_buffers.end(); it++) {
if (scheduler.IsFree(it->LastUsageTick())) { it->ResetUsageTracking();
it->ResetUsageTracking();
}
} }
} }
@@ -564,7 +556,7 @@ void BufferCacheRuntime::BindVertexBuffer(u32 index, VkBuffer buffer, u32 offset
if (index >= device.GetMaxVertexInputBindings()) { if (index >= device.GetMaxVertexInputBindings()) {
return; return;
} }
if (device.IsExtExtendedDynamicStateSupported() && !vertex_input_dynamic_state_active) { if (device.IsExtExtendedDynamicStateSupported()) {
scheduler.Record([index, buffer, offset, size, stride](vk::CommandBuffer cmdbuf) { scheduler.Record([index, buffer, offset, size, stride](vk::CommandBuffer cmdbuf) {
const VkDeviceSize vk_offset = buffer != VK_NULL_HANDLE ? offset : 0; const VkDeviceSize vk_offset = buffer != VK_NULL_HANDLE ? offset : 0;
const VkDeviceSize vk_size = buffer != VK_NULL_HANDLE ? size : VK_WHOLE_SIZE; const VkDeviceSize vk_size = buffer != VK_NULL_HANDLE ? size : VK_WHOLE_SIZE;
@@ -604,7 +596,7 @@ void BufferCacheRuntime::BindVertexBuffers(VideoCommon::HostBindings<Buffer>& bi
if (binding_count == 0) { if (binding_count == 0) {
return; return;
} }
if (device.IsExtExtendedDynamicStateSupported() && !vertex_input_dynamic_state_active) { if (device.IsExtExtendedDynamicStateSupported()) {
scheduler.Record([bindings_ = std::move(bindings), buffer_handles_ = std::move(buffer_handles), binding_count](vk::CommandBuffer cmdbuf) { scheduler.Record([bindings_ = std::move(bindings), buffer_handles_ = std::move(buffer_handles), binding_count](vk::CommandBuffer cmdbuf) {
cmdbuf.BindVertexBuffers2EXT(bindings_.min_index, binding_count, buffer_handles_.data(), bindings_.offsets.data(), bindings_.sizes.data(), bindings_.strides.data()); cmdbuf.BindVertexBuffers2EXT(bindings_.min_index, binding_count, buffer_handles_.data(), bindings_.offsets.data(), bindings_.sizes.data(), bindings_.strides.data());
}); });
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project // SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later // SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2019 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2019 yuzu Emulator Project
@@ -43,16 +43,14 @@ public:
return tracker.IsUsed(offset, size); return tracker.IsUsed(offset, size);
} }
void MarkUsage(u64 offset, u64 size) noexcept; void MarkUsage(u64 offset, u64 size) noexcept {
tracker.Track(offset, size);
}
void ResetUsageTracking() noexcept { void ResetUsageTracking() noexcept {
tracker.Reset(); tracker.Reset();
} }
[[nodiscard]] u64 LastUsageTick() const noexcept {
return last_usage_tick;
}
operator VkBuffer() const noexcept { operator VkBuffer() const noexcept {
return *buffer; return *buffer;
} }
@@ -66,11 +64,9 @@ private:
}; };
const Device* device{}; const Device* device{};
Scheduler* scheduler{};
vk::Buffer buffer; vk::Buffer buffer;
std::vector<BufferView> views; std::vector<BufferView> views;
VideoCommon::UsageTracker tracker; VideoCommon::UsageTracker tracker;
u64 last_usage_tick{};
bool is_null{}; bool is_null{};
}; };
@@ -131,10 +127,6 @@ public:
void BindVertexBuffers(VideoCommon::HostBindings<Buffer>& bindings); void BindVertexBuffers(VideoCommon::HostBindings<Buffer>& bindings);
void SetVertexInputDynamicState(bool is_active) {
vertex_input_dynamic_state_active = is_active;
}
void BindTransformFeedbackBuffer(u32 index, VkBuffer buffer, u32 offset, u32 size); void BindTransformFeedbackBuffer(u32 index, VkBuffer buffer, u32 offset, u32 size);
void BindTransformFeedbackBuffers(VideoCommon::HostBindings<Buffer>& bindings); void BindTransformFeedbackBuffers(VideoCommon::HostBindings<Buffer>& bindings);
@@ -171,11 +163,7 @@ public:
private: private:
void BindBuffer(VkBuffer buffer, u32 offset, u32 size) { void BindBuffer(VkBuffer buffer, u32 offset, u32 size) {
if (buffer == VK_NULL_HANDLE) { guest_descriptor_queue.AddBuffer(buffer, offset, size);
guest_descriptor_queue.AddBuffer(buffer, 0, VK_WHOLE_SIZE);
} else {
guest_descriptor_queue.AddBuffer(buffer, offset, size);
}
} }
void ReserveNullBuffer(); void ReserveNullBuffer();
@@ -197,8 +185,6 @@ private:
bool limit_dynamic_storage_buffers = false; bool limit_dynamic_storage_buffers = false;
u32 max_dynamic_storage_buffers = (std::numeric_limits<u32>::max)(); u32 max_dynamic_storage_buffers = (std::numeric_limits<u32>::max)();
bool vertex_input_dynamic_state_active = false;
}; };
struct BufferCacheParams { struct BufferCacheParams {
@@ -26,7 +26,6 @@
#include "video_core/host_shaders/vulkan_uint8_comp_spv.h" #include "video_core/host_shaders/vulkan_uint8_comp_spv.h"
#include "video_core/host_shaders/block_linear_unswizzle_3d_bcn_comp_spv.h" #include "video_core/host_shaders/block_linear_unswizzle_3d_bcn_comp_spv.h"
#include "video_core/renderer_vulkan/vk_compute_pass.h" #include "video_core/renderer_vulkan/vk_compute_pass.h"
#include "video_core/surface.h"
#include "video_core/renderer_vulkan/vk_descriptor_pool.h" #include "video_core/renderer_vulkan/vk_descriptor_pool.h"
#include "video_core/renderer_vulkan/vk_scheduler.h" #include "video_core/renderer_vulkan/vk_scheduler.h"
#include "video_core/renderer_vulkan/vk_staging_buffer_pool.h" #include "video_core/renderer_vulkan/vk_staging_buffer_pool.h"
@@ -746,8 +745,10 @@ void BlockLinearUnswizzle3DPass::Unswizzle(
const u32 MAX_BATCH_SLICES = (std::min)(z_count, image.info.size.depth); const u32 MAX_BATCH_SLICES = (std::min)(z_count, image.info.size.depth);
// Allocate or grow to cover this batch's slice count if (!image.has_compute_unswizzle_buffer) {
image.AllocateComputeUnswizzleBuffer(MAX_BATCH_SLICES); // Allocate exactly what this batch needs
image.AllocateComputeUnswizzleBuffer(MAX_BATCH_SLICES);
}
ASSERT(swizzles.size() == 1); ASSERT(swizzles.size() == 1);
const auto& sw = swizzles[0]; const auto& sw = swizzles[0];
@@ -193,14 +193,8 @@ void ComputePipeline::Configure(Tegra::Engines::KeplerCompute& kepler_compute,
is_written = desc.is_written; is_written = desc.is_written;
} }
ImageView& image_view = texture_cache.GetImageView(views[index].id); ImageView& image_view = texture_cache.GetImageView(views[index].id);
PixelFormat format{image_view.format};
if constexpr (is_image) {
if (const auto explicit_format{PixelFormatFromImageFormat(desc.format)}) {
format = *explicit_format;
}
}
buffer_cache.BindComputeTextureBuffer(index, image_view.GpuAddr(), buffer_cache.BindComputeTextureBuffer(index, image_view.GpuAddr(),
image_view.BufferSize(), format, image_view.BufferSize(), image_view.format,
is_written, is_image); is_written, is_image);
++index; ++index;
} }
@@ -426,14 +426,8 @@ bool GraphicsPipeline::ConfigureImpl(bool is_indexed) {
is_written = desc.is_written; is_written = desc.is_written;
} }
ImageView& image_view{texture_cache.GetImageView(texture_buffer_it->id)}; ImageView& image_view{texture_cache.GetImageView(texture_buffer_it->id)};
PixelFormat format{image_view.format};
if constexpr (is_image) {
if (const auto explicit_format{PixelFormatFromImageFormat(desc.format)}) {
format = *explicit_format;
}
}
buffer_cache.BindGraphicsTextureBuffer(stage, index, image_view.GpuAddr(), buffer_cache.BindGraphicsTextureBuffer(stage, index, image_view.GpuAddr(),
image_view.BufferSize(), format, image_view.BufferSize(), image_view.format,
is_written, is_image); is_written, is_image);
++index; ++index;
++texture_buffer_it; ++texture_buffer_it;
@@ -478,7 +472,6 @@ bool GraphicsPipeline::ConfigureImpl(bool is_indexed) {
} }
buffer_cache.UpdateGraphicsBuffers(is_indexed); buffer_cache.UpdateGraphicsBuffers(is_indexed);
buffer_cache.runtime.SetVertexInputDynamicState(HasDynamicVertexInput());
buffer_cache.BindHostGeometryBuffers(is_indexed); buffer_cache.BindHostGeometryBuffers(is_indexed);
guest_descriptor_queue.Acquire(scheduler, num_descriptor_entries); guest_descriptor_queue.Acquire(scheduler, num_descriptor_entries);
@@ -872,17 +865,18 @@ void GraphicsPipeline::MakePipeline(VkRenderPass render_pass) {
VK_DYNAMIC_STATE_DEPTH_BOUNDS_TEST_ENABLE_EXT, VK_DYNAMIC_STATE_DEPTH_BOUNDS_TEST_ENABLE_EXT,
VK_DYNAMIC_STATE_STENCIL_TEST_ENABLE_EXT, VK_DYNAMIC_STATE_STENCIL_TEST_ENABLE_EXT,
VK_DYNAMIC_STATE_STENCIL_OP_EXT, VK_DYNAMIC_STATE_STENCIL_OP_EXT,
VK_DYNAMIC_STATE_PRIMITIVE_TOPOLOGY_EXT,
}; };
dynamic_states.insert(dynamic_states.end(), extended.begin(), extended.end()); dynamic_states.insert(dynamic_states.end(), extended.begin(), extended.end());
// VK_DYNAMIC_STATE_VERTEX_INPUT_BINDING_STRIDE_EXT // VK_DYNAMIC_STATE_VERTEX_INPUT_BINDING_STRIDE_EXT is part of EDS1
// Only use it if VIDS is not active (VIDS replaces it with full vertex input control)
if (!key.state.dynamic_vertex_input) { if (!key.state.dynamic_vertex_input) {
dynamic_states.push_back(VK_DYNAMIC_STATE_VERTEX_INPUT_BINDING_STRIDE_EXT); dynamic_states.push_back(VK_DYNAMIC_STATE_VERTEX_INPUT_BINDING_STRIDE_EXT);
} }
} }
// VK_DYNAMIC_STATE_VERTEX_INPUT_EXT // VK_DYNAMIC_STATE_VERTEX_INPUT_EXT (VIDS) - Independent from EDS
// Provides full dynamic vertex input control, replaces VERTEX_INPUT_BINDING_STRIDE
if (key.state.dynamic_vertex_input) { if (key.state.dynamic_vertex_input) {
dynamic_states.push_back(VK_DYNAMIC_STATE_VERTEX_INPUT_EXT); dynamic_states.push_back(VK_DYNAMIC_STATE_VERTEX_INPUT_EXT);
} }
@@ -912,11 +906,6 @@ void GraphicsPipeline::MakePipeline(VkRenderPass render_pass) {
dynamic_states.insert(dynamic_states.end(), extended3.begin(), extended3.end()); dynamic_states.insert(dynamic_states.end(), extended3.begin(), extended3.end());
} }
// VK_EXT_color_write_enable fallback for fully on/off render targets when EDS3 blending is not available.
if (!key.state.extended_dynamic_state_3_blend && key.state.color_write_enable_dynamic) {
dynamic_states.push_back(VK_DYNAMIC_STATE_COLOR_WRITE_ENABLE_EXT);
}
// EDS3 - Enables (composite: per-feature) // EDS3 - Enables (composite: per-feature)
if (key.state.extended_dynamic_state_3_enables) { if (key.state.extended_dynamic_state_3_enables) {
if (device.SupportsDynamicState3DepthClampEnable()) { if (device.SupportsDynamicState3DepthClampEnable()) {
@@ -130,70 +130,6 @@ VkResult MasterSemaphore::SubmitQueueTimeline(vk::CommandBuffer& cmdbuf,
VkSemaphore wait_semaphore, u64 host_tick) { VkSemaphore wait_semaphore, u64 host_tick) {
const VkSemaphore timeline_semaphore = *semaphore; const VkSemaphore timeline_semaphore = *semaphore;
if (device.HasSynchronization2()) {
const std::array<VkCommandBufferSubmitInfo, 2> cmdbuffer_infos{{
{
.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_SUBMIT_INFO,
.pNext = nullptr,
.commandBuffer = *upload_cmdbuf,
.deviceMask = 0,
},
{
.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_SUBMIT_INFO,
.pNext = nullptr,
.commandBuffer = *cmdbuf,
.deviceMask = 0,
},
}};
std::array<VkSemaphoreSubmitInfo, 2> signal_infos{{
{
.sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO,
.pNext = nullptr,
.semaphore = timeline_semaphore,
.value = host_tick,
.stageMask = VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT,
.deviceIndex = 0,
},
{},
}};
u32 num_signal_semaphores = 1;
if (signal_semaphore) {
signal_infos[1] = VkSemaphoreSubmitInfo{
.sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO,
.pNext = nullptr,
.semaphore = signal_semaphore,
.value = 0,
.stageMask = VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT,
.deviceIndex = 0,
};
num_signal_semaphores = 2;
}
const u32 num_wait_semaphores = wait_semaphore ? 1 : 0;
const VkSemaphoreSubmitInfo wait_info{
.sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO,
.pNext = nullptr,
.semaphore = wait_semaphore,
.value = 0,
.stageMask = static_cast<VkPipelineStageFlags2>(wait_stage_mask),
.deviceIndex = 0,
};
const VkSubmitInfo2 submit_info2{
.sType = VK_STRUCTURE_TYPE_SUBMIT_INFO_2,
.pNext = nullptr,
.flags = 0,
.waitSemaphoreInfoCount = num_wait_semaphores,
.pWaitSemaphoreInfos = num_wait_semaphores ? &wait_info : nullptr,
.commandBufferInfoCount = static_cast<u32>(cmdbuffer_infos.size()),
.pCommandBufferInfos = cmdbuffer_infos.data(),
.signalSemaphoreInfoCount = num_signal_semaphores,
.pSignalSemaphoreInfos = signal_infos.data(),
};
return device.GetGraphicsQueue().Submit2(submit_info2);
}
const u32 num_signal_semaphores = signal_semaphore ? 2 : 1; const u32 num_signal_semaphores = signal_semaphore ? 2 : 1;
const std::array signal_values{host_tick, u64(0)}; const std::array signal_values{host_tick, u64(0)};
const std::array signal_semaphores{timeline_semaphore, signal_semaphore}; const std::array signal_semaphores{timeline_semaphore, signal_semaphore};
@@ -236,66 +172,6 @@ VkResult MasterSemaphore::SubmitQueueFence(vk::CommandBuffer& cmdbuf,
vk::CommandBuffer& upload_cmdbuf, vk::CommandBuffer& upload_cmdbuf,
VkSemaphore signal_semaphore, VkSemaphore wait_semaphore, VkSemaphore signal_semaphore, VkSemaphore wait_semaphore,
u64 host_tick) { u64 host_tick) {
if (device.HasSynchronization2()) {
const std::array<VkCommandBufferSubmitInfo, 2> cmdbuffer_infos{{
{
.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_SUBMIT_INFO,
.pNext = nullptr,
.commandBuffer = *upload_cmdbuf,
.deviceMask = 0,
},
{
.sType = VK_STRUCTURE_TYPE_COMMAND_BUFFER_SUBMIT_INFO,
.pNext = nullptr,
.commandBuffer = *cmdbuf,
.deviceMask = 0,
},
}};
const u32 num_signal_semaphores = signal_semaphore ? 1 : 0;
const VkSemaphoreSubmitInfo signal_info{
.sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO,
.pNext = nullptr,
.semaphore = signal_semaphore,
.value = 0,
.stageMask = VK_PIPELINE_STAGE_2_ALL_COMMANDS_BIT,
.deviceIndex = 0,
};
const u32 num_wait_semaphores = wait_semaphore ? 1 : 0;
const VkSemaphoreSubmitInfo wait_info{
.sType = VK_STRUCTURE_TYPE_SEMAPHORE_SUBMIT_INFO,
.pNext = nullptr,
.semaphore = wait_semaphore,
.value = 0,
.stageMask = static_cast<VkPipelineStageFlags2>(wait_stage_mask),
.deviceIndex = 0,
};
const VkSubmitInfo2 submit_info2{
.sType = VK_STRUCTURE_TYPE_SUBMIT_INFO_2,
.pNext = nullptr,
.flags = 0,
.waitSemaphoreInfoCount = num_wait_semaphores,
.pWaitSemaphoreInfos = num_wait_semaphores ? &wait_info : nullptr,
.commandBufferInfoCount = static_cast<u32>(cmdbuffer_infos.size()),
.pCommandBufferInfos = cmdbuffer_infos.data(),
.signalSemaphoreInfoCount = num_signal_semaphores,
.pSignalSemaphoreInfos = num_signal_semaphores ? &signal_info : nullptr,
};
auto fence = GetFreeFence();
auto result = device.GetGraphicsQueue().Submit2(submit_info2, *fence);
if (result == VK_SUCCESS) {
std::scoped_lock lock{wait_mutex};
wait_queue.emplace(host_tick, std::move(fence));
wait_cv.notify_one();
}
return result;
}
const u32 num_signal_semaphores = signal_semaphore ? 1 : 0; const u32 num_signal_semaphores = signal_semaphore ? 1 : 0;
const u32 num_wait_semaphores = wait_semaphore ? 1 : 0; const u32 num_wait_semaphores = wait_semaphore ? 1 : 0;
@@ -246,7 +246,32 @@ Shader::RuntimeInfo MakeRuntimeInfo(std::span<const Shader::IR::Program> program
key.state.UnpackComparisonOp(key.state.alpha_test_func.Value())); key.state.UnpackComparisonOp(key.state.alpha_test_func.Value()));
info.alpha_test_reference = std::bit_cast<float>(key.state.alpha_test_ref); info.alpha_test_reference = std::bit_cast<float>(key.state.alpha_test_ref);
info.dual_source_blend = key.state.attachment0_dual_source_blend != 0; // Check for dual source blending
const auto& blend0 = key.state.attachments[0];
if (blend0.enable != 0) {
using F = Maxwell::Blend::Factor;
const auto src_rgb = blend0.SourceRGBFactor();
const auto dst_rgb = blend0.DestRGBFactor();
const auto src_a = blend0.SourceAlphaFactor();
const auto dst_a = blend0.DestAlphaFactor();
info.dual_source_blend =
src_rgb == F::Source1Color_D3D || src_rgb == F::OneMinusSource1Color_D3D ||
src_rgb == F::Source1Alpha_D3D || src_rgb == F::OneMinusSource1Alpha_D3D ||
src_rgb == F::Source1Color_GL || src_rgb == F::OneMinusSource1Color_GL ||
src_rgb == F::Source1Alpha_GL || src_rgb == F::OneMinusSource1Alpha_GL ||
dst_rgb == F::Source1Color_D3D || dst_rgb == F::OneMinusSource1Color_D3D ||
dst_rgb == F::Source1Alpha_D3D || dst_rgb == F::OneMinusSource1Alpha_D3D ||
dst_rgb == F::Source1Color_GL || dst_rgb == F::OneMinusSource1Color_GL ||
dst_rgb == F::Source1Alpha_GL || dst_rgb == F::OneMinusSource1Alpha_GL ||
src_a == F::Source1Color_D3D || src_a == F::OneMinusSource1Color_D3D ||
src_a == F::Source1Alpha_D3D || src_a == F::OneMinusSource1Alpha_D3D ||
src_a == F::Source1Color_GL || src_a == F::OneMinusSource1Color_GL ||
src_a == F::Source1Alpha_GL || src_a == F::OneMinusSource1Alpha_GL ||
dst_a == F::Source1Color_D3D || dst_a == F::OneMinusSource1Color_D3D ||
dst_a == F::Source1Alpha_D3D || dst_a == F::OneMinusSource1Alpha_D3D ||
dst_a == F::Source1Color_GL || dst_a == F::OneMinusSource1Color_GL ||
dst_a == F::Source1Alpha_GL || dst_a == F::OneMinusSource1Alpha_GL;
}
if (device.IsMoltenVK()) { if (device.IsMoltenVK()) {
for (size_t i = 0; i < 8; ++i) { for (size_t i = 0; i < 8; ++i) {
@@ -353,30 +378,12 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
serialization_thread(1, "VkPipelineSerialization") { serialization_thread(1, "VkPipelineSerialization") {
const auto& float_control{device.FloatControlProperties()}; const auto& float_control{device.FloatControlProperties()};
const VkDriverId driver_id{device.GetDriverID()}; const VkDriverId driver_id{device.GetDriverID()};
const VkShaderStageFlags subgroup_stages{device.GetSubgroupSupportedStages()};
const auto subgroup_stage_bit{[subgroup_stages](VkShaderStageFlags flag, Shader::Stage stage) {
return (subgroup_stages & flag) != 0 ? (1u << static_cast<u32>(stage)) : 0u;
}};
const u32 supported_subgroup_stages{
subgroup_stage_bit(VK_SHADER_STAGE_VERTEX_BIT, Shader::Stage::VertexA) |
subgroup_stage_bit(VK_SHADER_STAGE_VERTEX_BIT, Shader::Stage::VertexB) |
subgroup_stage_bit(VK_SHADER_STAGE_TESSELLATION_CONTROL_BIT,
Shader::Stage::TessellationControl) |
subgroup_stage_bit(VK_SHADER_STAGE_TESSELLATION_EVALUATION_BIT,
Shader::Stage::TessellationEval) |
subgroup_stage_bit(VK_SHADER_STAGE_GEOMETRY_BIT, Shader::Stage::Geometry) |
subgroup_stage_bit(VK_SHADER_STAGE_FRAGMENT_BIT, Shader::Stage::Fragment) |
subgroup_stage_bit(VK_SHADER_STAGE_COMPUTE_BIT, Shader::Stage::Compute)};
profile = Shader::Profile{ profile = Shader::Profile{
.supported_spirv = device.SupportedSpirvVersion(), .supported_spirv = device.SupportedSpirvVersion(),
.unified_descriptor_binding = true, .unified_descriptor_binding = true,
.support_descriptor_aliasing = device.IsDescriptorAliasingSupported(), .support_descriptor_aliasing = device.IsDescriptorAliasingSupported(),
.support_int8 = device.IsInt8Supported(), .support_int8 = device.IsInt8Supported(),
.support_uniform_and_storage_buffer_8bit =
device.IsUniformAndStorageBuffer8BitAccessSupported(),
.support_int16 = device.IsShaderInt16Supported(), .support_int16 = device.IsShaderInt16Supported(),
.support_uniform_and_storage_buffer_16bit =
device.IsUniformAndStorageBuffer16BitAccessSupported(),
.support_int64 = device.IsShaderInt64Supported(), .support_int64 = device.IsShaderInt64Supported(),
.support_vertex_instance_id = false, .support_vertex_instance_id = false,
.support_float_controls = device.IsKhrShaderFloatControlsSupported(), .support_float_controls = device.IsKhrShaderFloatControlsSupported(),
@@ -396,7 +403,6 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
float_control.shaderSignedZeroInfNanPreserveFloat64 != VK_FALSE, float_control.shaderSignedZeroInfNanPreserveFloat64 != VK_FALSE,
.support_explicit_workgroup_layout = device.IsKhrWorkgroupMemoryExplicitLayoutSupported(), .support_explicit_workgroup_layout = device.IsKhrWorkgroupMemoryExplicitLayoutSupported(),
.support_vote = device.IsSubgroupFeatureSupported(VK_SUBGROUP_FEATURE_VOTE_BIT), .support_vote = device.IsSubgroupFeatureSupported(VK_SUBGROUP_FEATURE_VOTE_BIT),
.supported_subgroup_stages = supported_subgroup_stages,
.support_viewport_index_layer_non_geometry = .support_viewport_index_layer_non_geometry =
device.IsExtShaderViewportIndexLayerSupported(), device.IsExtShaderViewportIndexLayerSupported(),
.support_viewport_mask = device.IsNvViewportArray2Supported(), .support_viewport_mask = device.IsNvViewportArray2Supported(),
@@ -494,8 +500,6 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
device.IsExtExtendedDynamicState3BlendingSupported(); device.IsExtExtendedDynamicState3BlendingSupported();
dynamic_features.has_extended_dynamic_state_3_enables = dynamic_features.has_extended_dynamic_state_3_enables =
device.IsExtExtendedDynamicState3EnablesSupported(); device.IsExtExtendedDynamicState3EnablesSupported();
dynamic_features.has_color_write_enable =
device.IsExtColorWriteEnableSupported();
dynamic_features.has_dynamic_state3_depth_clamp_enable = dynamic_features.has_dynamic_state3_depth_clamp_enable =
dynamic_features.has_extended_dynamic_state_3_enables && dynamic_features.has_extended_dynamic_state_3_enables &&
device.SupportsDynamicState3DepthClampEnable(); device.SupportsDynamicState3DepthClampEnable();
@@ -628,10 +632,7 @@ void PipelineCache::LoadDiskResources(u64 title_id, std::stop_token stop_loading
dynamic_features.has_extended_dynamic_state_3_blend || dynamic_features.has_extended_dynamic_state_3_blend ||
(key.state.extended_dynamic_state_3_enables != 0) != (key.state.extended_dynamic_state_3_enables != 0) !=
dynamic_features.has_extended_dynamic_state_3_enables || dynamic_features.has_extended_dynamic_state_3_enables ||
(key.state.color_write_enable_dynamic != 0) != (key.state.dynamic_vertex_input != 0) != dynamic_features.has_dynamic_vertex_input) {
dynamic_features.has_color_write_enable ||
(key.state.dynamic_vertex_input != 0) !=
dynamic_features.has_dynamic_vertex_input) {
return; return;
} }
@@ -41,8 +41,8 @@ class SamplesQueryBank : public VideoCommon::BankBase {
public: public:
static constexpr size_t BANK_SIZE = 256; static constexpr size_t BANK_SIZE = 256;
static constexpr size_t QUERY_SIZE = 8; static constexpr size_t QUERY_SIZE = 8;
explicit SamplesQueryBank(const Device& device_, Scheduler& scheduler_, size_t index_) explicit SamplesQueryBank(const Device& device_, size_t index_)
: BankBase(BANK_SIZE), device{device_}, scheduler{scheduler_}, index{index_} { : BankBase(BANK_SIZE), device{device_}, index{index_} {
const auto& dev = device.GetLogical(); const auto& dev = device.GetLogical();
query_pool = dev.CreateQueryPool({ query_pool = dev.CreateQueryPool({
.sType = VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO, .sType = VK_STRUCTURE_TYPE_QUERY_POOL_CREATE_INFO,
@@ -60,28 +60,12 @@ public:
void Reset() override { void Reset() override {
ASSERT(references == 0); ASSERT(references == 0);
VideoCommon::BankBase::Reset(); VideoCommon::BankBase::Reset();
if (device.IsHostQueryResetSupported()) { const auto& dev = device.GetLogical();
const auto& dev = device.GetLogical(); dev.ResetQueryPool(*query_pool, 0, BANK_SIZE);
dev.ResetQueryPool(*query_pool, 0, BANK_SIZE);
} else {
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([pool = *query_pool](vk::CommandBuffer cmdbuf) {
cmdbuf.ResetQueryPool(pool, 0, BANK_SIZE);
});
}
host_results.fill(0ULL); host_results.fill(0ULL);
next_bank = 0; next_bank = 0;
} }
void AddReference(size_t how_many = 1) {
BankBase::AddReference(how_many);
last_used_tick = scheduler.CurrentTick();
}
[[nodiscard]] bool IsDead() const {
return BankBase::IsDead() && scheduler.IsFree(last_used_tick);
}
void Sync(size_t start, size_t size) { void Sync(size_t start, size_t size) {
const auto& dev = device.GetLogical(); const auto& dev = device.GetLogical();
const VkResult query_result = dev.GetQueryResults( const VkResult query_result = dev.GetQueryResults(
@@ -114,11 +98,9 @@ public:
private: private:
const Device& device; const Device& device;
Scheduler& scheduler;
const size_t index; const size_t index;
vk::QueryPool query_pool; vk::QueryPool query_pool;
std::array<u64, BANK_SIZE> host_results; std::array<u64, BANK_SIZE> host_results;
u64 last_used_tick{};
}; };
using BaseStreamer = VideoCommon::SimpleStreamer<VideoCommon::HostQueryBase>; using BaseStreamer = VideoCommon::SimpleStreamer<VideoCommon::HostQueryBase>;
@@ -236,8 +218,7 @@ public:
} }
PauseCounter(); PauseCounter();
const auto driver_id = device.GetDriverID(); const auto driver_id = device.GetDriverID();
if (driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY || if (driver_id == VK_DRIVER_ID_ARM_PROPRIETARY || driver_id == VK_DRIVER_ID_MESA_TURNIP) {
driver_id == VK_DRIVER_ID_ARM_PROPRIETARY || driver_id == VK_DRIVER_ID_MESA_TURNIP) {
pending_sync.clear(); pending_sync.clear();
sync_values_stash.clear(); sync_values_stash.clear();
return; return;
@@ -450,7 +431,7 @@ private:
void ReserveBank() { void ReserveBank() {
current_bank_id = current_bank_id =
bank_pool.ReserveBank([this](std::deque<SamplesQueryBank>& queue, size_t index) { bank_pool.ReserveBank([this](std::deque<SamplesQueryBank>& queue, size_t index) {
queue.emplace_back(device, scheduler, index); queue.emplace_back(device, index);
}); });
if (current_bank) { if (current_bank) {
current_bank->next_bank = current_bank_id + 1; current_bank->next_bank = current_bank_id + 1;
@@ -639,15 +620,6 @@ public:
VideoCommon::BankBase::Reset(); VideoCommon::BankBase::Reset();
} }
void AddReference(size_t how_many = 1) {
BankBase::AddReference(how_many);
last_used_tick = scheduler.CurrentTick();
}
[[nodiscard]] bool IsDead() const {
return BankBase::IsDead() && scheduler.IsFree(last_used_tick);
}
void Sync(StagingBufferRef& stagging_buffer, size_t extra_offset, size_t start, size_t size) { void Sync(StagingBufferRef& stagging_buffer, size_t extra_offset, size_t start, size_t size) {
scheduler.RequestOutsideRenderPassOperationContext(); scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([this, dst_buffer = stagging_buffer.buffer, extra_offset, start, scheduler.Record([this, dst_buffer = stagging_buffer.buffer, extra_offset, start,
@@ -673,7 +645,6 @@ private:
Scheduler& scheduler; Scheduler& scheduler;
const size_t index; const size_t index;
vk::Buffer buffer; vk::Buffer buffer;
u64 last_used_tick{};
}; };
class PrimitivesSucceededStreamer; class PrimitivesSucceededStreamer;
@@ -951,7 +922,7 @@ private:
return; return;
} }
has_flushed_end_pending = false; has_flushed_end_pending = false;
// Refresh buffer state before ending transform feedback to ensure counters_count is up-to-date. // Refresh buffer state before ending transform feedback to ensure counters_count is up-to-date.
UpdateBuffers(); UpdateBuffers();
if (buffers_count == 0) { if (buffers_count == 0) {
@@ -1521,7 +1492,7 @@ bool QueryCacheRuntime::HostConditionalRenderingCompareValues(VideoCommon::Looku
auto driver_id = impl->device.GetDriverID(); auto driver_id = impl->device.GetDriverID();
const bool is_gpu_high = Settings::IsGPULevelHigh(); const bool is_gpu_high = Settings::IsGPULevelHigh();
if ((!is_gpu_high && driver_id == VK_DRIVER_ID_INTEL_PROPRIETARY_WINDOWS) || driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY || driver_id == VK_DRIVER_ID_ARM_PROPRIETARY || driver_id == VK_DRIVER_ID_MESA_TURNIP) { if ((!is_gpu_high && driver_id == VK_DRIVER_ID_INTEL_PROPRIETARY_WINDOWS) || driver_id == VK_DRIVER_ID_ARM_PROPRIETARY || driver_id == VK_DRIVER_ID_MESA_TURNIP) {
EndHostConditionalRendering(); EndHostConditionalRendering();
return true; return true;
} }
@@ -223,10 +223,7 @@ RasterizerVulkan::RasterizerVulkan(Core::Frontend::EmuWindow& emu_window_, Tegra
scheduler.SetQueryCache(query_cache); scheduler.SetQueryCache(query_cache);
} }
RasterizerVulkan::~RasterizerVulkan() { RasterizerVulkan::~RasterizerVulkan() = default;
scheduler.WaitWorker();
scheduler.Finish();
}
template <typename Func> template <typename Func>
void RasterizerVulkan::PrepareDraw(bool is_indexed, Func&& draw_func) { void RasterizerVulkan::PrepareDraw(bool is_indexed, Func&& draw_func) {
@@ -1014,12 +1011,12 @@ void RasterizerVulkan::UpdateDynamicStates() {
auto& regs = maxwell3d->regs; auto& regs = maxwell3d->regs;
auto& flags = maxwell3d->dirty.flags; auto& flags = maxwell3d->dirty.flags;
const auto topology = maxwell3d->draw_manager.draw_state.topology; const auto topology = maxwell3d->draw_manager.draw_state.topology;
const bool topology_changed = state_tracker.ChangePrimitiveTopology(topology); if (state_tracker.ChangePrimitiveTopology(topology)) {
if (topology_changed) {
flags[Dirty::DepthBiasEnable] = true; flags[Dirty::DepthBiasEnable] = true;
flags[Dirty::PrimitiveRestartEnable] = true; flags[Dirty::PrimitiveRestartEnable] = true;
} }
// Core Dynamic States (Vulkan 1.0) - Always active regardless of dyna_state setting
UpdateViewportsState(regs); UpdateViewportsState(regs);
UpdateScissorsState(regs); UpdateScissorsState(regs);
UpdateDepthBias(regs); UpdateDepthBias(regs);
@@ -1028,6 +1025,7 @@ void RasterizerVulkan::UpdateDynamicStates() {
UpdateStencilFaces(regs); UpdateStencilFaces(regs);
UpdateLineWidth(regs); UpdateLineWidth(regs);
// EDS1: CullMode, DepthCompare, FrontFace, StencilOp, DepthBoundsTest, DepthTest, DepthWrite, StencilTest
if (device.IsExtExtendedDynamicStateSupported()) { if (device.IsExtExtendedDynamicStateSupported()) {
UpdateCullMode(regs); UpdateCullMode(regs);
UpdateDepthCompareOp(regs); UpdateDepthCompareOp(regs);
@@ -1039,24 +1037,21 @@ void RasterizerVulkan::UpdateDynamicStates() {
UpdateDepthWriteEnable(regs); UpdateDepthWriteEnable(regs);
UpdateStencilTestEnable(regs); UpdateStencilTestEnable(regs);
} }
if (topology_changed) {
scheduler.Record([topology_vk = MaxwellToVK::PrimitiveTopology(device, topology)](
vk::CommandBuffer cmdbuf) {
cmdbuf.SetPrimitiveTopologyEXT(topology_vk);
});
}
} }
// EDS2: PrimitiveRestart, RasterizerDiscard, DepthBias enable/disable
if (device.IsExtExtendedDynamicState2Supported()) { if (device.IsExtExtendedDynamicState2Supported()) {
UpdatePrimitiveRestartEnable(regs); UpdatePrimitiveRestartEnable(regs);
UpdateRasterizerDiscardEnable(regs); UpdateRasterizerDiscardEnable(regs);
UpdateDepthBiasEnable(regs); UpdateDepthBiasEnable(regs);
} }
// EDS2 Extras: LogicOp operation selection
if (device.IsExtExtendedDynamicState2ExtrasSupported()) { if (device.IsExtExtendedDynamicState2ExtrasSupported()) {
UpdateLogicOp(regs); UpdateLogicOp(regs);
} }
// EDS3 Enables: LogicOpEnable, DepthClamp, LineStipple, ConservativeRaster
if (device.IsExtExtendedDynamicState3EnablesSupported()) { if (device.IsExtExtendedDynamicState3EnablesSupported()) {
using namespace Tegra::Engines; using namespace Tegra::Engines;
// AMD Workaround: LogicOp incompatible with float render targets // AMD Workaround: LogicOp incompatible with float render targets
@@ -1081,12 +1076,12 @@ void RasterizerVulkan::UpdateDynamicStates() {
UpdateAlphaToOneEnable(regs); UpdateAlphaToOneEnable(regs);
} }
// EDS3 Blending: ColorBlendEnable, ColorBlendEquation, ColorWriteMask
if (device.IsExtExtendedDynamicState3BlendingSupported()) { if (device.IsExtExtendedDynamicState3BlendingSupported()) {
UpdateBlending(regs); UpdateBlending(regs);
} else if (device.IsExtColorWriteEnableSupported()) {
UpdateColorWriteEnable(regs);
} }
// Vertex Input Dynamic State: Independent from EDS levels
if (device.IsExtVertexInputDynamicStateSupported()) { if (device.IsExtVertexInputDynamicStateSupported()) {
if (auto* gp = pipeline_cache.CurrentGraphicsPipeline(); gp && gp->HasDynamicVertexInput()) { if (auto* gp = pipeline_cache.CurrentGraphicsPipeline(); gp && gp->HasDynamicVertexInput()) {
UpdateVertexInput(regs); UpdateVertexInput(regs);
@@ -1099,6 +1094,7 @@ void RasterizerVulkan::HandleTransformFeedback() {
const auto& regs = maxwell3d->regs; const auto& regs = maxwell3d->regs;
if (!device.IsExtTransformFeedbackSupported()) { if (!device.IsExtTransformFeedbackSupported()) {
// If the guest enabled transform feedback, warn once that the device lacks support.
if (regs.transform_feedback_enabled != 0) { if (regs.transform_feedback_enabled != 0) {
std::call_once(warn_unsupported, [&] { std::call_once(warn_unsupported, [&] {
LOG_WARNING(Render_Vulkan, "Transform feedback requested by guest but VK_EXT_transform_feedback is unavailable; queries disabled"); LOG_WARNING(Render_Vulkan, "Transform feedback requested by guest but VK_EXT_transform_feedback is unavailable; queries disabled");
@@ -1727,16 +1723,9 @@ void RasterizerVulkan::UpdateBlending(Tegra::Engines::Maxwell3D::Regs& regs) {
if (state_tracker.TouchBlendEnable()) { if (state_tracker.TouchBlendEnable()) {
std::array<VkBool32, Maxwell::NumRenderTargets> setup_enables{}; std::array<VkBool32, Maxwell::NumRenderTargets> setup_enables{};
for (size_t index = 0; index < Maxwell::NumRenderTargets; index++) { std::ranges::transform(
bool is_integer = false; regs.blend.enable, setup_enables.begin(),
if (regs.rt[index].format != Tegra::RenderTargetFormat::NONE) { [&](const auto& is_enabled) { return is_enabled != 0 ? VK_TRUE : VK_FALSE; });
const auto format =
VideoCore::Surface::PixelFormatFromRenderTargetFormat(regs.rt[index].format);
is_integer = IsPixelFormatInteger(format);
}
setup_enables[index] =
(!is_integer && regs.blend.enable[index] != 0) ? VK_TRUE : VK_FALSE;
}
scheduler.Record([setup_enables](vk::CommandBuffer cmdbuf) { scheduler.Record([setup_enables](vk::CommandBuffer cmdbuf) {
cmdbuf.SetColorBlendEnableEXT(0, setup_enables); cmdbuf.SetColorBlendEnableEXT(0, setup_enables);
}); });
@@ -1785,20 +1774,6 @@ void RasterizerVulkan::UpdateBlending(Tegra::Engines::Maxwell3D::Regs& regs) {
} }
} }
void RasterizerVulkan::UpdateColorWriteEnable(Tegra::Engines::Maxwell3D::Regs& regs) {
if (!state_tracker.TouchColorMask()) {
return;
}
std::array<VkBool32, Maxwell::NumRenderTargets> setup_enables{};
for (size_t index = 0; index < Maxwell::NumRenderTargets; index++) {
const auto& mask = regs.color_mask[regs.color_mask_common ? 0 : index];
setup_enables[index] = (mask.R || mask.G || mask.B || mask.A) ? VK_TRUE : VK_FALSE;
}
scheduler.Record([setup_enables](vk::CommandBuffer cmdbuf) {
cmdbuf.SetColorWriteEnableEXT(setup_enables);
});
}
void RasterizerVulkan::UpdateStencilTestEnable(Tegra::Engines::Maxwell3D::Regs& regs) { void RasterizerVulkan::UpdateStencilTestEnable(Tegra::Engines::Maxwell3D::Regs& regs) {
if (!state_tracker.TouchStencilTestEnable()) { if (!state_tracker.TouchStencilTestEnable()) {
return; return;
@@ -191,7 +191,6 @@ private:
void UpdateStencilTestEnable(Tegra::Engines::Maxwell3D::Regs& regs); void UpdateStencilTestEnable(Tegra::Engines::Maxwell3D::Regs& regs);
void UpdateLogicOp(Tegra::Engines::Maxwell3D::Regs& regs); void UpdateLogicOp(Tegra::Engines::Maxwell3D::Regs& regs);
void UpdateBlending(Tegra::Engines::Maxwell3D::Regs& regs); void UpdateBlending(Tegra::Engines::Maxwell3D::Regs& regs);
void UpdateColorWriteEnable(Tegra::Engines::Maxwell3D::Regs& regs);
void UpdateVertexInput(Tegra::Engines::Maxwell3D::Regs& regs); void UpdateVertexInput(Tegra::Engines::Maxwell3D::Regs& regs);
@@ -344,9 +344,7 @@ void Scheduler::EndRenderPass()
Record([num_images = num_renderpass_images, Record([num_images = num_renderpass_images,
images = renderpass_images, images = renderpass_images,
ranges = renderpass_image_ranges, ranges = renderpass_image_ranges](vk::CommandBuffer cmdbuf) {
has_transform_feedback = device.IsExtTransformFeedbackSupported()](
vk::CommandBuffer cmdbuf) {
std::array<VkImageMemoryBarrier, 9> barriers; std::array<VkImageMemoryBarrier, 9> barriers;
for (size_t i = 0; i < num_images; ++i) { for (size_t i = 0; i < num_images; ++i) {
const VkImageSubresourceRange& range = ranges[i]; const VkImageSubresourceRange& range = ranges[i];
@@ -386,17 +384,6 @@ void Scheduler::EndRenderPass()
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT | cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT | VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT |
VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, vk::PIPELINE_STAGE_GRAPHICS_COMPUTE, VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, vk::PIPELINE_STAGE_GRAPHICS_COMPUTE,
0, nullptr, nullptr, vk::Span(barriers.data(), num_images)); 0, nullptr, nullptr, vk::Span(barriers.data(), num_images));
if (has_transform_feedback) {
static constexpr VkMemoryBarrier XFB_OUTPUT_BARRIER{
.sType = VK_STRUCTURE_TYPE_MEMORY_BARRIER,
.pNext = nullptr,
.srcAccessMask = VK_ACCESS_TRANSFORM_FEEDBACK_WRITE_BIT_EXT,
.dstAccessMask = VK_ACCESS_VERTEX_ATTRIBUTE_READ_BIT | VK_ACCESS_TRANSFER_READ_BIT,
};
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_TRANSFORM_FEEDBACK_BIT_EXT,
VK_PIPELINE_STAGE_VERTEX_INPUT_BIT | VK_PIPELINE_STAGE_TRANSFER_BIT,
0, XFB_OUTPUT_BARRIER);
}
}); });
state.renderpass = VkRenderPass{}; state.renderpass = VkRenderPass{};
@@ -6,7 +6,6 @@
#include <algorithm> #include <algorithm>
#include <array> #include <array>
#include <optional>
#include <span> #include <span>
#include <memory> #include <memory>
#include <vector> #include <vector>
@@ -129,32 +128,9 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
return usage; return usage;
} }
[[nodiscard]] bool WillUseAcceleratedAstcDecode(const Device& device, const ImageInfo& info) { [[nodiscard]] VkImageCreateInfo MakeImageCreateInfo(const Device& device, const ImageInfo& info) {
if (!IsPixelFormatASTC(info.format) || device.IsOptimalAstcSupported()) { const auto format_info =
return false;
}
if (Settings::values.accelerate_astc.GetValue() != Settings::AstcDecodeMode::Gpu) {
return false;
}
return Settings::values.astc_recompression.GetValue() ==
Settings::AstcRecompression::Uncompressed &&
info.size.depth == 1;
}
[[nodiscard]] bool WillUseWidenedAstcFormat(const Device& device, const ImageInfo& info) {
return WillUseAcceleratedAstcDecode(device, info) &&
!VideoCore::Surface::IsPixelFormatSRGB(info.format);
}
[[nodiscard]] VkImageCreateInfo MakeImageCreateInfo(const Device& device, const ImageInfo& info,
std::optional<VkFormat> format_override = {}) {
auto format_info =
MaxwellToVK::SurfaceFormat(device, FormatType::Optimal, false, info.format); MaxwellToVK::SurfaceFormat(device, FormatType::Optimal, false, info.format);
if (format_override) {
format_info.format = *format_override;
format_info.attachable = false;
format_info.storage = true;
}
VkImageCreateFlags flags{}; VkImageCreateFlags flags{};
if (info.type == ImageType::e2D && info.resources.layers >= 6 && if (info.type == ImageType::e2D && info.resources.layers >= 6 &&
info.size.width == info.size.height && !device.HasBrokenCubeImageCompatibility()) { info.size.width == info.size.height && !device.HasBrokenCubeImageCompatibility()) {
@@ -188,12 +164,11 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
} }
[[nodiscard]] vk::Image MakeImage(const Device& device, const MemoryAllocator& allocator, [[nodiscard]] vk::Image MakeImage(const Device& device, const MemoryAllocator& allocator,
const ImageInfo& info, std::span<const VkFormat> view_formats, const ImageInfo& info, std::span<const VkFormat> view_formats) {
std::optional<VkFormat> format_override = {}) {
if (info.type == ImageType::Buffer) { if (info.type == ImageType::Buffer) {
return vk::Image{}; return vk::Image{};
} }
VkImageCreateInfo image_ci = MakeImageCreateInfo(device, info, format_override); VkImageCreateInfo image_ci = MakeImageCreateInfo(device, info);
const VkImageFormatListCreateInfo image_format_list = { const VkImageFormatListCreateInfo image_format_list = {
.sType = VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO, .sType = VK_STRUCTURE_TYPE_IMAGE_FORMAT_LIST_CREATE_INFO,
.pNext = nullptr, .pNext = nullptr,
@@ -1324,9 +1299,7 @@ void TextureCacheRuntime::ConvertImage(Framebuffer* dst, ImageView& dst_view, Im
case PixelFormat::R32G32_FLOAT: case PixelFormat::R32G32_FLOAT:
case PixelFormat::R32G32_SINT: case PixelFormat::R32G32_SINT:
case PixelFormat::R32_FLOAT: case PixelFormat::R32_FLOAT:
if (src_view.format == PixelFormat::D32_FLOAT && if ((src_view.format == PixelFormat::D32_FLOAT) && Settings::values.fix_bloom_effects.GetValue()) {
(dst_view.format == PixelFormat::B5G6R5_UNORM ||
Settings::values.fix_bloom_effects.GetValue())) {
const Region2D region{ const Region2D region{
.start = {0, 0}, .start = {0, 0},
.end = {static_cast<s32>(dst->RenderArea().width), .end = {static_cast<s32>(dst->RenderArea().width),
@@ -1582,19 +1555,15 @@ void TextureCacheRuntime::TickFrame() {}
Image::Image(TextureCacheRuntime& runtime_, const ImageInfo& info_, GPUVAddr gpu_addr_, Image::Image(TextureCacheRuntime& runtime_, const ImageInfo& info_, GPUVAddr gpu_addr_,
VAddr cpu_addr_) VAddr cpu_addr_)
: VideoCommon::ImageBase(info_, gpu_addr_, cpu_addr_), scheduler{&runtime_.scheduler}, : VideoCommon::ImageBase(info_, gpu_addr_, cpu_addr_), scheduler{&runtime_.scheduler},
runtime{&runtime_}, runtime{&runtime_}, original_image(MakeImage(runtime_.device, runtime_.memory_allocator, info,
original_image(MakeImage(runtime_.device, runtime_.memory_allocator, info, runtime->ViewFormats(info.format))),
WillUseWidenedAstcFormat(runtime_.device, info)
? std::span<const VkFormat>{}
: runtime->ViewFormats(info.format),
WillUseWidenedAstcFormat(runtime_.device, info)
? std::make_optional(VK_FORMAT_R32G32B32A32_SFLOAT)
: std::nullopt)),
aspect_mask(ImageAspectMask(info.format)) { aspect_mask(ImageAspectMask(info.format)) {
if (IsPixelFormatASTC(info.format) && !runtime->device.IsOptimalAstcSupported()) { if (IsPixelFormatASTC(info.format) && !runtime->device.IsOptimalAstcSupported()) {
switch (Settings::values.accelerate_astc.GetValue()) { switch (Settings::values.accelerate_astc.GetValue()) {
case Settings::AstcDecodeMode::Gpu: case Settings::AstcDecodeMode::Gpu:
if (WillUseAcceleratedAstcDecode(runtime->device, info)) { if (Settings::values.astc_recompression.GetValue() ==
Settings::AstcRecompression::Uncompressed &&
info.size.depth == 1) {
flags |= VideoCommon::ImageFlagBits::AcceleratedUpload; flags |= VideoCommon::ImageFlagBits::AcceleratedUpload;
} }
break; break;
@@ -1620,12 +1589,9 @@ Image::Image(TextureCacheRuntime& runtime_, const ImageInfo& info_, GPUVAddr gpu
Settings::values.astc_recompression.GetValue() == Settings::values.astc_recompression.GetValue() ==
Settings::AstcRecompression::Uncompressed) { Settings::AstcRecompression::Uncompressed) {
const auto& device = runtime->device.GetLogical(); const auto& device = runtime->device.GetLogical();
const VkFormat storage_format = WillUseWidenedAstcFormat(runtime->device, info)
? VK_FORMAT_R32G32B32A32_SFLOAT
: VK_FORMAT_A8B8G8R8_UNORM_PACK32;
for (s32 level = 0; level < info.resources.levels; ++level) { for (s32 level = 0; level < info.resources.levels; ++level) {
storage_image_views[level] = storage_image_views[level] =
MakeStorageView(device, level, *original_image, storage_format); MakeStorageView(device, level, *original_image, VK_FORMAT_A8B8G8R8_UNORM_PACK32);
} }
} }
} }
@@ -1635,6 +1601,9 @@ Image::Image(const VideoCommon::NullImageParams& params) : VideoCommon::ImageBas
Image::~Image() = default; Image::~Image() = default;
void Image::AllocateComputeUnswizzleBuffer(u32 max_slices) { void Image::AllocateComputeUnswizzleBuffer(u32 max_slices) {
if (has_compute_unswizzle_buffer)
return;
using VideoCore::Surface::BytesPerBlock; using VideoCore::Surface::BytesPerBlock;
const u32 block_bytes = BytesPerBlock(info.format); // 8 for BC1, 16 for BC6H const u32 block_bytes = BytesPerBlock(info.format); // 8 for BC1, 16 for BC6H
@@ -1651,12 +1620,7 @@ void Image::AllocateComputeUnswizzleBuffer(u32 max_slices) {
static_cast<u64>(blocks_y) * static_cast<u64>(blocks_y) *
static_cast<u64>(blocks_z); static_cast<u64>(blocks_z);
const VkDeviceSize required_size = block_count * block_bytes; compute_unswizzle_buffer_size = block_count * block_bytes;
if (has_compute_unswizzle_buffer && required_size <= compute_unswizzle_buffer_size) {
return;
}
compute_unswizzle_buffer_size = required_size;
VkBufferCreateInfo ci{ VkBufferCreateInfo ci{
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO, .sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
@@ -1976,13 +1940,8 @@ void Image::DownloadMemory(const StagingBufferRef& map, std::span<const BufferIm
VkImageView Image::StorageImageView(s32 level) noexcept { VkImageView Image::StorageImageView(s32 level) noexcept {
auto& view = storage_image_views[level]; auto& view = storage_image_views[level];
if (!view) { if (!view) {
auto format_info = const auto format_info =
MaxwellToVK::SurfaceFormat(runtime->device, FormatType::Optimal, true, info.format); MaxwellToVK::SurfaceFormat(runtime->device, FormatType::Optimal, true, info.format);
if (WillUseAcceleratedAstcDecode(runtime->device, info)) {
format_info.format = WillUseWidenedAstcFormat(runtime->device, info)
? VK_FORMAT_R32G32B32A32_SFLOAT
: VK_FORMAT_A8B8G8R8_UNORM_PACK32;
}
view = MakeStorageView(runtime->device.GetLogical(), level, *(this->*current_image), view = MakeStorageView(runtime->device.GetLogical(), level, *(this->*current_image),
format_info.format); format_info.format);
} }
@@ -2149,11 +2108,7 @@ ImageView::ImageView(TextureCacheRuntime& runtime, const VideoCommon::ImageViewI
SanitizeDepthStencilSwizzle(swizzle, device->SupportsDepthStencilSwizzleOne()); SanitizeDepthStencilSwizzle(swizzle, device->SupportsDepthStencilSwizzleOne());
} }
} }
uses_widened_astc_format = WillUseWidenedAstcFormat(*device, image.info); const auto format_info = MaxwellToVK::SurfaceFormat(*device, FormatType::Optimal, true, format);
auto format_info = MaxwellToVK::SurfaceFormat(*device, FormatType::Optimal, true, format);
if (uses_widened_astc_format) {
format_info.format = VK_FORMAT_R32G32B32A32_SFLOAT;
}
const VkImageUsageFlags requested_view_usage = ImageUsageFlags(format_info, format); const VkImageUsageFlags requested_view_usage = ImageUsageFlags(format_info, format);
const VkImageUsageFlags image_usage = image.UsageFlags(); const VkImageUsageFlags image_usage = image.UsageFlags();
const VkImageUsageFlags clamped_view_usage = requested_view_usage & image_usage; const VkImageUsageFlags clamped_view_usage = requested_view_usage & image_usage;
@@ -2287,14 +2242,7 @@ VkImageView ImageView::StorageView(Shader::TextureType texture_type,
Shader::ImageFormat image_format) { Shader::ImageFormat image_format) {
if (image_handle) { if (image_handle) {
if (image_format == Shader::ImageFormat::Typeless) { if (image_format == Shader::ImageFormat::Typeless) {
if (!typeless_storage_view) { return Handle(texture_type);
auto info = MaxwellToVK::SurfaceFormat(*device, FormatType::Optimal, true, format);
if (uses_widened_astc_format) {
info.format = VK_FORMAT_R32G32B32A32_SFLOAT;
}
typeless_storage_view = MakeView(info.format, VK_IMAGE_ASPECT_COLOR_BIT, texture_type);
}
return *typeless_storage_view;
} }
const bool is_signed = image_format == Shader::ImageFormat::R8_SINT const bool is_signed = image_format == Shader::ImageFormat::R8_SINT
|| image_format == Shader::ImageFormat::R16_SINT; || image_format == Shader::ImageFormat::R16_SINT;
@@ -2303,7 +2251,7 @@ VkImageView ImageView::StorageView(Shader::TextureType texture_type,
auto& views{is_signed ? storage_views->signeds : storage_views->unsigneds}; auto& views{is_signed ? storage_views->signeds : storage_views->unsigneds};
auto& view{views[size_t(texture_type)]}; auto& view{views[size_t(texture_type)]};
if (!view) if (!view)
view = MakeView(Format(image_format), VK_IMAGE_ASPECT_COLOR_BIT, texture_type); view = MakeView(Format(image_format), VK_IMAGE_ASPECT_COLOR_BIT);
return *view; return *view;
} }
return VK_NULL_HANDLE; return VK_NULL_HANDLE;
@@ -2313,28 +2261,13 @@ bool ImageView::IsRescaled() const noexcept {
return (*slot_images)[image_id].IsRescaled(); return (*slot_images)[image_id].IsRescaled();
} }
vk::ImageView ImageView::MakeView(VkFormat vk_format, VkImageAspectFlags aspect_mask, vk::ImageView ImageView::MakeView(VkFormat vk_format, VkImageAspectFlags aspect_mask) {
std::optional<Shader::TextureType> texture_type) {
VkImageViewType view_type = ImageViewType(type);
VkImageSubresourceRange subresource_range = MakeSubresourceRange(aspect_mask, range);
if (texture_type) {
view_type = ImageViewType(*texture_type);
switch (view_type) {
case VK_IMAGE_VIEW_TYPE_1D_ARRAY:
case VK_IMAGE_VIEW_TYPE_2D_ARRAY:
case VK_IMAGE_VIEW_TYPE_CUBE_ARRAY:
break;
default:
subresource_range.layerCount = 1;
break;
}
}
return device->GetLogical().CreateImageView({ return device->GetLogical().CreateImageView({
.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO, .sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO,
.pNext = nullptr, .pNext = nullptr,
.flags = 0, .flags = 0,
.image = image_handle, .image = image_handle,
.viewType = view_type, .viewType = ImageViewType(type),
.format = vk_format, .format = vk_format,
.components{ .components{
.r = VK_COMPONENT_SWIZZLE_IDENTITY, .r = VK_COMPONENT_SWIZZLE_IDENTITY,
@@ -2342,7 +2275,7 @@ vk::ImageView ImageView::MakeView(VkFormat vk_format, VkImageAspectFlags aspect_
.b = VK_COMPONENT_SWIZZLE_IDENTITY, .b = VK_COMPONENT_SWIZZLE_IDENTITY,
.a = VK_COMPONENT_SWIZZLE_IDENTITY, .a = VK_COMPONENT_SWIZZLE_IDENTITY,
}, },
.subresourceRange = subresource_range, .subresourceRange = MakeSubresourceRange(aspect_mask, range),
}); });
} }
@@ -2364,15 +2297,12 @@ Sampler::Sampler(TextureCacheRuntime& runtime, const Tegra::Texture::TSCEntry& t
const void* pnext = nullptr; const void* pnext = nullptr;
if (has_custom_border_colors) { if (has_custom_border_colors) {
pnext = &border_ci; pnext = &border_ci;
// Log extension usage for custom border color
if (GPU::Logging::IsActive()) { if (GPU::Logging::IsActive()) {
GPU::Logging::GPULogger::GetInstance().LogExtensionUsage( GPU::Logging::GPULogger::GetInstance().LogExtensionUsage(
"VK_EXT_custom_border_color", "Sampler::Sampler"); "VK_EXT_custom_border_color", "Sampler::Sampler");
} }
} }
if (device.IsExtBorderColorSwizzleSupported() && GPU::Logging::IsActive()) {
GPU::Logging::GPULogger::GetInstance().LogExtensionUsage(
"VK_EXT_border_color_swizzle", "Sampler::Sampler");
}
const VkSamplerReductionModeCreateInfoEXT reduction_ci{ const VkSamplerReductionModeCreateInfoEXT reduction_ci{
.sType = VK_STRUCTURE_TYPE_SAMPLER_REDUCTION_MODE_CREATE_INFO_EXT, .sType = VK_STRUCTURE_TYPE_SAMPLER_REDUCTION_MODE_CREATE_INFO_EXT,
.pNext = pnext, .pNext = pnext,
@@ -2386,28 +2316,20 @@ Sampler::Sampler(TextureCacheRuntime& runtime, const Tegra::Texture::TSCEntry& t
// Some games have samplers with garbage. Sanitize them here. // Some games have samplers with garbage. Sanitize them here.
const f32 max_anisotropy = std::clamp(tsc.MaxAnisotropy(), 1.0f, 16.0f); const f32 max_anisotropy = std::clamp(tsc.MaxAnisotropy(), 1.0f, 16.0f);
const VkFilter mag_filter{MaxwellToVK::Sampler::Filter(tsc.mag_filter)}; const auto create_sampler = [&](const f32 anisotropy) {
const VkFilter min_filter{MaxwellToVK::Sampler::Filter(tsc.min_filter)};
const VkSamplerMipmapMode mipmap_mode{MaxwellToVK::Sampler::MipmapMode(tsc.mipmap_filter)};
const bool has_linear_filtering{mag_filter == VK_FILTER_LINEAR ||
min_filter == VK_FILTER_LINEAR ||
mipmap_mode == VK_SAMPLER_MIPMAP_MODE_LINEAR};
const auto create_sampler = [&](const f32 anisotropy, bool force_nearest) {
return device.GetLogical().CreateSampler(VkSamplerCreateInfo{ return device.GetLogical().CreateSampler(VkSamplerCreateInfo{
.sType = VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO, .sType = VK_STRUCTURE_TYPE_SAMPLER_CREATE_INFO,
.pNext = pnext, .pNext = pnext,
.flags = 0, .flags = 0,
.magFilter = force_nearest ? VK_FILTER_NEAREST : mag_filter, .magFilter = MaxwellToVK::Sampler::Filter(tsc.mag_filter),
.minFilter = force_nearest ? VK_FILTER_NEAREST : min_filter, .minFilter = MaxwellToVK::Sampler::Filter(tsc.min_filter),
.mipmapMode = force_nearest ? VK_SAMPLER_MIPMAP_MODE_NEAREST : mipmap_mode, .mipmapMode = MaxwellToVK::Sampler::MipmapMode(tsc.mipmap_filter),
.addressModeU = MaxwellToVK::Sampler::WrapMode(device, tsc.wrap_u, tsc.mag_filter), .addressModeU = MaxwellToVK::Sampler::WrapMode(device, tsc.wrap_u, tsc.mag_filter),
.addressModeV = MaxwellToVK::Sampler::WrapMode(device, tsc.wrap_v, tsc.mag_filter), .addressModeV = MaxwellToVK::Sampler::WrapMode(device, tsc.wrap_v, tsc.mag_filter),
.addressModeW = MaxwellToVK::Sampler::WrapMode(device, tsc.wrap_p, tsc.mag_filter), .addressModeW = MaxwellToVK::Sampler::WrapMode(device, tsc.wrap_p, tsc.mag_filter),
.mipLodBias = tsc.LodBias(), .mipLodBias = tsc.LodBias(),
.anisotropyEnable = .anisotropyEnable = static_cast<VkBool32>(anisotropy > 1.0f ? VK_TRUE : VK_FALSE),
static_cast<VkBool32>(!force_nearest && anisotropy > 1.0f ? VK_TRUE : VK_FALSE), .maxAnisotropy = anisotropy,
.maxAnisotropy = force_nearest ? 1.0f : anisotropy,
.compareEnable = tsc.depth_compare_enabled, .compareEnable = tsc.depth_compare_enabled,
.compareOp = MaxwellToVK::Sampler::DepthCompareFunction(tsc.depth_compare_func), .compareOp = MaxwellToVK::Sampler::DepthCompareFunction(tsc.depth_compare_func),
.minLod = tsc.mipmap_filter == TextureMipmapFilter::None ? 0.0f : tsc.MinLod(), .minLod = tsc.mipmap_filter == TextureMipmapFilter::None ? 0.0f : tsc.MinLod(),
@@ -2418,18 +2340,11 @@ Sampler::Sampler(TextureCacheRuntime& runtime, const Tegra::Texture::TSCEntry& t
}); });
}; };
sampler = create_sampler(max_anisotropy, false); sampler = create_sampler(max_anisotropy);
const f32 max_anisotropy_default = static_cast<f32>(1U << tsc.max_anisotropy); const f32 max_anisotropy_default = static_cast<f32>(1U << tsc.max_anisotropy);
if (max_anisotropy > max_anisotropy_default) { if (max_anisotropy > max_anisotropy_default) {
sampler_default_anisotropy = create_sampler(max_anisotropy_default, false); sampler_default_anisotropy = create_sampler(max_anisotropy_default);
}
if (has_linear_filtering) {
// Integer-format image views can never be linearly filtered
// (VUID-vkCmdDraw*-magFilter-04553); this sampler is cached purely from the guest's TSC,
// decoupled from whichever ImageView it ends up paired with, so build a nearest-forced
// fallback here for callers to swap to when the paired view turns out to be integer.
sampler_nearest = create_sampler(1.0f, true);
} }
} }
@@ -370,15 +370,13 @@ private:
std::array<vk::ImageView, Shader::NUM_TEXTURE_TYPES> unsigneds; std::array<vk::ImageView, Shader::NUM_TEXTURE_TYPES> unsigneds;
}; };
[[nodiscard]] vk::ImageView MakeView(VkFormat vk_format, VkImageAspectFlags aspect_mask, [[nodiscard]] vk::ImageView MakeView(VkFormat vk_format, VkImageAspectFlags aspect_mask);
std::optional<Shader::TextureType> texture_type = std::nullopt);
const Device* device = nullptr; const Device* device = nullptr;
const SlotVector<Image>* slot_images = nullptr; const SlotVector<Image>* slot_images = nullptr;
std::array<vk::ImageView, Shader::NUM_TEXTURE_TYPES> image_views; std::array<vk::ImageView, Shader::NUM_TEXTURE_TYPES> image_views;
std::optional<StorageViews> storage_views; std::optional<StorageViews> storage_views;
vk::ImageView typeless_storage_view;
vk::ImageView depth_view; vk::ImageView depth_view;
vk::ImageView stencil_view; vk::ImageView stencil_view;
vk::ImageView color_view; vk::ImageView color_view;
@@ -387,8 +385,6 @@ private:
VkImageView render_target = VK_NULL_HANDLE; VkImageView render_target = VK_NULL_HANDLE;
VkSampleCountFlagBits samples = VK_SAMPLE_COUNT_1_BIT; VkSampleCountFlagBits samples = VK_SAMPLE_COUNT_1_BIT;
u32 buffer_size = 0; u32 buffer_size = 0;
bool uses_widened_astc_format = false;
}; };
class ImageAlloc : public VideoCommon::ImageAllocBase {}; class ImageAlloc : public VideoCommon::ImageAllocBase {};
@@ -409,18 +405,9 @@ public:
return static_cast<bool>(sampler_default_anisotropy); return static_cast<bool>(sampler_default_anisotropy);
} }
[[nodiscard]] VkSampler HandleWithNearestFilter() const noexcept {
return *sampler_nearest;
}
[[nodiscard]] bool HasLinearFiltering() const noexcept {
return static_cast<bool>(sampler_nearest);
}
private: private:
vk::Sampler sampler; vk::Sampler sampler;
vk::Sampler sampler_default_anisotropy; vk::Sampler sampler_default_anisotropy;
vk::Sampler sampler_nearest;
}; };
struct TextureCacheParams { struct TextureCacheParams {
-4
View File
@@ -344,10 +344,6 @@ bool IsPixelFormatETC2(PixelFormat format) {
case PixelFormat::ETC2_RGB_SRGB: case PixelFormat::ETC2_RGB_SRGB:
case PixelFormat::ETC2_RGBA_SRGB: case PixelFormat::ETC2_RGBA_SRGB:
case PixelFormat::ETC2_RGB_PTA_SRGB: case PixelFormat::ETC2_RGB_PTA_SRGB:
case PixelFormat::EAC_R11_UNORM:
case PixelFormat::EAC_R11_SNORM:
case PixelFormat::EAC_R11G11_UNORM:
case PixelFormat::EAC_R11G11_SNORM:
return true; return true;
default: default:
return false; return false;
-4
View File
@@ -119,10 +119,6 @@ namespace VideoCore::Surface {
PIXEL_FORMAT_ELEM(ETC2_RGB_SRGB, 4, 4, 64) \ PIXEL_FORMAT_ELEM(ETC2_RGB_SRGB, 4, 4, 64) \
PIXEL_FORMAT_ELEM(ETC2_RGBA_SRGB, 4, 4, 128) \ PIXEL_FORMAT_ELEM(ETC2_RGBA_SRGB, 4, 4, 128) \
PIXEL_FORMAT_ELEM(ETC2_RGB_PTA_SRGB, 4, 4, 64) \ PIXEL_FORMAT_ELEM(ETC2_RGB_PTA_SRGB, 4, 4, 64) \
PIXEL_FORMAT_ELEM(EAC_R11_UNORM, 4, 4, 64) \
PIXEL_FORMAT_ELEM(EAC_R11_SNORM, 4, 4, 64) \
PIXEL_FORMAT_ELEM(EAC_R11G11_UNORM, 4, 4, 128) \
PIXEL_FORMAT_ELEM(EAC_R11G11_SNORM, 4, 4, 128) \
/* Depth formats */ \ /* Depth formats */ \
PIXEL_FORMAT_ELEM(D32_FLOAT, 1, 1, 32) \ PIXEL_FORMAT_ELEM(D32_FLOAT, 1, 1, 32) \
PIXEL_FORMAT_ELEM(D16_UNORM, 1, 1, 16) \ PIXEL_FORMAT_ELEM(D16_UNORM, 1, 1, 16) \
@@ -204,15 +204,6 @@ PixelFormat PixelFormatFromTextureInfo(TextureFormat format, ComponentType red,
return PixelFormat::ETC2_RGB_PTA_SRGB; return PixelFormat::ETC2_RGB_PTA_SRGB;
case Hash(TextureFormat::ETC2_RGBA, UNORM, SRGB): case Hash(TextureFormat::ETC2_RGBA, UNORM, SRGB):
return PixelFormat::ETC2_RGBA_SRGB; return PixelFormat::ETC2_RGBA_SRGB;
/* EAC */
case Hash(TextureFormat::EAC, UNORM):
return PixelFormat::EAC_R11_UNORM;
case Hash(TextureFormat::EAC, SNORM):
return PixelFormat::EAC_R11_SNORM;
case Hash(TextureFormat::EACX2, UNORM):
return PixelFormat::EAC_R11G11_UNORM;
case Hash(TextureFormat::EACX2, SNORM):
return PixelFormat::EAC_R11G11_SNORM;
/* ASTC */ /* ASTC */
case Hash(TextureFormat::ASTC_2D_4X4, UNORM, LINEAR): case Hash(TextureFormat::ASTC_2D_4X4, UNORM, LINEAR):
return PixelFormat::ASTC_2D_4X4_UNORM; return PixelFormat::ASTC_2D_4X4_UNORM;
+10 -361
View File
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project // SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later // SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: 2016 The University of North Carolina at Chapel Hill // SPDX-FileCopyrightText: 2016 The University of North Carolina at Chapel Hill
@@ -1269,220 +1269,6 @@ static inline u32 Select2DPartition(s32 seed, s32 x, s32 y, s32 partitionCount,
return SelectPartition(seed, x, y, 0, partitionCount, smallBlock); return SelectPartition(seed, x, y, 0, partitionCount, smallBlock);
} }
static constexpr bool IsHDRColorEndpointMode(u32 cem) {
switch (cem) {
case 2:
case 3:
case 7:
case 11:
case 14:
case 15:
return true;
default:
return false;
}
}
// Sign-extends the low nbits of value (a 2's complement field packed into
// the bottom of an otherwise-unsigned integer), per C.2.15's HDR endpoint
// bitfield unpacking.
static constexpr s32 SignExtend(s32 value, u32 nbits) {
const s32 sign_bit = 1 << (nbits - 1);
return (value ^ sign_bit) - sign_bit;
}
struct HDREndpointRGB {
s32 r0, g0, b0;
s32 r1, g1, b1;
};
// HDR Endpoint Mode 7 (C.2.15): base RGB + scale factor.
static void DecodeHDREndpointMode7(u32 v0, u32 v1, u32 v2, u32 v3, s32& r0, s32& g0, s32& b0,
s32& r1, s32& g1, s32& b1) {
const u32 modeval = ((v0 & 0xC0) >> 6) | ((v1 & 0x80) >> 5) | ((v2 & 0x80) >> 4);
u32 majcomp;
u32 mode;
if ((modeval & 0xC) != 0xC) {
majcomp = modeval >> 2;
mode = modeval & 3;
} else if (modeval != 0xF) {
majcomp = modeval & 3;
mode = 4;
} else {
majcomp = 0;
mode = 5;
}
s32 red = static_cast<s32>(v0 & 0x3f);
s32 green = static_cast<s32>(v1 & 0x1f);
s32 blue = static_cast<s32>(v2 & 0x1f);
s32 scale = static_cast<s32>(v3 & 0x1f);
const u32 x0 = (v1 >> 6) & 1;
const u32 x1 = (v1 >> 5) & 1;
const u32 x2 = (v2 >> 6) & 1;
const u32 x3 = (v2 >> 5) & 1;
const u32 x4 = (v3 >> 7) & 1;
const u32 x5 = (v3 >> 6) & 1;
const u32 x6 = (v3 >> 5) & 1;
const u32 ohm = 1u << mode;
if (ohm & 0x30)
green |= static_cast<s32>(x0 << 6);
if (ohm & 0x3A)
green |= static_cast<s32>(x1 << 5);
if (ohm & 0x30)
blue |= static_cast<s32>(x2 << 6);
if (ohm & 0x3A)
blue |= static_cast<s32>(x3 << 5);
if (ohm & 0x3D)
scale |= static_cast<s32>(x6 << 5);
if (ohm & 0x2D)
scale |= static_cast<s32>(x5 << 6);
if (ohm & 0x04)
scale |= static_cast<s32>(x4 << 7);
if (ohm & 0x3B)
red |= static_cast<s32>(x4 << 6);
if (ohm & 0x04)
red |= static_cast<s32>(x3 << 6);
if (ohm & 0x10)
red |= static_cast<s32>(x5 << 7);
if (ohm & 0x0F)
red |= static_cast<s32>(x2 << 7);
if (ohm & 0x05)
red |= static_cast<s32>(x1 << 8);
if (ohm & 0x0A)
red |= static_cast<s32>(x0 << 8);
if (ohm & 0x05)
red |= static_cast<s32>(x0 << 9);
if (ohm & 0x02)
red |= static_cast<s32>(x6 << 9);
if (ohm & 0x01)
red |= static_cast<s32>(x3 << 10);
if (ohm & 0x02)
red |= static_cast<s32>(x5 << 10);
static constexpr int shamts[6] = {1, 1, 2, 3, 4, 5};
const s32 shamt = shamts[mode];
red <<= shamt;
green <<= shamt;
blue <<= shamt;
scale <<= shamt;
if (mode != 5) {
green = red - green;
blue = red - blue;
}
if (majcomp == 1)
std::swap(red, green);
if (majcomp == 2)
std::swap(red, blue);
r1 = std::clamp(red, 0, 0xFFF);
g1 = std::clamp(green, 0, 0xFFF);
b1 = std::clamp(blue, 0, 0xFFF);
r0 = std::clamp(red - scale, 0, 0xFFF);
g0 = std::clamp(green - scale, 0, 0xFFF);
b0 = std::clamp(blue - scale, 0, 0xFFF);
}
// HDR Endpoint Mode 11 (C.2.15): direct RGB pair. Shared by modes 11, 14 and 15,
// which all decode their RGB the same way and only differ in how alpha is filled in.
static HDREndpointRGB DecodeHDREndpointMode11(u32 v0, u32 v1, u32 v2, u32 v3, u32 v4, u32 v5) {
const u32 majcomp = ((v4 & 0x80) >> 7) | ((v5 & 0x80) >> 6);
if (majcomp == 3) {
HDREndpointRGB result;
result.r0 = static_cast<s32>(v0 << 4);
result.g0 = static_cast<s32>(v2 << 4);
result.b0 = static_cast<s32>((v4 & 0x7f) << 5);
result.r1 = static_cast<s32>(v1 << 4);
result.g1 = static_cast<s32>(v3 << 4);
result.b1 = static_cast<s32>((v5 & 0x7f) << 5);
return result;
}
const u32 mode = ((v1 & 0x80) >> 7) | ((v2 & 0x80) >> 6) | ((v3 & 0x80) >> 5);
s32 va = static_cast<s32>(v0 | ((v1 & 0x40) << 2));
s32 vb0 = static_cast<s32>(v2 & 0x3f);
s32 vb1 = static_cast<s32>(v3 & 0x3f);
s32 vc = static_cast<s32>(v1 & 0x3f);
s32 vd0 = static_cast<s32>(v4 & 0x7f);
s32 vd1 = static_cast<s32>(v5 & 0x7f);
static constexpr int dbitstab[8] = {7, 6, 7, 6, 5, 6, 5, 6};
vd0 = SignExtend(vd0, dbitstab[mode]);
vd1 = SignExtend(vd1, dbitstab[mode]);
const u32 x0 = (v2 >> 6) & 1;
const u32 x1 = (v3 >> 6) & 1;
const u32 x2 = (v4 >> 6) & 1;
const u32 x3 = (v5 >> 6) & 1;
const u32 x4 = (v4 >> 5) & 1;
const u32 x5 = (v5 >> 5) & 1;
const u32 ohm = 1u << mode;
if (ohm & 0xA4)
va |= static_cast<s32>(x0 << 9);
if (ohm & 0x08)
va |= static_cast<s32>(x2 << 9);
if (ohm & 0x50)
va |= static_cast<s32>(x4 << 9);
if (ohm & 0x50)
va |= static_cast<s32>(x5 << 10);
if (ohm & 0xA0)
va |= static_cast<s32>(x1 << 10);
if (ohm & 0xC0)
va |= static_cast<s32>(x2 << 11);
if (ohm & 0x04)
vc |= static_cast<s32>(x1 << 6);
if (ohm & 0xE8)
vc |= static_cast<s32>(x3 << 6);
if (ohm & 0x20)
vc |= static_cast<s32>(x2 << 7);
if (ohm & 0x5B)
vb0 |= static_cast<s32>(x0 << 6);
if (ohm & 0x5B)
vb1 |= static_cast<s32>(x1 << 6);
if (ohm & 0x12)
vb0 |= static_cast<s32>(x2 << 7);
if (ohm & 0x12)
vb1 |= static_cast<s32>(x3 << 7);
// NOTE: the published spec text says "modeval >> 1" here, but no "modeval" is defined
// in this decode (that name belongs to Mode 7's unrelated decode) -- substituting the
// "mode" computed just above reproduces exactly Table C.2.23's per-mode shift amounts
// (3,3,2,2,1,1,0,0 for modes 0..7), so this is a spec transcription error, not a real
// "modeval" this function forgot to compute.
const s32 shamt = (static_cast<s32>(mode) >> 1) ^ 3;
va <<= shamt;
vb0 <<= shamt;
vb1 <<= shamt;
vc <<= shamt;
vd0 <<= shamt;
vd1 <<= shamt;
HDREndpointRGB result;
result.r1 = std::clamp(va, 0, 0xFFF);
result.g1 = std::clamp(va - vb0, 0, 0xFFF);
result.b1 = std::clamp(va - vb1, 0, 0xFFF);
result.r0 = std::clamp(va - vc, 0, 0xFFF);
result.g0 = std::clamp(va - vb0 - vc - vd0, 0, 0xFFF);
result.b0 = std::clamp(va - vb1 - vc - vd1, 0, 0xFFF);
if (majcomp == 1) {
std::swap(result.r0, result.g0);
std::swap(result.r1, result.g1);
} else if (majcomp == 2) {
std::swap(result.r0, result.b0);
std::swap(result.r1, result.b1);
}
return result;
}
// Section C.2.14 // Section C.2.14
static void ComputeEndpoints(Pixel& ep1, Pixel& ep2, const u32*& colorValues, static void ComputeEndpoints(Pixel& ep1, Pixel& ep2, const u32*& colorValues,
u32 colorEndpointMode) { u32 colorEndpointMode) {
@@ -1596,85 +1382,8 @@ static void ComputeEndpoints(Pixel& ep1, Pixel& ep2, const u32*& colorValues,
ep2.ClampByte(); ep2.ClampByte();
} break; } break;
case 2: {
READ_UINT_VALUES(2)
u32 y0, y1;
if (v[1] >= v[0]) {
y0 = v[0] << 4;
y1 = v[1] << 4;
} else {
y0 = (v[1] << 4) + 8;
y1 = (v[0] << 4) - 8;
}
ep1 = Pixel(0x780, y0, y0, y0);
ep2 = Pixel(0x780, y1, y1, y1);
} break;
case 3: {
READ_UINT_VALUES(2)
u32 y0, d;
if (v[0] & 0x80) {
y0 = ((v[1] & 0xE0) << 4) | ((v[0] & 0x7F) << 2);
d = (v[1] & 0x1F) << 2;
} else {
y0 = ((v[1] & 0xF0) << 4) | ((v[0] & 0x7F) << 1);
d = (v[1] & 0x0F) << 1;
}
const u32 y1 = (std::min)(y0 + d, 0xFFFU);
ep1 = Pixel(0x780, y0, y0, y0);
ep2 = Pixel(0x780, y1, y1, y1);
} break;
case 7: {
READ_UINT_VALUES(4)
s32 r0, g0, b0, r1, g1, b1;
DecodeHDREndpointMode7(v[0], v[1], v[2], v[3], r0, g0, b0, r1, g1, b1);
ep1 = Pixel(0x780, r0, g0, b0);
ep2 = Pixel(0x780, r1, g1, b1);
} break;
case 11: {
READ_UINT_VALUES(6)
const HDREndpointRGB rgb = DecodeHDREndpointMode11(v[0], v[1], v[2], v[3], v[4], v[5]);
ep1 = Pixel(0x780, rgb.r0, rgb.g0, rgb.b0);
ep2 = Pixel(0x780, rgb.r1, rgb.g1, rgb.b1);
} break;
case 14: {
READ_UINT_VALUES(8)
const HDREndpointRGB rgb = DecodeHDREndpointMode11(v[0], v[1], v[2], v[3], v[4], v[5]);
// Only mode with LDR (8-bit UNORM)-interpreted alpha; left as-is (0-255).
ep1 = Pixel(v[6], rgb.r0, rgb.g0, rgb.b0);
ep2 = Pixel(v[7], rgb.r1, rgb.g1, rgb.b1);
} break;
case 15: {
READ_UINT_VALUES(8)
const HDREndpointRGB rgb = DecodeHDREndpointMode11(v[0], v[1], v[2], v[3], v[4], v[5]);
const u32 mode = ((v[6] >> 7) & 1) | ((v[7] >> 6) & 2);
s32 a6 = static_cast<s32>(v[6] & 0x7F);
s32 a7 = static_cast<s32>(v[7] & 0x7F);
s32 alpha0, alpha1;
if (mode == 3) {
alpha0 = a6 << 5;
alpha1 = a7 << 5;
} else {
a6 |= (a7 << (mode + 1)) & 0x780;
a7 &= (0x3F >> mode);
a7 ^= 0x20 >> mode;
a7 -= 0x20 >> mode;
a6 <<= (4 - mode);
a7 <<= (4 - mode);
a7 += a6;
alpha0 = a6;
alpha1 = std::clamp(a7, 0, 0xFFF);
}
ep1 = Pixel(alpha0, rgb.r0, rgb.g0, rgb.b0);
ep2 = Pixel(alpha1, rgb.r1, rgb.g1, rgb.b1);
} break;
default: default:
assert(false && "Unsupported color endpoint mode"); assert(false && "Unsupported color endpoint mode (is it HDR?)");
break; break;
} }
@@ -1705,39 +1414,6 @@ static void FillVoidExtentLDR(InputBitStream& strm, std::span<u32> outBuf, u32 b
} }
} }
static float HalfToFloat(u16 h) {
const u32 sign = static_cast<u32>(h & 0x8000) << 16;
u32 exp = (h & 0x7C00) >> 10;
u32 mant = h & 0x3FF;
u32 bits;
if (exp == 0) {
if (mant == 0) {
bits = sign;
} else {
s32 e = 127 - 15 + 1;
while ((mant & 0x400) == 0) {
mant <<= 1;
--e;
}
mant &= 0x3FF;
bits = sign | (static_cast<u32>(e) << 23) | (mant << 13);
}
} else if (exp == 0x1F) {
bits = sign | 0x7F800000 | (mant << 13);
} else {
bits = sign | ((exp - 15 + 127) << 23) | (mant << 13);
}
float result;
std::memcpy(&result, &bits, sizeof(result));
return result;
}
static u16 HalfToClampedByte(u16 half_bits) {
const float value = HalfToFloat(half_bits);
const float clamped = std::clamp(value, 0.0f, 1.0f);
return static_cast<u16>(clamped * 255.0f + 0.5f);
}
static void FillError(std::span<u32> outBuf, u32 blockWidth, u32 blockHeight) { static void FillError(std::span<u32> outBuf, u32 blockWidth, u32 blockHeight) {
for (u32 j = 0; j < blockHeight; j++) { for (u32 j = 0; j < blockHeight; j++) {
for (u32 i = 0; i < blockWidth; i++) { for (u32 i = 0; i < blockWidth; i++) {
@@ -1955,49 +1631,22 @@ static void DecompressBlock(std::span<const u8, 16> inBuf, const u32 blockWidth,
Pixel p; Pixel p;
for (u32 c = 0; c < 4; c++) { for (u32 c = 0; c < 4; c++) {
u32 C0 = endpoints[partition][0].Component(c); u32 C0 = endpoints[partition][0].Component(c);
C0 = ReplicateByteTo16(C0);
u32 C1 = endpoints[partition][1].Component(c); u32 C1 = endpoints[partition][1].Component(c);
C1 = ReplicateByteTo16(C1);
u32 plane = 0; u32 plane = 0;
if (weightParams.m_bDualPlane && (((planeIdx + 1) & 3) == c)) { if (weightParams.m_bDualPlane && (((planeIdx + 1) & 3) == c)) {
plane = 1; plane = 1;
} }
u32 weight = weights[plane][j * blockWidth + i]; u32 weight = weights[plane][j * blockWidth + i];
u32 C = (C0 * (64 - weight) + C1 * weight + 32) / 64;
// Mode 14 is RGB-HDR but keeps an LDR (8-bit UNORM)-interpreted alpha if (C == 65535) {
// (component 0 here, see Pixel::A()) -- the only HDR mode with this split. p.Component(c) = 255;
const bool is_hdr = IsHDRColorEndpointMode(colorEndpointMode[partition]) &&
!(colorEndpointMode[partition] == 14 && c == 0);
if (is_hdr) {
// Endpoints are raw 12-bit pseudo-logarithmic values; shift left 4 bits
// to become 16-bit before interpolating, per C.2.19.
C0 <<= 4;
C1 <<= 4;
const u32 C = (C0 * (64 - weight) + C1 * weight + 32) / 64;
const u32 E = (C & 0xF800) >> 11;
const u32 M = C & 0x7FF;
u32 Mt;
if (M < 512) {
Mt = 3 * M;
} else if (M >= 1536) {
Mt = 5 * M - 2048;
} else {
Mt = 4 * M - 512;
}
const u32 Cf = (E << 10) + (Mt >> 3);
// +Inf/NaN clamps to the largest finite FP16 value (0x7BFF).
const u16 half_bits = (Cf >= 0x7C00) ? u16{0x7BFF} : static_cast<u16>(Cf);
p.Component(c) = HalfToClampedByte(half_bits);
} else { } else {
C0 = ReplicateByteTo16(C0); double Cf = static_cast<double>(C);
C1 = ReplicateByteTo16(C1); p.Component(c) = static_cast<u16>(255.0 * (Cf / 65536.0) + 0.5);
const u32 C = (C0 * (64 - weight) + C1 * weight + 32) / 64;
if (C == 65535) {
p.Component(c) = 255;
} else {
double Cf = static_cast<double>(C);
p.Component(c) = static_cast<u16>(255.0 * (Cf / 65536.0) + 0.5);
}
} }
} }
+21 -46
View File
@@ -95,12 +95,6 @@ constexpr std::array VK_FORMAT_A4B4G4R4_UNORM_PACK16{
VK_FORMAT_UNDEFINED, VK_FORMAT_UNDEFINED,
}; };
constexpr std::array B10G11R11_UFLOAT_PACK32{
VK_FORMAT_R16G16B16A16_SFLOAT,
VK_FORMAT_A8B8G8R8_SRGB_PACK32,
VK_FORMAT_UNDEFINED,
};
} // namespace Alternatives } // namespace Alternatives
template <typename T> template <typename T>
@@ -133,8 +127,6 @@ constexpr const VkFormat* GetFormatAlternatives(VkFormat format) {
return Alternatives::VK_FORMAT_R32G32B32_SFLOAT.data(); return Alternatives::VK_FORMAT_R32G32B32_SFLOAT.data();
case VK_FORMAT_A4B4G4R4_UNORM_PACK16_EXT: case VK_FORMAT_A4B4G4R4_UNORM_PACK16_EXT:
return Alternatives::VK_FORMAT_A4B4G4R4_UNORM_PACK16.data(); return Alternatives::VK_FORMAT_A4B4G4R4_UNORM_PACK16.data();
case VK_FORMAT_B10G11R11_UFLOAT_PACK32:
return Alternatives::B10G11R11_UFLOAT_PACK32.data();
default: default:
return nullptr; return nullptr;
} }
@@ -300,10 +292,6 @@ ankerl::unordered_dense::map<VkFormat, VkFormatProperties> GetFormatProperties(v
VK_FORMAT_ETC2_R8G8B8_SRGB_BLOCK, VK_FORMAT_ETC2_R8G8B8_SRGB_BLOCK,
VK_FORMAT_ETC2_R8G8B8A8_SRGB_BLOCK, VK_FORMAT_ETC2_R8G8B8A8_SRGB_BLOCK,
VK_FORMAT_ETC2_R8G8B8A1_SRGB_BLOCK, VK_FORMAT_ETC2_R8G8B8A1_SRGB_BLOCK,
VK_FORMAT_EAC_R11_UNORM_BLOCK,
VK_FORMAT_EAC_R11_SNORM_BLOCK,
VK_FORMAT_EAC_R11G11_UNORM_BLOCK,
VK_FORMAT_EAC_R11G11_SNORM_BLOCK,
}; };
ankerl::unordered_dense::map<VkFormat, VkFormatProperties> format_properties; ankerl::unordered_dense::map<VkFormat, VkFormatProperties> format_properties;
for (const auto format : formats) { for (const auto format : formats) {
@@ -504,20 +492,18 @@ Device::Device(VkInstance instance_, vk::PhysicalDevice physical_, VkSurfaceKHR
CollectToolingInfo(); CollectToolingInfo();
if (is_qualcomm) { if (is_qualcomm) {
LOG_WARNING(Render_Vulkan, "Qualcomm drivers require scaled vertex format emulation"); LOG_WARNING(Render_Vulkan,
"Qualcomm drivers require scaled vertex format emulation");
must_emulate_scaled_formats = true; must_emulate_scaled_formats = true;
LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken custom border color."); LOG_WARNING(Render_Vulkan,
RemoveExtensionFeature(extensions.custom_border_color, features.custom_border_color, "Qualcomm drivers have broken provoking vertex");
VK_EXT_CUSTOM_BORDER_COLOR_EXTENSION_NAME); RemoveExtension(extensions.provoking_vertex, VK_EXT_PROVOKING_VERTEX_EXTENSION_NAME);
LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken border color swizzle."); LOG_WARNING(Render_Vulkan,
RemoveExtensionFeature(extensions.border_color_swizzle, features.border_color_swizzle, "Qualcomm drivers have slow push descriptor implementation");
VK_EXT_BORDER_COLOR_SWIZZLE_EXTENSION_NAME); RemoveExtension(extensions.push_descriptor, VK_KHR_PUSH_DESCRIPTOR_EXTENSION_NAME);
LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken color write enable."); LOG_WARNING(Render_Vulkan,
RemoveExtensionFeature(extensions.color_write_enable, features.color_write_enable, "Disabling shader float controls and 64-bit integer features on Qualcomm proprietary drivers");
VK_EXT_COLOR_WRITE_ENABLE_EXTENSION_NAME);
LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken shader float controls.");
RemoveExtension(extensions.shader_float_controls, VK_KHR_SHADER_FLOAT_CONTROLS_EXTENSION_NAME); RemoveExtension(extensions.shader_float_controls, VK_KHR_SHADER_FLOAT_CONTROLS_EXTENSION_NAME);
LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken shader atomic int64.");
RemoveExtensionFeature(extensions.shader_atomic_int64, features.shader_atomic_int64, RemoveExtensionFeature(extensions.shader_atomic_int64, features.shader_atomic_int64,
VK_KHR_SHADER_ATOMIC_INT64_EXTENSION_NAME); VK_KHR_SHADER_ATOMIC_INT64_EXTENSION_NAME);
features.shader_atomic_int64.shaderBufferInt64Atomics = false; features.shader_atomic_int64.shaderBufferInt64Atomics = false;
@@ -575,6 +561,7 @@ Device::Device(VkInstance instance_, vk::PhysicalDevice physical_, VkSurfaceKHR
features.shader_float16_int8.shaderFloat16 = false; features.shader_float16_int8.shaderFloat16 = false;
} }
// Mali/ NVIDIA proprietary drivers: Shader stencil export not supported
// Use hardware depth/stencil blits instead when available // Use hardware depth/stencil blits instead when available
if (!extensions.shader_stencil_export) { if (!extensions.shader_stencil_export) {
LOG_INFO(Render_Vulkan, LOG_INFO(Render_Vulkan,
@@ -670,6 +657,14 @@ Device::Device(VkInstance instance_, vk::PhysicalDevice physical_, VkSurfaceKHR
} }
const auto dyna_state = Settings::values.dyna_state.GetValue(); const auto dyna_state = Settings::values.dyna_state.GetValue();
// Base dynamic states (VIEWPORT, SCISSOR, DEPTH_BIAS, etc.) are ALWAYS active in vk_graphics_pipeline.cpp
// This slider controls EXTENDED dynamic states with accumulative levels per Vulkan specs:
// Level 0 = Core Dynamic States only (Vulkan 1.0)
// Level 1 = Core + VK_EXT_extended_dynamic_state
// Level 2 = Core + VK_EXT_extended_dynamic_state + VK_EXT_extended_dynamic_state2
// Level 3 = Core + VK_EXT_extended_dynamic_state + VK_EXT_extended_dynamic_state2 + VK_EXT_extended_dynamic_state3
switch (dyna_state) { switch (dyna_state) {
case Settings::ExtendedDynamicState::Disabled: case Settings::ExtendedDynamicState::Disabled:
// Level 0: Disable all extended dynamic state extensions // Level 0: Disable all extended dynamic state extensions
@@ -704,7 +699,8 @@ Device::Device(VkInstance instance_, vk::PhysicalDevice physical_, VkSurfaceKHR
break; break;
} }
// VK_EXT_vertex_input_dynamic_state // VK_EXT_vertex_input_dynamic_state is independent from EDS
// It can be enabled even without extended_dynamic_state
if (!Settings::values.vertex_input_dynamic_state.GetValue()) { if (!Settings::values.vertex_input_dynamic_state.GetValue()) {
RemoveExtensionFeature(extensions.vertex_input_dynamic_state, features.vertex_input_dynamic_state, VK_EXT_VERTEX_INPUT_DYNAMIC_STATE_EXTENSION_NAME); RemoveExtensionFeature(extensions.vertex_input_dynamic_state, features.vertex_input_dynamic_state, VK_EXT_VERTEX_INPUT_DYNAMIC_STATE_EXTENSION_NAME);
} }
@@ -1174,11 +1170,6 @@ bool Device::GetSuitability(bool requires_swapchain) {
} }
void Device::RemoveUnsuitableExtensions() { void Device::RemoveUnsuitableExtensions() {
// VK_EXT_color_write_enable
extensions.color_write_enable = features.color_write_enable.colorWriteEnable;
RemoveExtensionFeatureIfUnsuitable(extensions.color_write_enable, features.color_write_enable,
VK_EXT_COLOR_WRITE_ENABLE_EXTENSION_NAME);
// VK_EXT_custom_border_color // VK_EXT_custom_border_color
if (extensions.custom_border_color) { if (extensions.custom_border_color) {
extensions.custom_border_color = extensions.custom_border_color =
@@ -1188,17 +1179,6 @@ void Device::RemoveUnsuitableExtensions() {
RemoveExtensionFeatureIfUnsuitable(extensions.custom_border_color, features.custom_border_color, RemoveExtensionFeatureIfUnsuitable(extensions.custom_border_color, features.custom_border_color,
VK_EXT_CUSTOM_BORDER_COLOR_EXTENSION_NAME); VK_EXT_CUSTOM_BORDER_COLOR_EXTENSION_NAME);
// VK_EXT_border_color_swizzle
if (extensions.border_color_swizzle) {
extensions.border_color_swizzle =
extensions.custom_border_color &&
features.border_color_swizzle.borderColorSwizzle &&
features.border_color_swizzle.borderColorSwizzleFromImage;
}
RemoveExtensionFeatureIfUnsuitable(extensions.border_color_swizzle,
features.border_color_swizzle,
VK_EXT_BORDER_COLOR_SWIZZLE_EXTENSION_NAME);
// VK_EXT_depth_bias_control // VK_EXT_depth_bias_control
extensions.depth_bias_control = extensions.depth_bias_control =
features.depth_bias_control.depthBiasControl && features.depth_bias_control.depthBiasControl &&
@@ -1395,11 +1375,6 @@ void Device::RemoveUnsuitableExtensions() {
// VK_KHR_maintenance8 // VK_KHR_maintenance8
extensions.maintenance8 = loaded_extensions.contains(VK_KHR_MAINTENANCE_8_EXTENSION_NAME); extensions.maintenance8 = loaded_extensions.contains(VK_KHR_MAINTENANCE_8_EXTENSION_NAME);
RemoveExtensionIfUnsuitable(extensions.maintenance8, VK_KHR_MAINTENANCE_8_EXTENSION_NAME); RemoveExtensionIfUnsuitable(extensions.maintenance8, VK_KHR_MAINTENANCE_8_EXTENSION_NAME);
// VK_KHR_synchronization2
extensions.synchronization2 = features.synchronization2.synchronization2;
RemoveExtensionFeatureIfUnsuitable(extensions.synchronization2, features.synchronization2,
VK_KHR_SYNCHRONIZATION_2_EXTENSION_NAME);
} }
void Device::SetupFamilies(VkSurfaceKHR surface) { void Device::SetupFamilies(VkSurfaceKHR surface) {
+5 -60
View File
@@ -43,15 +43,12 @@ VK_DEFINE_HANDLE(VmaAllocator)
FEATURE(EXT, ShaderDemoteToHelperInvocation, SHADER_DEMOTE_TO_HELPER_INVOCATION, \ FEATURE(EXT, ShaderDemoteToHelperInvocation, SHADER_DEMOTE_TO_HELPER_INVOCATION, \
shader_demote_to_helper_invocation) \ shader_demote_to_helper_invocation) \
FEATURE(EXT, SubgroupSizeControl, SUBGROUP_SIZE_CONTROL, subgroup_size_control) \ FEATURE(EXT, SubgroupSizeControl, SUBGROUP_SIZE_CONTROL, subgroup_size_control) \
FEATURE(KHR, Maintenance4, MAINTENANCE_4, maintenance4) \ FEATURE(KHR, Maintenance4, MAINTENANCE_4, maintenance4)
FEATURE(KHR, Synchronization2, SYNCHRONIZATION_2, synchronization2)
#define FOR_EACH_VK_FEATURE_1_4(FEATURE) #define FOR_EACH_VK_FEATURE_1_4(FEATURE)
// Define all features which may be used by the implementation and require an extension here. // Define all features which may be used by the implementation and require an extension here.
#define FOR_EACH_VK_FEATURE_EXT(FEATURE) \ #define FOR_EACH_VK_FEATURE_EXT(FEATURE) \
FEATURE(EXT, BorderColorSwizzle, BORDER_COLOR_SWIZZLE, border_color_swizzle) \
FEATURE(EXT, ColorWriteEnable, COLOR_WRITE_ENABLE, color_write_enable) \
FEATURE(EXT, CustomBorderColor, CUSTOM_BORDER_COLOR, custom_border_color) \ FEATURE(EXT, CustomBorderColor, CUSTOM_BORDER_COLOR, custom_border_color) \
FEATURE(EXT, DepthBiasControl, DEPTH_BIAS_CONTROL, depth_bias_control) \ FEATURE(EXT, DepthBiasControl, DEPTH_BIAS_CONTROL, depth_bias_control) \
FEATURE(EXT, DepthClipControl, DEPTH_CLIP_CONTROL, depth_clip_control) \ FEATURE(EXT, DepthClipControl, DEPTH_CLIP_CONTROL, depth_clip_control) \
@@ -72,9 +69,7 @@ VK_DEFINE_HANDLE(VmaAllocator)
FEATURE(KHR, PipelineExecutableProperties, PIPELINE_EXECUTABLE_PROPERTIES, \ FEATURE(KHR, PipelineExecutableProperties, PIPELINE_EXECUTABLE_PROPERTIES, \
pipeline_executable_properties) \ pipeline_executable_properties) \
FEATURE(KHR, WorkgroupMemoryExplicitLayout, WORKGROUP_MEMORY_EXPLICIT_LAYOUT, \ FEATURE(KHR, WorkgroupMemoryExplicitLayout, WORKGROUP_MEMORY_EXPLICIT_LAYOUT, \
workgroup_memory_explicit_layout) \ workgroup_memory_explicit_layout)
FEATURE(EXT, TextureCompressionASTCHDR, TEXTURE_COMPRESSION_ASTC_HDR, \
texture_compression_astc_hdr)
// Define miscellaneous extensions which may be used by the implementation here. // Define miscellaneous extensions which may be used by the implementation here.
@@ -185,7 +180,6 @@ VK_DEFINE_HANDLE(VmaAllocator)
FEATURE_NAME(robustness2, nullDescriptor) \ FEATURE_NAME(robustness2, nullDescriptor) \
FEATURE_NAME(shader_float16_int8, shaderFloat16) \ FEATURE_NAME(shader_float16_int8, shaderFloat16) \
FEATURE_NAME(shader_float16_int8, shaderInt8) \ FEATURE_NAME(shader_float16_int8, shaderInt8) \
FEATURE_NAME(synchronization2, synchronization2) \
FEATURE_NAME(timeline_semaphore, timelineSemaphore) \ FEATURE_NAME(timeline_semaphore, timelineSemaphore) \
FEATURE_NAME(transform_feedback, transformFeedback) \ FEATURE_NAME(transform_feedback, transformFeedback) \
FEATURE_NAME(uniform_buffer_standard_layout, uniformBufferStandardLayout) \ FEATURE_NAME(uniform_buffer_standard_layout, uniformBufferStandardLayout) \
@@ -350,10 +344,9 @@ FN_MAX_LIMIT_LIST
return properties.float_controls; return properties.float_controls;
} }
/// Returns true if ASTC is natively supported, including HDR-profile (non-LDR-only) /// Returns true if ASTC is natively supported.
bool IsOptimalAstcSupported() const { bool IsOptimalAstcSupported() const {
return features.features.textureCompressionASTC_LDR && return features.features.textureCompressionASTC_LDR;
features.texture_compression_astc_hdr.textureCompressionASTC_HDR;
} }
/// Returns true if BCn is natively supported. /// Returns true if BCn is natively supported.
@@ -390,26 +383,6 @@ FN_MAX_LIMIT_LIST
return features.shader_float16_int8.shaderInt8; return features.shader_float16_int8.shaderInt8;
} }
/// Returns true if the device allows 8-bit integer members in uniform/storage buffers.
bool IsUniformAndStorageBuffer8BitAccessSupported() const {
return features.bit8_storage.uniformAndStorageBuffer8BitAccess;
}
/// Returns true if the device allows 16-bit integer members in uniform/storage buffers.
bool IsUniformAndStorageBuffer16BitAccessSupported() const {
return features.bit16_storage.uniformAndStorageBuffer16BitAccess;
}
/// Returns true if the device supports reading 8-bit values from a storage buffer.
bool IsStorageBuffer8BitAccessSupported() const {
return features.bit8_storage.storageBuffer8BitAccess;
}
/// Returns true if the device supports reading 16-bit values from a storage buffer.
bool IsStorageBuffer16BitAccessSupported() const {
return features.bit16_storage.storageBuffer16BitAccess;
}
/// Returns true if the device supports binding multisample images as storage images. /// Returns true if the device supports binding multisample images as storage images.
bool IsStorageImageMultisampleSupported() const { bool IsStorageImageMultisampleSupported() const {
return features.features.shaderStorageImageMultisample; return features.features.shaderStorageImageMultisample;
@@ -430,10 +403,6 @@ FN_MAX_LIMIT_LIST
return properties.subgroup_properties.supportedOperations & feature; return properties.subgroup_properties.supportedOperations & feature;
} }
VkShaderStageFlags GetSubgroupSupportedStages() const {
return properties.subgroup_properties.supportedStages;
}
/// Returns the maximum number of push descriptors. /// Returns the maximum number of push descriptors.
u32 MaxPushDescriptors() const { u32 MaxPushDescriptors() const {
return properties.push_descriptor.maxPushDescriptors; return properties.push_descriptor.maxPushDescriptors;
@@ -540,6 +509,7 @@ FN_MAX_LIMIT_LIST
} }
/// Returns true if the device supports VK_EXT_shader_stencil_export. /// Returns true if the device supports VK_EXT_shader_stencil_export.
/// Note: Most Mali/NVIDIA drivers don't support this. Use hardware blits as fallback.
bool IsExtShaderStencilExportSupported() const { bool IsExtShaderStencilExportSupported() const {
return extensions.shader_stencil_export; return extensions.shader_stencil_export;
} }
@@ -576,11 +546,6 @@ FN_MAX_LIMIT_LIST
return extensions.subgroup_size_control; return extensions.subgroup_size_control;
} }
/// Returns true if vkResetQueryPool (host-side query reset) is supported.
bool IsHostQueryResetSupported() const {
return features.host_query_reset.hostQueryReset != VK_FALSE;
}
/// Returns true if the device supports VK_EXT_transform_feedback. /// Returns true if the device supports VK_EXT_transform_feedback.
bool IsExtTransformFeedbackSupported() const { bool IsExtTransformFeedbackSupported() const {
return extensions.transform_feedback; return extensions.transform_feedback;
@@ -617,21 +582,6 @@ FN_MAX_LIMIT_LIST
return features.custom_border_color.customBorderColorWithoutFormat; return features.custom_border_color.customBorderColorWithoutFormat;
} }
/// Returns true if the device supports VK_EXT_color_write_enable.
bool IsExtColorWriteEnableSupported() const {
return extensions.color_write_enable;
}
/// Returns true if the device supports VK_EXT_border_color_swizzle.
bool IsExtBorderColorSwizzleSupported() const {
return extensions.border_color_swizzle;
}
/// Returns true if borderColorSwizzleFromImage is available.
bool IsBorderColorSwizzleFromImageSupported() const {
return features.border_color_swizzle.borderColorSwizzleFromImage;
}
/// Returns true if the device supports VK_EXT_extended_dynamic_state. /// Returns true if the device supports VK_EXT_extended_dynamic_state.
bool IsExtExtendedDynamicStateSupported() const { bool IsExtExtendedDynamicStateSupported() const {
return extensions.extended_dynamic_state; return extensions.extended_dynamic_state;
@@ -772,11 +722,6 @@ FN_MAX_LIMIT_LIST
bool HasTimelineSemaphore() const; bool HasTimelineSemaphore() const;
/// Returns true if the device supports VK_KHR_synchronization2.
bool HasSynchronization2() const {
return extensions.synchronization2;
}
/// Returns the minimum supported version of SPIR-V. /// Returns the minimum supported version of SPIR-V.
u32 SupportedSpirvVersion() const { u32 SupportedSpirvVersion() const {
if (instance_version >= VK_API_VERSION_1_3) { if (instance_version >= VK_API_VERSION_1_3) {
@@ -123,7 +123,6 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
X(vkCmdEndDebugUtilsLabelEXT); X(vkCmdEndDebugUtilsLabelEXT);
X(vkCmdFillBuffer); X(vkCmdFillBuffer);
X(vkCmdPipelineBarrier); X(vkCmdPipelineBarrier);
X(vkCmdPipelineBarrier2);
X(vkCmdPushConstants); X(vkCmdPushConstants);
X(vkCmdPushDescriptorSetWithTemplateKHR); X(vkCmdPushDescriptorSetWithTemplateKHR);
X(vkCmdSetBlendConstants); X(vkCmdSetBlendConstants);
@@ -162,10 +161,8 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
X(vkCmdSetStencilTestEnableEXT); X(vkCmdSetStencilTestEnableEXT);
X(vkCmdSetVertexInputEXT); X(vkCmdSetVertexInputEXT);
X(vkCmdSetColorWriteMaskEXT); X(vkCmdSetColorWriteMaskEXT);
X(vkCmdSetColorWriteEnableEXT);
X(vkCmdSetColorBlendEnableEXT); X(vkCmdSetColorBlendEnableEXT);
X(vkCmdSetColorBlendEquationEXT); X(vkCmdSetColorBlendEquationEXT);
X(vkCmdResetQueryPool);
X(vkCmdResolveImage); X(vkCmdResolveImage);
X(vkCreateBuffer); X(vkCreateBuffer);
X(vkCreateBufferView); X(vkCreateBufferView);
@@ -229,7 +226,6 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
X(vkGetSemaphoreCounterValue); X(vkGetSemaphoreCounterValue);
X(vkMapMemory); X(vkMapMemory);
X(vkQueueSubmit); X(vkQueueSubmit);
X(vkQueueSubmit2);
X(vkResetFences); X(vkResetFences);
X(vkResetQueryPool); X(vkResetQueryPool);
X(vkSetDebugUtilsObjectNameEXT); X(vkSetDebugUtilsObjectNameEXT);
@@ -256,14 +252,6 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
Proc(dld.vkCmdDrawIndirectCount, dld, "vkCmdDrawIndirectCountKHR", device); Proc(dld.vkCmdDrawIndirectCount, dld, "vkCmdDrawIndirectCountKHR", device);
Proc(dld.vkCmdDrawIndexedIndirectCount, dld, "vkCmdDrawIndexedIndirectCountKHR", device); Proc(dld.vkCmdDrawIndexedIndirectCount, dld, "vkCmdDrawIndexedIndirectCountKHR", device);
} }
// Synchronization2 is core in Vulkan 1.3, otherwise requires VK_KHR_synchronization2
if (!dld.vkCmdPipelineBarrier2) {
Proc(dld.vkCmdPipelineBarrier2, dld, "vkCmdPipelineBarrier2KHR", device);
}
if (!dld.vkQueueSubmit2) {
Proc(dld.vkQueueSubmit2, dld, "vkQueueSubmit2KHR", device);
}
#undef X #undef X
} }
@@ -6,7 +6,6 @@
#pragma once #pragma once
#include <array>
#include <exception> #include <exception>
#include <limits> #include <limits>
#include <memory> #include <memory>
@@ -238,10 +237,8 @@ struct DeviceDispatch : InstanceDispatch {
PFN_vkCmdEndTransformFeedbackEXT vkCmdEndTransformFeedbackEXT{}; PFN_vkCmdEndTransformFeedbackEXT vkCmdEndTransformFeedbackEXT{};
PFN_vkCmdFillBuffer vkCmdFillBuffer{}; PFN_vkCmdFillBuffer vkCmdFillBuffer{};
PFN_vkCmdPipelineBarrier vkCmdPipelineBarrier{}; PFN_vkCmdPipelineBarrier vkCmdPipelineBarrier{};
PFN_vkCmdPipelineBarrier2 vkCmdPipelineBarrier2{};
PFN_vkCmdPushConstants vkCmdPushConstants{}; PFN_vkCmdPushConstants vkCmdPushConstants{};
PFN_vkCmdPushDescriptorSetWithTemplateKHR vkCmdPushDescriptorSetWithTemplateKHR{}; PFN_vkCmdPushDescriptorSetWithTemplateKHR vkCmdPushDescriptorSetWithTemplateKHR{};
PFN_vkCmdResetQueryPool vkCmdResetQueryPool{};
PFN_vkCmdResolveImage vkCmdResolveImage{}; PFN_vkCmdResolveImage vkCmdResolveImage{};
PFN_vkCmdSetBlendConstants vkCmdSetBlendConstants{}; PFN_vkCmdSetBlendConstants vkCmdSetBlendConstants{};
PFN_vkCmdSetCullModeEXT vkCmdSetCullModeEXT{}; PFN_vkCmdSetCullModeEXT vkCmdSetCullModeEXT{};
@@ -278,7 +275,6 @@ struct DeviceDispatch : InstanceDispatch {
PFN_vkCmdSetVertexInputEXT vkCmdSetVertexInputEXT{}; PFN_vkCmdSetVertexInputEXT vkCmdSetVertexInputEXT{};
PFN_vkCmdSetViewport vkCmdSetViewport{}; PFN_vkCmdSetViewport vkCmdSetViewport{};
PFN_vkCmdSetColorWriteMaskEXT vkCmdSetColorWriteMaskEXT{}; PFN_vkCmdSetColorWriteMaskEXT vkCmdSetColorWriteMaskEXT{};
PFN_vkCmdSetColorWriteEnableEXT vkCmdSetColorWriteEnableEXT{};
PFN_vkCmdSetColorBlendEnableEXT vkCmdSetColorBlendEnableEXT{}; PFN_vkCmdSetColorBlendEnableEXT vkCmdSetColorBlendEnableEXT{};
PFN_vkCmdSetColorBlendEquationEXT vkCmdSetColorBlendEquationEXT{}; PFN_vkCmdSetColorBlendEquationEXT vkCmdSetColorBlendEquationEXT{};
PFN_vkCmdWaitEvents vkCmdWaitEvents{}; PFN_vkCmdWaitEvents vkCmdWaitEvents{};
@@ -344,7 +340,6 @@ struct DeviceDispatch : InstanceDispatch {
PFN_vkGetSemaphoreCounterValue vkGetSemaphoreCounterValue{}; PFN_vkGetSemaphoreCounterValue vkGetSemaphoreCounterValue{};
PFN_vkMapMemory vkMapMemory{}; PFN_vkMapMemory vkMapMemory{};
PFN_vkQueueSubmit vkQueueSubmit{}; PFN_vkQueueSubmit vkQueueSubmit{};
PFN_vkQueueSubmit2 vkQueueSubmit2{};
PFN_vkResetFences vkResetFences{}; PFN_vkResetFences vkResetFences{};
PFN_vkResetQueryPool vkResetQueryPool{}; PFN_vkResetQueryPool vkResetQueryPool{};
PFN_vkSetDebugUtilsObjectNameEXT vkSetDebugUtilsObjectNameEXT{}; PFN_vkSetDebugUtilsObjectNameEXT vkSetDebugUtilsObjectNameEXT{};
@@ -824,13 +819,6 @@ public:
return dld->vkQueueSubmit(queue, submit_infos.size(), submit_infos.data(), fence); return dld->vkQueueSubmit(queue, submit_infos.size(), submit_infos.data(), fence);
} }
/// Submits using VK_KHR_synchronization2 / Vulkan 1.3 vkQueueSubmit2.
/// Only valid to call when the device dispatch table has vkQueueSubmit2 loaded.
VkResult Submit2(Span<VkSubmitInfo2> submit_infos,
VkFence fence = VK_NULL_HANDLE) const noexcept {
return dld->vkQueueSubmit2(queue, submit_infos.size(), submit_infos.data(), fence);
}
VkResult Present(const VkPresentInfoKHR& present_info) const noexcept { VkResult Present(const VkPresentInfoKHR& present_info) const noexcept {
return dld->vkQueuePresentKHR(queue, &present_info); return dld->vkQueuePresentKHR(queue, &present_info);
} }
@@ -1192,10 +1180,6 @@ public:
dld->vkCmdEndQuery(handle, query_pool, query); dld->vkCmdEndQuery(handle, query_pool, query);
} }
void ResetQueryPool(VkQueryPool query_pool, u32 first, u32 count) const noexcept {
dld->vkCmdResetQueryPool(handle, query_pool, first, count);
}
void BindDescriptorSets(VkPipelineBindPoint bind_point, VkPipelineLayout layout, u32 first, void BindDescriptorSets(VkPipelineBindPoint bind_point, VkPipelineLayout layout, u32 first,
Span<VkDescriptorSet> sets, Span<u32> dynamic_offsets) const noexcept { Span<VkDescriptorSet> sets, Span<u32> dynamic_offsets) const noexcept {
dld->vkCmdBindDescriptorSets(handle, bind_point, layout, first, sets.size(), sets.data(), dld->vkCmdBindDescriptorSets(handle, bind_point, layout, first, sets.size(), sets.data(),
@@ -1303,74 +1287,6 @@ public:
VkDependencyFlags dependency_flags, Span<VkMemoryBarrier> memory_barriers, VkDependencyFlags dependency_flags, Span<VkMemoryBarrier> memory_barriers,
Span<VkBufferMemoryBarrier> buffer_barriers, Span<VkBufferMemoryBarrier> buffer_barriers,
Span<VkImageMemoryBarrier> image_barriers) const noexcept { Span<VkImageMemoryBarrier> image_barriers) const noexcept {
// Legacy VkPipelineStageFlagBits/VkAccessFlagBits are bit-compatible with their
// Synchronization2 *2 counterparts, so barriers can be widened without a lookup table.
static constexpr u32 MaxBarriers = 16;
if (dld->vkCmdPipelineBarrier2 && memory_barriers.size() <= MaxBarriers &&
buffer_barriers.size() <= MaxBarriers && image_barriers.size() <= MaxBarriers) {
const auto src_stage_mask2 = static_cast<VkPipelineStageFlags2>(src_stage_mask);
const auto dst_stage_mask2 = static_cast<VkPipelineStageFlags2>(dst_stage_mask);
std::array<VkMemoryBarrier2, MaxBarriers> memory_barriers2;
for (u32 i = 0; i < memory_barriers.size(); ++i) {
memory_barriers2[i] = VkMemoryBarrier2{
.sType = VK_STRUCTURE_TYPE_MEMORY_BARRIER_2,
.pNext = nullptr,
.srcStageMask = src_stage_mask2,
.srcAccessMask = static_cast<VkAccessFlags2>(memory_barriers[i].srcAccessMask),
.dstStageMask = dst_stage_mask2,
.dstAccessMask = static_cast<VkAccessFlags2>(memory_barriers[i].dstAccessMask),
};
}
std::array<VkBufferMemoryBarrier2, MaxBarriers> buffer_barriers2;
for (u32 i = 0; i < buffer_barriers.size(); ++i) {
const auto& barrier = buffer_barriers[i];
buffer_barriers2[i] = VkBufferMemoryBarrier2{
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER_2,
.pNext = nullptr,
.srcStageMask = src_stage_mask2,
.srcAccessMask = static_cast<VkAccessFlags2>(barrier.srcAccessMask),
.dstStageMask = dst_stage_mask2,
.dstAccessMask = static_cast<VkAccessFlags2>(barrier.dstAccessMask),
.srcQueueFamilyIndex = barrier.srcQueueFamilyIndex,
.dstQueueFamilyIndex = barrier.dstQueueFamilyIndex,
.buffer = barrier.buffer,
.offset = barrier.offset,
.size = barrier.size,
};
}
std::array<VkImageMemoryBarrier2, MaxBarriers> image_barriers2;
for (u32 i = 0; i < image_barriers.size(); ++i) {
const auto& barrier = image_barriers[i];
image_barriers2[i] = VkImageMemoryBarrier2{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2,
.pNext = nullptr,
.srcStageMask = src_stage_mask2,
.srcAccessMask = static_cast<VkAccessFlags2>(barrier.srcAccessMask),
.dstStageMask = dst_stage_mask2,
.dstAccessMask = static_cast<VkAccessFlags2>(barrier.dstAccessMask),
.oldLayout = barrier.oldLayout,
.newLayout = barrier.newLayout,
.srcQueueFamilyIndex = barrier.srcQueueFamilyIndex,
.dstQueueFamilyIndex = barrier.dstQueueFamilyIndex,
.image = barrier.image,
.subresourceRange = barrier.subresourceRange,
};
}
const VkDependencyInfo dependency_info{
.sType = VK_STRUCTURE_TYPE_DEPENDENCY_INFO,
.pNext = nullptr,
.dependencyFlags = dependency_flags,
.memoryBarrierCount = memory_barriers.size(),
.pMemoryBarriers = memory_barriers2.data(),
.bufferMemoryBarrierCount = buffer_barriers.size(),
.pBufferMemoryBarriers = buffer_barriers2.data(),
.imageMemoryBarrierCount = image_barriers.size(),
.pImageMemoryBarriers = image_barriers2.data(),
};
dld->vkCmdPipelineBarrier2(handle, &dependency_info);
return;
}
dld->vkCmdPipelineBarrier(handle, src_stage_mask, dst_stage_mask, dependency_flags, dld->vkCmdPipelineBarrier(handle, src_stage_mask, dst_stage_mask, dependency_flags,
memory_barriers.size(), memory_barriers.data(), memory_barriers.size(), memory_barriers.data(),
buffer_barriers.size(), buffer_barriers.data(), buffer_barriers.size(), buffer_barriers.data(),
@@ -1595,10 +1511,6 @@ public:
dld->vkCmdSetColorWriteMaskEXT(handle, first, masks.size(), masks.data()); dld->vkCmdSetColorWriteMaskEXT(handle, first, masks.size(), masks.data());
} }
void SetColorWriteEnableEXT(Span<VkBool32> enables) const noexcept {
dld->vkCmdSetColorWriteEnableEXT(handle, enables.size(), enables.data());
}
void SetColorBlendEnableEXT(u32 first, Span<VkBool32> enables) const noexcept { void SetColorBlendEnableEXT(u32 first, Span<VkBool32> enables) const noexcept {
dld->vkCmdSetColorBlendEnableEXT(handle, first, enables.size(), enables.data()); dld->vkCmdSetColorBlendEnableEXT(handle, first, enables.size(), enables.data());
} }
-1
View File
@@ -371,7 +371,6 @@ int main(int argc, char** argv) {
#ifdef _WIN32 #ifdef _WIN32
Common::Windows::SetCurrentTimerResolutionToMaximum(); Common::Windows::SetCurrentTimerResolutionToMaximum();
system.CoreTiming().SetTimerResolutionNs(Common::Windows::GetCurrentTimerResolution());
#endif #endif
system.SetContentProvider(std::make_unique<FileSys::ContentProviderUnion>()); system.SetContentProvider(std::make_unique<FileSys::ContentProviderUnion>());