Compare commits

..

17 Commits

Author SHA1 Message Date
xbzk 6b36035a1b fix annoying ninja if constexpr warning 2026-02-15 04:01:13 -03:00
xbzk 56f0701b5a VUID-vkCmdPipelineBarrier-dstAccessMask-02816 2026-02-15 00:40:39 -03:00
xbzk 9ab3e9f482 VUID-vkCmdBeginRenderPass-initialLayout-00900 2026-02-15 00:40:39 -03:00
xbzk c372c88813 VUID-vkCmdDrawIndexedIndirect-imageLayout-00344 2026-02-15 00:40:39 -03:00
xbzk f46321da03 VK_FORMAT_R8G8B8A8_UNORM 2026-02-14 23:31:56 -03:00
xbzk 5fd780f721 layerCount_VK_REMAINING_ARRAY_LAYERS 2026-02-14 23:04:28 -03:00
xbzk d03ca7181a VUID-vkCmdBeginQuery-None-00807 2026-02-14 23:03:42 -03:00
xbzk a08114bd5d minTexelBufferOffsetAlignment 2026-02-14 23:03:42 -03:00
xbzk aefdd92834 VUID-VkWriteDescriptorSet-descriptorType-00339 2026-02-14 23:03:42 -03:00
xbzk 3249462561 VUID-VkWriteDescriptorSet-descriptorType-00336+00339 2026-02-14 23:03:42 -03:00
xbzk 60fd8380fb VUID-vkCmdDraw-None-09600 2026-02-14 23:03:20 -03:00
xbzk 068c272e27 VUID-VkImageViewCreateInfo-pNext-02662 2026-02-14 23:02:33 -03:00
xbzk 3ee14ed99a VUID-vkCmdDrawIndexed-format-07753 (FLOAT/SINT/UINT mistmatches) 2026-02-14 23:02:33 -03:00
xbzk 42451f0c3d versionwise workgroup vars 2026-02-14 23:02:33 -03:00
xbzk 1ffa381d74 vulkan vkQueue dispute fix 2026-02-14 23:02:33 -03:00
xbzk a72f444753 spirv logs cleanup 2 2026-02-14 23:02:33 -03:00
xbzk 2ba40a452f spirv log cleanup 1 2026-02-14 23:02:32 -03:00
46 changed files with 696 additions and 337 deletions
+1 -1
View File
@@ -37,7 +37,7 @@ set(GIT_DESC ${BUILD_VERSION})
# Auto-updater metadata! Must somewhat mirror GitHub API endpoint
set(BUILD_AUTO_UPDATE_WEBSITE "https://github.com")
set(BUILD_AUTO_UPDATE_API "https://api.github.com")
set(BUILD_AUTO_UPDATE_API "http://api.github.com")
if (NIGHTLY_BUILD)
set(BUILD_AUTO_UPDATE_REPO "Eden-CI/Nightly")
+4 -1
View File
@@ -6,4 +6,7 @@ Debugging on physical hardware can get tedious and time consuming. Users are emp
**Standard key prefix**: Allows to redirect the key manager to a file other than `prod.keys` (for example `other` would redirect to `other.keys`). This is useful for testing multiple keysets. Default is `prod`.
**Changing serial**: Very basic way to set debug values for the serial (and battery number). Developers do not need to write the full serial as only the first digits (excluding the last) will be accoutned for. Region settings will affect the generated serial. The serial corresponds to a non-OLED/Lite console.
**Changing serial**: Very basic way to set debug values for the serial (and battery number). Developers do not need to write the full serial as it will be writen in-place (that is, it will be filled with the default serial and then overwrite the serial from the beginning).
- Battery serial: `YUZU0EMULATOR14022024`
- Board serial: `YUZ10000000001`
If the user were to set their board serial as `ABC`, then it will be written in-place and the resulting serial would be `ABC10000000001`. There are no underlying checks to ensure correctness of serials other than a hard limit of 16-characters for both.
@@ -534,8 +534,8 @@
<item>@string/frame_pacing_mode_target_Auto</item>
<item>@string/frame_pacing_mode_target_30</item>
<item>@string/frame_pacing_mode_target_60</item>
<item>@string/frame_pacing_mode_target_90</item>
<item>@string/frame_pacing_mode_target_120</item>
<item>@string/frame_pacing_mode_target_240</item>
</string-array>
<integer-array name="framePacingModeValues">
<item>0</item>
@@ -1038,8 +1038,8 @@
<string name="frame_pacing_mode_target_Auto">Auto</string>
<string name="frame_pacing_mode_target_30">30 FPS</string>
<string name="frame_pacing_mode_target_60">60 FPS</string>
<string name="frame_pacing_mode_target_90">90 FPS</string>
<string name="frame_pacing_mode_target_120">120 FPS</string>
<string name="frame_pacing_mode_target_240">240 FPS</string>
<!-- ASTC Decoding Method Choices -->
<string name="accelerate_astc_cpu" translatable="false">CPU</string>
+2 -1
View File
@@ -25,7 +25,8 @@ void AssertFailSoftImpl();
#define ASSERT_MSG(_a_, ...) \
([&]() YUZU_NO_INLINE { \
if (!(_a_)) [[unlikely]] { \
auto&& _a_eval_ = (_a_); \
if (!(_a_eval_)) [[unlikely]] { \
LOG_CRITICAL(Debug, __FILE__ ": assert " __VA_ARGS__); \
AssertFailSoftImpl(); \
} \
+8 -9
View File
@@ -462,12 +462,9 @@ struct Values {
SwitchableSetting<FramePacingMode, true> frame_pacing_mode{linkage,
FramePacingMode::Target_Auto,
FramePacingMode::Target_Auto,
FramePacingMode::Target_120,
FramePacingMode::Target_240,
"frame_pacing_mode",
Category::RendererAdvanced,
Specialization::Default,
true,
true};
Category::RendererAdvanced};
SwitchableSetting<AstcRecompression, true> astc_recompression{linkage,
AstcRecompression::Uncompressed,
@@ -623,11 +620,11 @@ struct Values {
"language_index",
Category::System};
SwitchableSetting<Region, true> region_index{linkage, Region::Usa, "region_index", Category::System};
SwitchableSetting<TimeZone, true> time_zone_index{linkage, TimeZone::Auto, "time_zone_index", Category::System};
Setting<u32> serial_battery{linkage, 0, "serial_battery", Category::System};
Setting<u32> serial_unit{linkage, 0, "serial_unit", Category::System};
SwitchableSetting<TimeZone, true> time_zone_index{linkage, TimeZone::Auto,
"time_zone_index", Category::System};
// Measured in seconds since epoch
SwitchableSetting<bool> custom_rtc_enabled{linkage, false, "custom_rtc_enabled", Category::System, Specialization::Paired, true, true};
SwitchableSetting<bool> custom_rtc_enabled{
linkage, false, "custom_rtc_enabled", Category::System, Specialization::Paired, true, true};
SwitchableSetting<s64> custom_rtc{
linkage, 0, "custom_rtc", Category::System, Specialization::Time,
false, true, &custom_rtc_enabled};
@@ -797,6 +794,8 @@ struct Values {
true};
// Miscellaneous
Setting<std::string> serial_battery{linkage, std::string(), "serial_battery", Category::Miscellaneous};
Setting<std::string> serial_unit{linkage, std::string(), "serial_unit", Category::Miscellaneous};
Setting<std::string> log_filter{linkage, "*:Info", "log_filter", Category::Miscellaneous};
Setting<bool> log_flush_line{linkage, false, "flush_line", Category::Miscellaneous, Specialization::Default, true, true};
Setting<bool> censor_username{linkage, true, "censor_username", Category::Miscellaneous};
+1 -1
View File
@@ -129,7 +129,7 @@ ENUM(TimeZone, Auto, Default, Cet, Cst6Cdt, Cuba, Eet, Egypt, Eire, Est, Est5Edt
ENUM(AnisotropyMode, Automatic, Default, X2, X4, X8, X16, X32, X64, None);
ENUM(AstcDecodeMode, Cpu, Gpu, CpuAsynchronous);
ENUM(AstcRecompression, Uncompressed, Bc1, Bc3);
ENUM(FramePacingMode, Target_Auto, Target_30, Target_60, Target_90, Target_120);
ENUM(FramePacingMode, Target_Auto, Target_30, Target_60, Target_120, Target_240);
ENUM(VSyncMode, Immediate, Mailbox, Fifo, FifoRelaxed);
ENUM(VramUsageMode, Conservative, Aggressive);
ENUM(RendererBackend, OpenGL_GLSL, Vulkan, Null, OpenGL_GLASM, OpenGL_SPIRV);
+2 -44
View File
@@ -13,11 +13,6 @@
#include "common/windows/timer_resolution.h"
#endif
#if defined(_WIN32) && defined(ARCHITECTURE_x86_64) && defined(__MINGW64__)
#include "common/x64/cpu_detect.h"
#include "common/x64/rdtsc.h"
#endif
#include "common/settings.h"
#include "core/core_timing.h"
#include "core/hardware_properties.h"
@@ -287,45 +282,8 @@ void CoreTiming::ThreadLoop() {
const auto next_time = Advance();
if (next_time) {
// There are more events left in the queue, wait until the next event.
if (auto wait_time = *next_time - GetGlobalTimeNs().count(); wait_time > 0) {
#if defined(_WIN32) && defined(ARCHITECTURE_x86_64) && defined(__MINGW64__)
while (!paused && !event.IsSet() && wait_time > 0) {
wait_time = *next_time - GetGlobalTimeNs().count();
if (wait_time >= timer_resolution_ns) {
Common::Windows::SleepForOneTick();
} else {
// 100,000 cycles is a reasonable amount of time to wait to save on CPU resources.
// For reference:
// At 1 GHz, 100K cycles is 100us
// At 2 GHz, 100K cycles is 50us
// At 4 GHz, 100K cycles is 25us
constexpr auto PauseCycles = 100'000U;
auto const& caps = Common::GetCPUCaps();
if (caps.waitpkg) {
static constexpr auto RequestC02State = 0U;
const auto tsc = Common::X64::FencedRDTSC() + PauseCycles;
const auto eax = u32(tsc & 0xFFFFFFFF);
const auto edx = u32(tsc >> 32);
asm volatile("tpause %0" : : "r"(RequestC02State), "d"(edx), "a"(eax));
} else if (caps.monitorx) {
static constexpr auto EnableWaitTimeFlag = 1U << 1;
static constexpr auto RequestC1State = 0U;
// monitor_var should be aligned to a cache line.
alignas(64) u64 monitor_var{};
asm volatile("monitorx" : : "a"(&monitor_var), "c"(0), "d"(0));
asm volatile("mwaitx" : : "a"(RequestC1State), "b"(PauseCycles), "c"(EnableWaitTimeFlag));
} else {
std::this_thread::yield();
}
}
}
if (event.IsSet()) {
event.Reset();
}
#else
event.WaitFor(std::chrono::nanoseconds(wait_time));
#endif
}
auto wait_time = *next_time - GetGlobalTimeNs().count();
event.WaitFor(std::chrono::nanoseconds(wait_time));
} else {
// Queue is empty, wait until another event is scheduled and signals us to
// continue.
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project
@@ -980,82 +980,25 @@ Result ISystemSettingsServer::SetPrimaryAlbumStorage(PrimaryAlbumStorage primary
R_SUCCEED();
}
static void Fill3DS_CRC(u32 d, char* data) {
std::array<u8, 10> digits = {
u8((d / 1000000000) % 100),
u8((d / 100000000) % 10),
u8((d / 10000000) % 10),
u8((d / 1000000) % 10),
u8((d / 100000) % 10),
u8((d / 10000) % 10),
u8((d / 1000) % 10),
u8((d / 100) % 10),
u8((d / 10) % 10),
u8(d % 10),
};
// Normalize to retail values
std::array<u8, 4> retail_digits = { 1, 4, 5, 7 };
digits[0] = retail_digits[(d % 10) % 4];
digits[1] = 0;
//
for (size_t i = 0; i < sizeof(digits); ++i)
data[i] = char(digits[i] + '0');
u8 sum_odd = 0, sum_even = 0;
for (size_t i = 0; i < sizeof(digits); i += 2) {
sum_odd += digits[i + 0];
sum_even += digits[i + 1];
}
u8 sum_digit = u8(((sum_even * 3) + sum_odd) % 10);
if (sum_digit != 0)
sum_digit = 10 - sum_digit;
data[sizeof(digits)] = char(sum_digit + '0');
}
Result ISystemSettingsServer::GetBatteryLot(Out<BatteryLot> out_battery_lot) {
LOG_INFO(Service_SET, "called");
*out_battery_lot = []{
u32 d = ::Settings::values.serial_battery.GetValue();
BatteryLot c{};
c.lot_number[0] = 'B';
c.lot_number[1] = 'H';
c.lot_number[2] = 'A';
c.lot_number[3] = 'C';
// TODO: I have no fucking idea what the letters mean
c.lot_number[4] = 'H';
c.lot_number[5] = 'Z';
c.lot_number[6] = 'Z';
c.lot_number[7] = 'A';
c.lot_number[8] = 'D';
c.lot_number[9] = char(((d / 100000) % 26) + 'A');
Fill3DS_CRC(d, c.lot_number.data() + 10);
return c;
}();
*out_battery_lot = {"YUZU0EMULATOR14022024"};
if (auto const s = ::Settings::values.serial_battery.GetValue(); !s.empty()) {
auto const max_size = out_battery_lot->lot_number.size();
auto const end = s.size() > max_size ? s.begin() + max_size : s.end();
std::copy(s.begin(), end, out_battery_lot->lot_number.begin());
}
R_SUCCEED();
}
Result ISystemSettingsServer::GetSerialNumber(Out<SerialNumber> out_console_serial) {
LOG_INFO(Service_SET, "called");
*out_console_serial = []{
u32 d = ::Settings::values.serial_unit.GetValue();
SerialNumber c{};
c.serial_number[0] = 'X';
c.serial_number[1] = 'A';
c.serial_number[2] = [] {
// Adding another setting would be tedious so... let's just reuse region_index :)
switch (::Settings::values.region_index.GetValue()) {
case ::Settings::Region::Japan: return 'J';
case ::Settings::Region::Usa: return 'W';
case ::Settings::Region::Europe: return 'E';
case ::Settings::Region::Australia: return 'M'; //pretend its Malaysia
case ::Settings::Region::China:
case ::Settings::Region::Taiwan: return 'C';
case ::Settings::Region::Korea: return 'K';
default: return 'W';
}
}();
Fill3DS_CRC(d, c.serial_number.data() + 3);
return c;
}();
*out_console_serial = {"YUZ10000000001"};
if (auto const s = ::Settings::values.serial_unit.GetValue(); !s.empty()) {
auto const max_size = out_console_serial->serial_number.size();
auto const end = s.size() > max_size ? s.begin() + max_size : s.end();
std::copy(s.begin(), end, out_console_serial->serial_number.begin());
}
R_SUCCEED();
}
+2 -2
View File
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project
@@ -178,7 +178,7 @@ void LoopProcess(Core::System& system) {
auto module = std::make_shared<Module>();
server_manager->RegisterNamedService("csrng", std::make_shared<CSRNG>(system, module));
server_manager->RegisterNamedService("spl:", std::make_shared<SPL>(system, module));
server_manager->RegisterNamedService("spl", std::make_shared<SPL>(system, module));
server_manager->RegisterNamedService("spl:mig", std::make_shared<SPL_MIG>(system, module));
server_manager->RegisterNamedService("spl:fs", std::make_shared<SPL_FS>(system, module));
server_manager->RegisterNamedService("spl:ssl", std::make_shared<SPL_SSL>(system, module));
+5 -13
View File
@@ -11,8 +11,9 @@ namespace FrontendCommon {
void GenerateSettings() {
static std::random_device rd;
// Web Token
if (Settings::values.eden_token.GetValue().empty()) {
// Web Token //
auto &token_setting = Settings::values.eden_token;
if (token_setting.GetValue().empty()) {
static constexpr const size_t token_length = 48;
static constexpr const frozen::string token_set = "abcdefghijklmnopqrstuvwxyz";
static std::uniform_int_distribution<int> token_dist(0, token_set.size() - 1);
@@ -22,18 +23,9 @@ void GenerateSettings() {
size_t idx = token_dist(rd);
result += token_set[idx];
}
Settings::values.eden_token.SetValue(result);
token_setting.SetValue(result);
}
// Randomly generated number because, well, we fill the rest automagically ;)
// Other serial parts are filled by Region_Index
std::random_device device;
std::mt19937 gen(device());
std::uniform_int_distribution<u32> distribution(1, (std::numeric_limits<u32>::max)());
if (Settings::values.serial_unit.GetValue() == 0)
Settings::values.serial_unit.SetValue(distribution(gen));
if (Settings::values.serial_battery.GetValue() == 0)
Settings::values.serial_battery.SetValue(distribution(gen));
}
}
+1 -1
View File
@@ -517,8 +517,8 @@ std::unique_ptr<ComboboxTranslationMap> ComboboxEnumeration(QObject* parent)
PAIR(FramePacingMode, Target_Auto, tr("Auto")),
PAIR(FramePacingMode, Target_30, tr("30 FPS")),
PAIR(FramePacingMode, Target_60, tr("60 FPS")),
PAIR(FramePacingMode, Target_90, tr("90 FPS")),
PAIR(FramePacingMode, Target_120, tr("120 FPS")),
PAIR(FramePacingMode, Target_240, tr("240 FPS")),
}});
translations->insert({Settings::EnumMetadata<Settings::VramUsageMode>::Index(),
{
@@ -206,10 +206,16 @@ std::string_view FormatStorage(ImageFormat format) {
return "S16";
case ImageFormat::R32_UINT:
return "U32";
case ImageFormat::R32_SINT:
return "S32";
case ImageFormat::R32G32_UINT:
return "U32X2";
case ImageFormat::R32G32_SINT:
return "S32X2";
case ImageFormat::R32G32B32A32_UINT:
return "U32X4";
case ImageFormat::R32G32B32A32_SINT:
return "S32X4";
}
throw InvalidArgument("Invalid image format {}", format);
}
@@ -148,10 +148,16 @@ std::string_view ImageFormatString(ImageFormat format) {
return ",r16i";
case ImageFormat::R32_UINT:
return ",r32ui";
case ImageFormat::R32_SINT:
return ",r32i";
case ImageFormat::R32G32_UINT:
return ",rg32ui";
case ImageFormat::R32G32_SINT:
return ",rg32i";
case ImageFormat::R32G32B32A32_UINT:
return ",rgba32ui";
case ImageFormat::R32G32B32A32_SINT:
return ",rgba32i";
default:
throw NotImplementedException("Image format: {}", format);
}
@@ -142,15 +142,22 @@ Id GetCbuf(EmitContext& ctx, Id result_type, Id UniformDefinitions::*member_ptr,
const auto is_float = UniformDefinitions::IsFloat(member_ptr);
const auto num_elements = UniformDefinitions::NumElements(member_ptr);
const std::array zero_vec{
is_float ? ctx.Const(0.0f) : ctx.Const(0u),
is_float ? ctx.Const(0.0f) : ctx.Const(0u),
is_float ? ctx.Const(0.0f) : ctx.Const(0u),
is_float ? ctx.Const(0.0f) : ctx.Const(0u),
};
const Id zero_element = is_float ? ctx.Const(0.0f) : ctx.Const(0u);
const Id cond = ctx.OpULessThanEqual(ctx.TypeBool(), buffer_offset, ctx.Const(0xFFFFu));
const Id zero = ctx.OpCompositeConstruct(result_type, std::span(zero_vec.data(), num_elements));
return ctx.OpSelect(result_type, cond, val, zero);
// OpSelect with vector result requires vector condition, scalar uses scalar directly
if (num_elements > 1) {
const std::array zero_vec{zero_element, zero_element, zero_element, zero_element};
const Id zero = ctx.OpCompositeConstruct(result_type, std::span(zero_vec.data(), num_elements));
const Id bool_vector_type = ctx.TypeVector(ctx.U1, num_elements);
const std::array cond_vec{cond, cond, cond, cond};
const Id vector_cond = ctx.OpCompositeConstruct(bool_vector_type, std::span(cond_vec.data(), num_elements));
return ctx.OpSelect(result_type, vector_cond, val, zero);
} else {
return ctx.OpSelect(result_type, cond, val, zero_element);
}
}
Id GetCbufU32(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset) {
@@ -205,7 +205,7 @@ Id TextureImage(EmitContext& ctx, IR::TextureInstInfo info, const IR::Value& ind
if (def.count > 1) {
throw NotImplementedException("Indirect texture sample");
}
return ctx.OpLoad(ctx.image_buffer_type, def.id);
return ctx.OpLoad(def.image_type, def.id);
} else {
const TextureDefinition& def{ctx.textures.at(info.descriptor_index)};
if (def.count > 1) {
@@ -215,16 +215,22 @@ Id TextureImage(EmitContext& ctx, IR::TextureInstInfo info, const IR::Value& ind
}
}
std::pair<Id, bool> Image(EmitContext& ctx, const IR::Value& index, IR::TextureInstInfo info) {
struct ImageInfo {
Id image;
bool is_integer;
bool is_signed;
};
ImageInfo Image(EmitContext& ctx, const IR::Value& index, IR::TextureInstInfo info) {
if (!index.IsImmediate() || index.U32() != 0) {
throw NotImplementedException("Indirect image indexing");
}
if (info.type == TextureType::Buffer) {
const ImageBufferDefinition def{ctx.image_buffers.at(info.descriptor_index)};
return {ctx.OpLoad(def.image_type, def.id), def.is_integer};
return {ctx.OpLoad(def.image_type, def.id), def.is_integer, def.is_signed};
} else {
const ImageDefinition def{ctx.images.at(info.descriptor_index)};
return {ctx.OpLoad(def.image_type, def.id), def.is_integer};
return {ctx.OpLoad(def.image_type, def.id), def.is_integer, def.is_signed};
}
}
@@ -550,8 +556,25 @@ Id EmitImageFetch(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id c
lod = Id{};
}
const ImageOperands operands(lod, ms);
return Emit(&EmitContext::OpImageSparseFetch, &EmitContext::OpImageFetch, ctx, inst, ctx.F32[4],
TextureImage(ctx, info, index), coords, operands.MaskOptional(), operands.Span());
bool is_integer = false;
bool is_signed = false;
Id result_type{ctx.F32[4]};
if (info.type == TextureType::Buffer) {
const TextureBufferDefinition& def{ctx.texture_buffers.at(info.descriptor_index)};
is_integer = def.is_integer;
is_signed = def.is_signed;
if (is_integer) {
result_type = is_signed ? ctx.S32[4] : ctx.U32[4];
}
}
Id fetched = Emit(&EmitContext::OpImageSparseFetch, &EmitContext::OpImageFetch, ctx, inst,
result_type, TextureImage(ctx, info, index), coords,
operands.MaskOptional(), operands.Span());
if (is_integer) {
// IR expects F32x4 from ImageFetch; bitcast integer results to float vector.
fetched = ctx.OpBitcast(ctx.F32[4], fetched);
}
return fetched;
}
Id EmitImageQueryDimensions(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id lod,
@@ -612,21 +635,25 @@ Id EmitImageRead(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id co
LOG_WARNING(Shader_SPIRV, "Typeless image read not supported by host");
return ctx.ConstantNull(ctx.U32[4]);
}
const auto [image, is_integer] = Image(ctx, index, info);
const Id result_type{is_integer ? ctx.U32[4] : ctx.F32[4]};
const auto [image, is_integer, is_signed] = Image(ctx, index, info);
const Id result_type{is_integer ? (is_signed ? ctx.S32[4] : ctx.U32[4]) : ctx.F32[4]};
Id color{Emit(&EmitContext::OpImageSparseRead, &EmitContext::OpImageRead, ctx, inst,
result_type, image, coords, std::nullopt, std::span<const Id>{})};
if (!is_integer) {
color = ctx.OpBitcast(ctx.U32[4], color);
} else if (is_signed) {
color = ctx.OpBitcast(ctx.U32[4], color);
}
return color;
}
void EmitImageWrite(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords, Id color) {
const auto info{inst->Flags<IR::TextureInstInfo>()};
const auto [image, is_integer] = Image(ctx, index, info);
const auto [image, is_integer, is_signed] = Image(ctx, index, info);
if (!is_integer) {
color = ctx.OpBitcast(ctx.F32[4], color);
} else if (is_signed) {
color = ctx.OpBitcast(ctx.S32[4], color);
}
ctx.OpImageWrite(image, coords, color);
}
@@ -7,13 +7,18 @@
namespace Shader::Backend::SPIRV {
namespace {
Id Image(EmitContext& ctx, IR::TextureInstInfo info) {
struct ImageInfo {
Id id;
bool is_signed;
};
ImageInfo Image(EmitContext& ctx, IR::TextureInstInfo info) {
if (info.type == TextureType::Buffer) {
const ImageBufferDefinition def{ctx.image_buffers.at(info.descriptor_index)};
return def.id;
return {def.id, def.is_signed};
} else {
const ImageDefinition def{ctx.images.at(info.descriptor_index)};
return def.id;
return {def.id, def.is_signed};
}
}
@@ -24,42 +29,66 @@ std::pair<Id, Id> AtomicArgs(EmitContext& ctx) {
}
Id ImageAtomicU32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords, Id value,
Id (Sirit::Module::*atomic_func)(Id, Id, Id, Id, Id)) {
Id (Sirit::Module::*atomic_func)(Id, Id, Id, Id, Id), bool value_signed) {
if (!index.IsImmediate() || index.U32() != 0) {
// TODO: handle layers
throw NotImplementedException("Image indexing");
}
const auto info{inst->Flags<IR::TextureInstInfo>()};
const Id image{Image(ctx, info)};
const Id pointer{ctx.OpImageTexelPointer(ctx.image_u32, image, coords, ctx.Const(0U))};
const auto image_info{Image(ctx, info)};
Id pointer_type{image_info.is_signed ? ctx.image_s32 : ctx.image_u32};
if (!Sirit::ValidId(pointer_type)) {
const Id element_type{image_info.is_signed ? ctx.S32[1] : ctx.U32[1]};
pointer_type = ctx.TypePointer(spv::StorageClass::Image, element_type);
}
const Id image{image_info.id};
const Id pointer{ctx.OpImageTexelPointer(pointer_type, image, coords, ctx.Const(0U))};
const auto [scope, semantics]{AtomicArgs(ctx)};
return (ctx.*atomic_func)(ctx.U32[1], pointer, scope, semantics, value);
const Id result_type{image_info.is_signed ? ctx.S32[1] : ctx.U32[1]};
// Ensure value type matches result_type's pointee type
Id cast_value{value};
if (image_info.is_signed) {
// Result type is signed s32, ensure value is also s32
cast_value = ctx.OpBitcast(ctx.S32[1], value);
} else {
// Result type is unsigned u32, ensure value is also u32
cast_value = ctx.OpBitcast(ctx.U32[1], value);
}
Id result{(ctx.*atomic_func)(result_type, pointer, scope, semantics, cast_value)};
// Convert result back to u32 for IR compatibility
if (image_info.is_signed) {
result = ctx.OpBitcast(ctx.U32[1], result);
}
return result;
}
} // Anonymous namespace
Id EmitImageAtomicIAdd32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicIAdd);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicIAdd, false);
}
Id EmitImageAtomicSMin32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicSMin);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicSMin, true);
}
Id EmitImageAtomicUMin32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicUMin);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicUMin, false);
}
Id EmitImageAtomicSMax32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicSMax);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicSMax, true);
}
Id EmitImageAtomicUMax32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicUMax);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicUMax, false);
}
Id EmitImageAtomicInc32(EmitContext&, IR::Inst*, const IR::Value&, Id, Id) {
@@ -74,22 +103,22 @@ Id EmitImageAtomicDec32(EmitContext&, IR::Inst*, const IR::Value&, Id, Id) {
Id EmitImageAtomicAnd32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicAnd);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicAnd, false);
}
Id EmitImageAtomicOr32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicOr);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicOr, false);
}
Id EmitImageAtomicXor32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicXor);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicXor, false);
}
Id EmitImageAtomicExchange32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicExchange);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicExchange, false);
}
Id EmitBindlessImageAtomicIAdd32(EmitContext&) {
@@ -69,10 +69,16 @@ spv::ImageFormat GetImageFormat(ImageFormat format) {
return spv::ImageFormat::R16i;
case ImageFormat::R32_UINT:
return spv::ImageFormat::R32ui;
case ImageFormat::R32_SINT:
return spv::ImageFormat::R32i;
case ImageFormat::R32G32_UINT:
return spv::ImageFormat::Rg32ui;
case ImageFormat::R32G32_SINT:
return spv::ImageFormat::Rg32i;
case ImageFormat::R32G32B32A32_UINT:
return spv::ImageFormat::Rgba32ui;
case ImageFormat::R32G32B32A32_SINT:
return spv::ImageFormat::Rgba32i;
}
throw InvalidArgument("Invalid image format {}", format);
}
@@ -613,7 +619,10 @@ void EmitContext::DefineSharedMemory(const IR::Program& program) {
const Id element_pointer{TypePointer(spv::StorageClass::Workgroup, element_type)};
const Id variable{AddGlobalVariable(pointer, spv::StorageClass::Workgroup)};
Decorate(variable, spv::Decoration::Aliased);
interfaces.push_back(variable);
// Workgroup variables in EntryPoint interfaces are only supported in SPIR-V 1.4+
if (profile.supported_spirv >= 0x00010400) {
interfaces.push_back(variable);
}
return std::make_tuple(variable, element_pointer, pointer);
}};
@@ -642,7 +651,10 @@ void EmitContext::DefineSharedMemory(const IR::Program& program) {
shared_u32 = TypePointer(spv::StorageClass::Workgroup, U32[1]);
shared_memory_u32 = AddGlobalVariable(shared_memory_u32_type, spv::StorageClass::Workgroup);
interfaces.push_back(shared_memory_u32);
// Workgroup variables in EntryPoint interfaces are only supported in SPIR-V 1.4+
if (profile.supported_spirv >= 0x00010400) {
interfaces.push_back(shared_memory_u32);
}
const Id func_type{TypeFunction(void_id, U32[1], U32[1])};
const auto make_function{[&](u32 mask, u32 size) {
@@ -1305,20 +1317,25 @@ void EmitContext::DefineTextureBuffers(const Info& info, u32& binding) {
return;
}
const spv::ImageFormat format{spv::ImageFormat::Unknown};
image_buffer_type = TypeImage(F32[1], spv::Dim::Buffer, 0U, false, false, 1, format);
const Id type{TypePointer(spv::StorageClass::UniformConstant, image_buffer_type)};
texture_buffers.reserve(info.texture_buffer_descriptors.size());
for (const TextureBufferDescriptor& desc : info.texture_buffer_descriptors) {
if (desc.count != 1) {
throw NotImplementedException("Array of texture buffers");
}
// Use the correct sampled type based on the descriptor's data format
const Id sampled_type{desc.is_integer ? (desc.is_signed ? S32[1] : U32[1]) : F32[1]};
image_buffer_type = TypeImage(sampled_type, spv::Dim::Buffer, 0U, false, false, 1, format);
const Id type{TypePointer(spv::StorageClass::UniformConstant, image_buffer_type)};
const Id id{AddGlobalVariable(type, spv::StorageClass::UniformConstant)};
Decorate(id, spv::Decoration::Binding, binding);
Decorate(id, spv::Decoration::DescriptorSet, 0U);
Name(id, NameOf(stage, desc, "texbuf"));
texture_buffers.push_back({
.id = id,
.image_type = image_buffer_type,
.is_integer = desc.is_integer,
.is_signed = desc.is_signed,
.count = desc.count,
});
if (profile.supported_spirv >= 0x00010400) {
@@ -1335,19 +1352,23 @@ void EmitContext::DefineImageBuffers(const Info& info, u32& binding) {
throw NotImplementedException("Array of image buffers");
}
const spv::ImageFormat format{GetImageFormat(desc.format)};
const Id sampled_type{desc.is_integer ? U32[1] : F32[1]};
const Id sampled_type{desc.is_integer ? (desc.is_signed ? S32[1] : U32[1]) : F32[1]};
const Id image_type{
TypeImage(sampled_type, spv::Dim::Buffer, false, false, false, 2, format)};
const Id pointer_type{TypePointer(spv::StorageClass::UniformConstant, image_type)};
const Id id{AddGlobalVariable(pointer_type, spv::StorageClass::UniformConstant)};
Decorate(id, spv::Decoration::Binding, binding);
Decorate(id, spv::Decoration::DescriptorSet, 0U);
if (format == spv::ImageFormat::Unknown) {
Decorate(id, spv::Decoration::NonReadable);
}
Name(id, NameOf(stage, desc, "imgbuf"));
image_buffers.push_back({
.id = id,
.image_type = image_type,
.count = desc.count,
.is_integer = desc.is_integer,
.is_signed = desc.is_signed,
});
if (profile.supported_spirv >= 0x00010400) {
interfaces.push_back(id);
@@ -1384,6 +1405,9 @@ void EmitContext::DefineTextures(const Info& info, u32& binding, u32& scaling_in
if (info.uses_atomic_image_u32) {
image_u32 = TypePointer(spv::StorageClass::Image, U32[1]);
}
if (info.uses_atomic_s32_min || info.uses_atomic_s32_max) {
image_s32 = TypePointer(spv::StorageClass::Image, S32[1]);
}
}
void EmitContext::DefineImages(const Info& info, u32& binding, u32& scaling_index) {
@@ -1392,18 +1416,23 @@ void EmitContext::DefineImages(const Info& info, u32& binding, u32& scaling_inde
if (desc.count != 1) {
throw NotImplementedException("Array of images");
}
const Id sampled_type{desc.is_integer ? U32[1] : F32[1]};
const Id sampled_type{desc.is_integer ? (desc.is_signed ? S32[1] : U32[1]) : F32[1]};
const Id image_type{ImageType(*this, desc, sampled_type)};
const Id pointer_type{TypePointer(spv::StorageClass::UniformConstant, image_type)};
const Id id{AddGlobalVariable(pointer_type, spv::StorageClass::UniformConstant)};
Decorate(id, spv::Decoration::Binding, binding);
Decorate(id, spv::Decoration::DescriptorSet, 0U);
const spv::ImageFormat format{GetImageFormat(desc.format)};
if (format == spv::ImageFormat::Unknown) {
Decorate(id, spv::Decoration::NonReadable);
}
Name(id, NameOf(stage, desc, "img"));
images.push_back({
.id = id,
.image_type = image_type,
.count = desc.count,
.is_integer = desc.is_integer,
.is_signed = desc.is_signed,
});
if (profile.supported_spirv >= 0x00010400) {
interfaces.push_back(id);
@@ -45,6 +45,9 @@ struct TextureDefinition {
struct TextureBufferDefinition {
Id id;
Id image_type; // Stores the correct buffer image type (F32 or U32 based on descriptor)
bool is_integer;
bool is_signed;
u32 count;
};
@@ -53,6 +56,7 @@ struct ImageBufferDefinition {
Id image_type;
u32 count;
bool is_integer;
bool is_signed;
};
struct ImageDefinition {
@@ -60,6 +64,7 @@ struct ImageDefinition {
Id image_type;
u32 count;
bool is_integer;
bool is_signed;
};
struct UniformDefinitions {
@@ -250,6 +255,7 @@ public:
Id image_buffer_type{};
Id image_u32{};
Id image_s32{};
std::array<UniformDefinitions, Info::MAX_CBUFS> cbufs{};
std::array<StorageDefinitions, Info::MAX_SSBOS> ssbos{};
+74 -2
View File
@@ -204,6 +204,65 @@ static inline bool IsTexturePixelFormatIntegerCached(Environment& env,
return env.IsTexturePixelFormatInteger(GetTextureHandleCached(env, cbuf));
}
static inline bool IsTexturePixelFormatSignedCached(Environment& env,
const ConstBufferAddr& cbuf) {
switch (ReadTexturePixelFormatCached(env, cbuf)) {
case TexturePixelFormat::A8B8G8R8_SINT:
case TexturePixelFormat::R8_SINT:
case TexturePixelFormat::R8G8_SINT:
case TexturePixelFormat::R16_SINT:
case TexturePixelFormat::R16G16_SINT:
case TexturePixelFormat::R16G16B16A16_SINT:
case TexturePixelFormat::R32_SINT:
case TexturePixelFormat::R32G32_SINT:
case TexturePixelFormat::R32G32B32A32_SINT:
return true;
default:
return false;
}
}
static inline bool IsImageFormatSigned(ImageFormat format) {
switch (format) {
case ImageFormat::R8_SINT:
case ImageFormat::R16_SINT:
case ImageFormat::R32_SINT:
case ImageFormat::R32G32_SINT:
case ImageFormat::R32G32B32A32_SINT:
return true;
default:
return false;
}
}
static inline std::optional<ImageFormat> BufferImageFormatFromPixelFormat(
TexturePixelFormat pixel_format) {
switch (pixel_format) {
case TexturePixelFormat::R8_UINT:
return ImageFormat::R8_UINT;
case TexturePixelFormat::R8_SINT:
return ImageFormat::R8_SINT;
case TexturePixelFormat::R16_UINT:
return ImageFormat::R16_UINT;
case TexturePixelFormat::R16_SINT:
return ImageFormat::R16_SINT;
case TexturePixelFormat::R32_UINT:
return ImageFormat::R32_UINT;
case TexturePixelFormat::R32_SINT:
return ImageFormat::R32_SINT;
case TexturePixelFormat::R32G32_UINT:
return ImageFormat::R32G32_UINT;
case TexturePixelFormat::R32G32_SINT:
return ImageFormat::R32G32_SINT;
case TexturePixelFormat::R32G32B32A32_UINT:
return ImageFormat::R32G32B32A32_UINT;
case TexturePixelFormat::R32G32B32A32_SINT:
return ImageFormat::R32G32B32A32_SINT;
default:
return std::nullopt;
}
}
std::optional<ConstBufferAddr> Track(const IR::Value& value, Environment& env);
static inline std::optional<ConstBufferAddr> TrackCached(const IR::Value& v, Environment& env) {
@@ -652,12 +711,21 @@ void TexturePass(Environment& env, IR::Program& program, const HostTranslateInfo
const bool is_written{inst->GetOpcode() != IR::Opcode::ImageRead};
const bool is_read{inst->GetOpcode() != IR::Opcode::ImageWrite};
const bool is_integer{IsTexturePixelFormatIntegerCached(env, cbuf)};
ImageFormat image_format = flags.image_format;
if (flags.type == TextureType::Buffer) {
const auto pixel_format = ReadTexturePixelFormatCached(env, cbuf);
if (const auto mapped = BufferImageFormatFromPixelFormat(pixel_format)) {
image_format = *mapped;
}
}
const bool is_signed{IsImageFormatSigned(image_format)};
if (flags.type == TextureType::Buffer) {
index = descriptors.Add(ImageBufferDescriptor{
.format = flags.image_format,
.format = image_format,
.is_written = is_written,
.is_read = is_read,
.is_integer = is_integer,
.is_signed = is_signed,
.cbuf_index = cbuf.index,
.cbuf_offset = cbuf.offset,
.count = cbuf.count,
@@ -666,22 +734,26 @@ void TexturePass(Environment& env, IR::Program& program, const HostTranslateInfo
} else {
index = descriptors.Add(ImageDescriptor{
.type = flags.type,
.format = flags.image_format,
.format = image_format,
.is_written = is_written,
.is_read = is_read,
.is_integer = is_integer,
.is_signed = is_signed,
.cbuf_index = cbuf.index,
.cbuf_offset = cbuf.offset,
.count = cbuf.count,
.size_shift = DESCRIPTOR_SIZE_SHIFT,
});
}
flags.image_format.Assign(image_format);
break;
}
default:
if (flags.type == TextureType::Buffer) {
index = descriptors.Add(TextureBufferDescriptor{
.has_secondary = cbuf.has_secondary,
.is_integer = IsTexturePixelFormatIntegerCached(env, cbuf),
.is_signed = IsTexturePixelFormatSignedCached(env, cbuf),
.cbuf_index = cbuf.index,
.cbuf_offset = cbuf.offset,
.shift_left = cbuf.shift_left,
+7
View File
@@ -150,8 +150,11 @@ enum class ImageFormat : u32 {
R16_UINT,
R16_SINT,
R32_UINT,
R32_SINT,
R32G32_UINT,
R32G32_SINT,
R32G32B32A32_UINT,
R32G32B32A32_SINT,
};
enum class Interpolation {
@@ -178,6 +181,8 @@ struct StorageBufferDescriptor {
struct TextureBufferDescriptor {
bool has_secondary;
bool is_integer; // True if data is SINT/UINT (from R_type in TIC), false if FLOAT
bool is_signed; // True if integer data is signed
u32 cbuf_index;
u32 cbuf_offset;
u32 shift_left;
@@ -196,6 +201,7 @@ struct ImageBufferDescriptor {
bool is_written;
bool is_read;
bool is_integer;
bool is_signed;
u32 cbuf_index;
u32 cbuf_offset;
u32 count;
@@ -229,6 +235,7 @@ struct ImageDescriptor {
bool is_written;
bool is_read;
bool is_integer;
bool is_signed;
u32 cbuf_index;
u32 cbuf_offset;
u32 count;
+19 -9
View File
@@ -458,14 +458,14 @@ void BufferCache<P>::UnbindGraphicsTextureBuffers(size_t stage) {
template <class P>
void BufferCache<P>::BindGraphicsTextureBuffer(size_t stage, size_t tbo_index, GPUVAddr gpu_addr,
u32 size, PixelFormat format, bool is_written,
bool is_image) {
bool is_image, bool is_integer, bool is_signed) {
channel_state->enabled_texture_buffers[stage] |= 1U << tbo_index;
channel_state->written_texture_buffers[stage] |= (is_written ? 1U : 0U) << tbo_index;
if constexpr (SEPARATE_IMAGE_BUFFERS_BINDINGS) {
channel_state->image_texture_buffers[stage] |= (is_image ? 1U : 0U) << tbo_index;
}
channel_state->texture_buffers[stage][tbo_index] =
GetTextureBufferBinding(gpu_addr, size, format);
GetTextureBufferBinding(gpu_addr, size, format, is_integer, is_signed);
}
template <class P>
@@ -532,7 +532,8 @@ void BufferCache<P>::UnbindComputeTextureBuffers() {
template <class P>
void BufferCache<P>::BindComputeTextureBuffer(size_t tbo_index, GPUVAddr gpu_addr, u32 size,
PixelFormat format, bool is_written, bool is_image) {
PixelFormat format, bool is_written, bool is_image,
bool is_integer, bool is_signed) {
if (tbo_index >= channel_state->compute_texture_buffers.size()) [[unlikely]] {
LOG_ERROR(HW_GPU, "Texture buffer index {} exceeds maximum texture buffer count",
tbo_index);
@@ -544,7 +545,7 @@ void BufferCache<P>::BindComputeTextureBuffer(size_t tbo_index, GPUVAddr gpu_add
channel_state->image_compute_texture_buffers |= (is_image ? 1U : 0U) << tbo_index;
}
channel_state->compute_texture_buffers[tbo_index] =
GetTextureBufferBinding(gpu_addr, size, format);
GetTextureBufferBinding(gpu_addr, size, format, is_integer, is_signed);
}
template <class P>
@@ -955,15 +956,17 @@ void BufferCache<P>::BindHostGraphicsTextureBuffers(size_t stage) {
const u32 offset = buffer.Offset(binding.device_addr);
const PixelFormat format = binding.format;
const bool is_integer = binding.is_integer;
const bool is_signed = binding.is_signed;
buffer.MarkUsage(offset, size);
if constexpr (SEPARATE_IMAGE_BUFFERS_BINDINGS) {
if (((channel_state->image_texture_buffers[stage] >> index) & 1) != 0) {
runtime.BindImageBuffer(buffer, offset, size, format);
} else {
runtime.BindTextureBuffer(buffer, offset, size, format);
runtime.BindTextureBuffer(buffer, offset, size, format, is_integer, is_signed);
}
} else {
runtime.BindTextureBuffer(buffer, offset, size, format);
runtime.BindTextureBuffer(buffer, offset, size, format, is_integer, is_signed);
}
});
}
@@ -1090,15 +1093,17 @@ void BufferCache<P>::BindHostComputeTextureBuffers() {
const u32 offset = buffer.Offset(binding.device_addr);
const PixelFormat format = binding.format;
const bool is_integer = binding.is_integer;
const bool is_signed = binding.is_signed;
buffer.MarkUsage(offset, size);
if constexpr (SEPARATE_IMAGE_BUFFERS_BINDINGS) {
if (((channel_state->image_compute_texture_buffers >> index) & 1) != 0) {
runtime.BindImageBuffer(buffer, offset, size, format);
} else {
runtime.BindTextureBuffer(buffer, offset, size, format);
runtime.BindTextureBuffer(buffer, offset, size, format, is_integer, is_signed);
}
} else {
runtime.BindTextureBuffer(buffer, offset, size, format);
runtime.BindTextureBuffer(buffer, offset, size, format, is_integer, is_signed);
}
});
}
@@ -1833,7 +1838,8 @@ Binding BufferCache<P>::StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
template <class P>
TextureBufferBinding BufferCache<P>::GetTextureBufferBinding(GPUVAddr gpu_addr, u32 size,
PixelFormat format) {
PixelFormat format, bool is_integer,
bool is_signed) {
const std::optional<DAddr> device_addr = gpu_memory->GpuToCpuAddress(gpu_addr);
TextureBufferBinding binding;
if (!device_addr || size == 0) {
@@ -1841,11 +1847,15 @@ TextureBufferBinding BufferCache<P>::GetTextureBufferBinding(GPUVAddr gpu_addr,
binding.size = 0;
binding.buffer_id = NULL_BUFFER_ID;
binding.format = PixelFormat::Invalid;
binding.is_integer = false;
binding.is_signed = false;
} else {
binding.device_addr = *device_addr;
binding.size = size;
binding.buffer_id = BufferId{};
binding.format = format;
binding.is_integer = is_integer;
binding.is_signed = is_signed;
}
return binding;
}
@@ -84,6 +84,8 @@ struct Binding {
struct TextureBufferBinding : Binding {
PixelFormat format;
bool is_integer{}; // True if data is SINT/UINT, false if FLOAT
bool is_signed{}; // True if integer data is signed
};
static constexpr Binding NULL_BINDING{
@@ -251,7 +253,8 @@ public:
void UnbindGraphicsTextureBuffers(size_t stage);
void BindGraphicsTextureBuffer(size_t stage, size_t tbo_index, GPUVAddr gpu_addr, u32 size,
PixelFormat format, bool is_written, bool is_image);
PixelFormat format, bool is_written, bool is_image,
bool is_integer, bool is_signed);
void UnbindComputeStorageBuffers();
@@ -261,7 +264,7 @@ public:
void UnbindComputeTextureBuffers();
void BindComputeTextureBuffer(size_t tbo_index, GPUVAddr gpu_addr, u32 size, PixelFormat format,
bool is_written, bool is_image);
bool is_written, bool is_image, bool is_integer, bool is_signed);
[[nodiscard]] std::pair<Buffer*, u32> ObtainBuffer(GPUVAddr gpu_addr, u32 size,
ObtainBufferSynchronize sync_info,
@@ -445,7 +448,8 @@ private:
bool is_written) const;
[[nodiscard]] TextureBufferBinding GetTextureBufferBinding(GPUVAddr gpu_addr, u32 size,
PixelFormat format);
PixelFormat format, bool is_integer,
bool is_signed);
[[nodiscard]] std::span<const u8> ImmediateBufferWithData(DAddr device_addr, size_t size);
@@ -88,7 +88,7 @@ void Buffer::MakeResident(GLenum access) noexcept {
glMakeNamedBufferResidentNV(buffer.handle, access);
}
GLuint Buffer::View(u32 offset, u32 size, PixelFormat format) {
GLuint Buffer::View(u32 offset, u32 size, PixelFormat format, bool is_integer, bool is_signed) {
const auto it{std::ranges::find_if(views, [offset, size, format](const BufferView& view) {
return offset == view.offset && size == view.size && format == view.format;
})};
@@ -370,8 +370,8 @@ void BufferCacheRuntime::BindTransformFeedbackBuffers(VideoCommon::HostBindings<
}
void BufferCacheRuntime::BindTextureBuffer(Buffer& buffer, u32 offset, u32 size,
PixelFormat format) {
*texture_handles++ = buffer.View(offset, size, format);
PixelFormat format, bool is_integer, bool is_signed) {
*texture_handles++ = buffer.View(offset, size, format, is_integer, is_signed);
}
void BufferCacheRuntime::BindImageBuffer(Buffer& buffer, u32 offset, u32 size, PixelFormat format) {
@@ -34,7 +34,8 @@ public:
void MarkUsage(u64 offset, u64 size) {}
[[nodiscard]] GLuint View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format);
[[nodiscard]] GLuint View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format,
bool is_integer = false, bool is_signed = false);
[[nodiscard]] GLuint64EXT HostGpuAddr() const noexcept {
return address;
@@ -118,7 +119,7 @@ public:
void BindTransformFeedbackBuffers(VideoCommon::HostBindings<Buffer>& bindings);
void BindTextureBuffer(Buffer& buffer, u32 offset, u32 size,
VideoCore::Surface::PixelFormat format);
VideoCore::Surface::PixelFormat format, bool is_integer, bool is_signed);
void BindImageBuffer(Buffer& buffer, u32 offset, u32 size,
VideoCore::Surface::PixelFormat format);
@@ -177,7 +177,8 @@ void ComputePipeline::Configure() {
ImageView& image_view{texture_cache.GetImageView(views[texbuf_index].id)};
buffer_cache.BindComputeTextureBuffer(texbuf_index, image_view.GpuAddr(),
image_view.BufferSize(), image_view.format,
is_written, is_image);
is_written, is_image, desc.is_integer,
desc.is_signed);
++texbuf_index;
}
}};
@@ -397,7 +397,8 @@ bool GraphicsPipeline::ConfigureImpl(bool is_indexed) {
ImageView& image_view{texture_cache.GetImageView(texture_buffer_it->id)};
buffer_cache.BindGraphicsTextureBuffer(stage, index, image_view.GpuAddr(),
image_view.BufferSize(), image_view.format,
is_written, is_image);
is_written, is_image, desc.is_integer,
desc.is_signed);
++index;
++texture_buffer_it;
}
@@ -433,10 +433,16 @@ OGLTexture MakeImage(const VideoCommon::ImageInfo& info, GLenum gl_internal_form
return GL_R16I;
case Shader::ImageFormat::R32_UINT:
return GL_R32UI;
case Shader::ImageFormat::R32_SINT:
return GL_R32I;
case Shader::ImageFormat::R32G32_UINT:
return GL_RG32UI;
case Shader::ImageFormat::R32G32_SINT:
return GL_RG32I;
case Shader::ImageFormat::R32G32B32A32_UINT:
return GL_RGBA32UI;
case Shader::ImageFormat::R32G32B32A32_SINT:
return GL_RGBA32I;
}
ASSERT_MSG(false, "Invalid image format={}", format);
return GL_R32UI;
@@ -177,6 +177,10 @@ try
RendererVulkan::~RendererVulkan() {
scheduler.RegisterOnSubmit([] {});
scheduler.WaitWorker();
// vkDeviceWaitIdle MUST be called only after all queue submissions are complete
// to avoid threading errors on VkQueue simultaneous access
std::scoped_lock lock{scheduler.submit_mutex};
void(device.GetLogical().WaitIdle());
}
@@ -30,6 +30,9 @@ BlitScreen::~BlitScreen() = default;
void BlitScreen::WaitIdle() {
present_manager.WaitPresent();
scheduler.Finish();
std::scoped_lock lock{scheduler.submit_mutex};
// vkDeviceWaitIdle MUST be protected by submit_mutex to prevent racing with queue submissions
// from the worker thread. This ensures no simultaneous access to VkQueue.
device.GetLogical().WaitIdle();
}
@@ -83,7 +83,7 @@ vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allo
} // Anonymous namespace
Buffer::Buffer(BufferCacheRuntime& runtime, VideoCommon::NullBufferParams null_params)
: VideoCommon::BufferBase(null_params), tracker{4096} {
: VideoCommon::BufferBase(null_params), runtime{&runtime}, tracker{4096} {
if (runtime.device.HasNullDescriptor()) {
return;
}
@@ -93,14 +93,53 @@ Buffer::Buffer(BufferCacheRuntime& runtime, VideoCommon::NullBufferParams null_p
}
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_)
: VideoCommon::BufferBase(cpu_addr_, size_bytes_), device{&runtime.device},
: VideoCommon::BufferBase(cpu_addr_, size_bytes_), runtime{&runtime}, device{&runtime.device},
buffer{CreateBuffer(*device, runtime.memory_allocator, SizeBytes())}, tracker{SizeBytes()} {
if (runtime.device.HasDebuggingToolAttached()) {
buffer.SetObjectNameEXT(fmt::format("Buffer 0x{:x}", CpuAddr()).c_str());
}
}
VkBufferView Buffer::View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format) {
VkFormat SelectTexelBufferFormat(VkFormat float_format, bool is_integer, bool is_signed) {
// If the buffer stores integer data but Vulkan reports float format,
// we need to map to appropriate integer formats for type compatibility
if (!is_integer) {
// Non-integer buffer, use the original float format
return float_format;
}
// Integer buffer: map float formats to signed/unsigned equivalents
if (is_signed) {
// Signed integer
switch (float_format) {
case VK_FORMAT_R32_SFLOAT:
return VK_FORMAT_R32_SINT;
case VK_FORMAT_R32G32_SFLOAT:
return VK_FORMAT_R32G32_SINT;
case VK_FORMAT_R32G32B32A32_SFLOAT:
return VK_FORMAT_R32G32B32A32_SINT;
default:
// For non-float formats, use as-is
return float_format;
}
} else {
// Unsigned integer
switch (float_format) {
case VK_FORMAT_R32_SFLOAT:
return VK_FORMAT_R32_UINT;
case VK_FORMAT_R32G32_SFLOAT:
return VK_FORMAT_R32G32_UINT;
case VK_FORMAT_R32G32B32A32_SFLOAT:
return VK_FORMAT_R32G32B32A32_UINT;
default:
// For non-float formats, use as-is
return float_format;
}
}
}
VkBufferView Buffer::View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format,
bool is_integer, bool is_signed) {
if (!device) {
// Null buffer supported, return a null descriptor
return VK_NULL_HANDLE;
@@ -109,25 +148,44 @@ VkBufferView Buffer::View(u32 offset, u32 size, VideoCore::Surface::PixelFormat
offset = 0;
size = 0;
}
const VkDeviceSize alignment = device->GetTexelBufferOffsetAlignment();
const bool needs_alignment = alignment != 0 && (offset % alignment) != 0;
const auto it{std::ranges::find_if(views, [offset, size, format](const BufferView& view) {
return offset == view.offset && size == view.size && format == view.format;
})};
if (it != views.end()) {
return *it->handle;
}
vk::Buffer backing_buffer{};
VkBuffer view_buffer = *buffer;
u32 view_offset = offset;
if (needs_alignment && runtime && size != 0) {
backing_buffer = CreateBuffer(*device, runtime->memory_allocator, size);
VideoCommon::BufferCopy copy{
.src_offset = offset,
.dst_offset = 0,
.size = size,
};
runtime->CopyBuffer(*backing_buffer, *buffer, std::span{&copy, 1}, true);
view_buffer = *backing_buffer;
view_offset = 0;
}
views.push_back({
.offset = offset,
.size = size,
.format = format,
.is_integer = is_integer,
.is_signed = is_signed,
.handle = device->GetLogical().CreateBufferView({
.sType = VK_STRUCTURE_TYPE_BUFFER_VIEW_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.buffer = *buffer,
.buffer = view_buffer,
.format = MaxwellToVK::SurfaceFormat(*device, FormatType::Buffer, false, format).format,
.offset = offset,
.offset = view_offset,
.range = size,
}),
.backing_buffer = std::move(backing_buffer),
});
return *views.back().handle;
}
@@ -494,6 +552,7 @@ void BufferCacheRuntime::PostCopyBarrier() {
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([](vk::CommandBuffer cmdbuf) {
const VkPipelineStageFlags dst_stages =
VK_PIPELINE_STAGE_DRAW_INDIRECT_BIT |
VK_PIPELINE_STAGE_VERTEX_INPUT_BIT |
VK_PIPELINE_STAGE_VERTEX_SHADER_BIT |
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT |
@@ -33,7 +33,8 @@ public:
explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params);
explicit Buffer(BufferCacheRuntime& runtime, VAddr cpu_addr_, u64 size_bytes_);
[[nodiscard]] VkBufferView View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format);
[[nodiscard]] VkBufferView View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format,
bool is_integer = false, bool is_signed = false);
[[nodiscard]] VkBuffer Handle() const noexcept {
return *buffer;
@@ -60,9 +61,13 @@ private:
u32 offset;
u32 size;
VideoCore::Surface::PixelFormat format;
bool is_integer;
bool is_signed;
vk::BufferView handle;
vk::Buffer backing_buffer;
};
BufferCacheRuntime* runtime{};
const Device* device{};
vk::Buffer buffer;
std::vector<BufferView> views;
@@ -149,8 +154,8 @@ public:
}
void BindTextureBuffer(Buffer& buffer, u32 offset, u32 size,
VideoCore::Surface::PixelFormat format) {
guest_descriptor_queue.AddTexelBuffer(buffer.View(offset, size, format));
VideoCore::Surface::PixelFormat format, bool is_integer, bool is_signed) {
guest_descriptor_queue.AddTexelBuffer(buffer.View(offset, size, format, is_integer, is_signed));
}
bool ShouldLimitDynamicStorageBuffers() const {
@@ -916,7 +916,7 @@ void BlockLinearUnswizzle3DPass::UnswizzleChunk(
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
.subresourceRange = {aspect, 0, 1, 0, 1},
.subresourceRange{aspect, 0, VK_REMAINING_MIP_LEVELS, 0, VK_REMAINING_ARRAY_LAYERS},
};
// Single barrier handles both buffer and image
@@ -950,7 +950,7 @@ void BlockLinearUnswizzle3DPass::UnswizzleChunk(
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
.subresourceRange = {aspect, 0, 1, 0, 1},
.subresourceRange{aspect, 0, VK_REMAINING_MIP_LEVELS, 0, VK_REMAINING_ARRAY_LAYERS},
};
cmdbuf.PipelineBarrier(
@@ -194,7 +194,8 @@ void ComputePipeline::Configure(Tegra::Engines::KeplerCompute& kepler_compute,
ImageView& image_view = texture_cache.GetImageView(views[index].id);
buffer_cache.BindComputeTextureBuffer(index, image_view.GpuAddr(),
image_view.BufferSize(), image_view.format,
is_written, is_image);
is_written, is_image, desc.is_integer,
desc.is_signed);
++index;
}
}};
@@ -422,7 +422,8 @@ bool GraphicsPipeline::ConfigureImpl(bool is_indexed) {
ImageView& image_view{texture_cache.GetImageView(texture_buffer_it->id)};
buffer_cache.BindGraphicsTextureBuffer(stage, index, image_view.GpuAddr(),
image_view.BufferSize(), image_view.format,
is_written, is_image);
is_written, is_image, desc.is_integer,
desc.is_signed);
++index;
++texture_buffer_it;
}
@@ -157,10 +157,10 @@ public:
ReserveHostQuery();
scheduler.Record([query_pool = current_query_pool,
query_index = current_bank_slot](vk::CommandBuffer cmdbuf) {
query_index = current_bank_slot](vk::CommandBuffer cmdbuf) {
const bool use_precise = Settings::IsGPULevelHigh();
cmdbuf.BeginQuery(query_pool, static_cast<u32>(query_index),
use_precise ? VK_QUERY_CONTROL_PRECISE_BIT : 0);
use_precise ? VK_QUERY_CONTROL_PRECISE_BIT : 0);
});
has_started = true;
@@ -454,6 +454,11 @@ private:
void ReserveHostQuery() {
size_t new_slot = ReserveBankSlot();
scheduler.RecordWithUploadBuffer([query_pool = current_query_pool,
query_index = current_bank_slot](vk::CommandBuffer,
vk::CommandBuffer upload_cmdbuf) {
upload_cmdbuf.ResetQueryPool(query_pool, static_cast<u32>(query_index), 1);
});
current_bank->AddReference(1);
num_slots_used++;
if (current_query) {
@@ -52,7 +52,8 @@ using VideoCore::Surface::SurfaceType;
const bool has_stencil = surface_type == SurfaceType::DepthStencil ||
surface_type == SurfaceType::Stencil;
// Use optimal layouts for attachments - this allows drivers to optimize tiling and access patterns
// Attachments are tracked as GENERAL outside render passes; render-pass begin performs
// the transition into attachment-optimal layout for the subpass.
const VkImageLayout attachment_layout = is_depth_stencil
? VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL
: VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL;
@@ -67,7 +68,7 @@ using VideoCore::Surface::SurfaceType;
: VK_ATTACHMENT_LOAD_OP_DONT_CARE,
.stencilStoreOp = has_stencil ? VK_ATTACHMENT_STORE_OP_STORE
: VK_ATTACHMENT_STORE_OP_DONT_CARE,
.initialLayout = attachment_layout,
.initialLayout = VK_IMAGE_LAYOUT_GENERAL,
.finalLayout = attachment_layout,
};
}
@@ -90,7 +91,7 @@ VkRenderPass RenderPassCache::Get(const RenderPassKey& key) {
const bool is_valid{format != PixelFormat::Invalid};
references[index] = VkAttachmentReference{
.attachment = is_valid ? num_colors : VK_ATTACHMENT_UNUSED,
.layout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL,
.layout = VK_IMAGE_LAYOUT_GENERAL,
};
if (is_valid) {
descriptions.push_back(AttachmentDescription(*device, format, key.samples, false));
@@ -103,7 +104,7 @@ VkRenderPass RenderPassCache::Get(const RenderPassKey& key) {
if (key.depth_format != PixelFormat::Invalid) {
depth_reference = VkAttachmentReference{
.attachment = num_colors,
.layout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL,
.layout = VK_IMAGE_LAYOUT_GENERAL,
};
descriptions.push_back(AttachmentDescription(*device, key.depth_format, key.samples, true));
}
@@ -347,7 +347,7 @@ void Scheduler::EndRenderPass()
Record([num_images = num_renderpass_images,
images = renderpass_images,
ranges = renderpass_image_ranges](vk::CommandBuffer cmdbuf) {
std::array<VkImageMemoryBarrier, 9> barriers;
std::vector<VkImageMemoryBarrier> barriers(num_images);
VkPipelineStageFlags src_stages = 0;
for (size_t i = 0; i < num_images; ++i) {
const VkImageSubresourceRange& range = ranges[i];
@@ -365,20 +365,22 @@ void Scheduler::EndRenderPass()
VkImageLayout new_layout;
if (is_color) {
// Color attachments can be read as textures or used as attachments again
// Keep GENERAL to match descriptor image layouts used across the renderer.
src_access = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT;
this_stage = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT;
new_layout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
new_layout = VK_IMAGE_LAYOUT_GENERAL;
dst_access = VK_ACCESS_SHADER_READ_BIT
| VK_ACCESS_SHADER_WRITE_BIT
| VK_ACCESS_COLOR_ATTACHMENT_READ_BIT
| VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT;
} else if (is_depth_stencil) {
// Depth attachments can be read as textures or used as attachments again
// Keep GENERAL to match descriptor image layouts used across the renderer.
src_access = VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT;
this_stage = VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT
| VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT;
new_layout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
new_layout = VK_IMAGE_LAYOUT_GENERAL;
dst_access = VK_ACCESS_SHADER_READ_BIT
| VK_ACCESS_SHADER_WRITE_BIT
| VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT
| VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT;
} else {
+20 -21
View File
@@ -115,27 +115,28 @@ public:
/// Waits for the given GPU tick, optionally pacing frames.
void Wait(u64 tick, double target_fps = 0.0) {
if (Settings::values.use_speed_limit.GetValue() && target_fps > 0.0) {
const auto now = std::chrono::steady_clock::now();
if (start_time == std::chrono::steady_clock::time_point{} || current_target_fps != target_fps) {
start_time = now;
frame_counter = 0;
current_target_fps = target_fps;
}
frame_counter++;
std::chrono::duration<double> frame_interval(1.0 / current_target_fps);
auto target_time = start_time + frame_interval * frame_counter;
if (target_time > now) {
std::this_thread::sleep_until(target_time);
auto frame_duration = std::chrono::duration_cast<std::chrono::steady_clock::duration>(std::chrono::duration<double>(1.0 / target_fps));
auto now = std::chrono::steady_clock::now();
if (now < next_frame_time) {
std::this_thread::sleep_until(next_frame_time);
next_frame_time += frame_duration;
} else {
start_time = now;
frame_counter = 0;
next_frame_time = now + frame_duration;
}
}
if (tick > 0) {
if (tick >= master_semaphore->CurrentTick()) {
Flush();
}
master_semaphore->Wait(tick);
if (tick > master_semaphore->CurrentTick() && !chunk->Empty()) {
Flush();
}
master_semaphore->Wait(tick);
}
/// Resets the frame pacing state by setting the next frame time.
void ResetFramePacing(double target_fps = 0.0) {
if (target_fps > 0.0) {
auto frame_duration = std::chrono::duration_cast<std::chrono::steady_clock::duration>(std::chrono::duration<double>(1.0 / target_fps));
next_frame_time = std::chrono::steady_clock::now() + frame_duration;
} else {
next_frame_time = std::chrono::steady_clock::time_point{};
}
}
@@ -281,9 +282,7 @@ private:
std::condition_variable_any event_cv;
std::jthread worker_thread;
std::chrono::steady_clock::time_point start_time{};
u64 frame_counter{};
double current_target_fps{};
std::chrono::steady_clock::time_point next_frame_time{};
};
} // namespace Vulkan
+37 -16
View File
@@ -146,6 +146,25 @@ void Swapchain::Create(
{
is_outdated = false;
is_suboptimal = false;
switch (Settings::values.frame_pacing_mode.GetValue()) {
case Settings::FramePacingMode::Target_Auto:
scheduler.ResetFramePacing();
break;
case Settings::FramePacingMode::Target_30:
scheduler.ResetFramePacing(30.0);
break;
case Settings::FramePacingMode::Target_60:
scheduler.ResetFramePacing(60.0);
break;
case Settings::FramePacingMode::Target_120:
scheduler.ResetFramePacing(120.0);
break;
case Settings::FramePacingMode::Target_240:
scheduler.ResetFramePacing(240.0);
break;
}
width = width_;
height = height_;
#ifdef ANDROID
@@ -194,22 +213,24 @@ bool Swapchain::AcquireNextImage() {
break;
}
switch (Settings::values.frame_pacing_mode.GetValue()) {
case Settings::FramePacingMode::Target_Auto:
scheduler.Wait(resource_ticks[image_index]);
break;
case Settings::FramePacingMode::Target_30:
scheduler.Wait(resource_ticks[image_index], 30.0);
break;
case Settings::FramePacingMode::Target_60:
scheduler.Wait(resource_ticks[image_index], 60.0);
break;
case Settings::FramePacingMode::Target_90:
scheduler.Wait(resource_ticks[image_index], 90.0);
break;
case Settings::FramePacingMode::Target_120:
scheduler.Wait(resource_ticks[image_index], 120.0);
break;
if (resource_ticks[image_index] != 0 && !scheduler.IsFree(resource_ticks[image_index])) {
switch (Settings::values.frame_pacing_mode.GetValue()) {
case Settings::FramePacingMode::Target_Auto:
scheduler.Wait(resource_ticks[image_index]);
break;
case Settings::FramePacingMode::Target_30:
scheduler.Wait(resource_ticks[image_index], 30.0);
break;
case Settings::FramePacingMode::Target_60:
scheduler.Wait(resource_ticks[image_index], 60.0);
break;
case Settings::FramePacingMode::Target_120:
scheduler.Wait(resource_ticks[image_index], 120.0);
break;
case Settings::FramePacingMode::Target_240:
scheduler.Wait(resource_ticks[image_index], 240.0);
break;
}
}
resource_ticks[image_index] = scheduler.CurrentTick();
@@ -47,10 +47,32 @@ using VideoCommon::SubresourceRange;
using VideoCore::Surface::BytesPerBlock;
using VideoCore::Surface::HasAlpha;
using VideoCore::Surface::IsPixelFormatASTC;
using VideoCore::Surface::IsPixelFormatBCn;
using VideoCore::Surface::IsPixelFormatSRGB;
using VideoCore::Surface::IsPixelFormatInteger;
using VideoCore::Surface::SurfaceType;
namespace {
PixelFormat StorageCompatibleBaseFormat(const Device& device, PixelFormat format) {
if (!IsPixelFormatSRGB(format)) {
return format;
}
PixelFormat candidate = format;
switch (format) {
case PixelFormat::A8B8G8R8_SRGB:
candidate = PixelFormat::A8B8G8R8_UNORM;
break;
case PixelFormat::B8G8R8A8_SRGB:
candidate = PixelFormat::B8G8R8A8_UNORM;
break;
default:
return format;
}
const auto base_info =
MaxwellToVK::SurfaceFormat(device, FormatType::Optimal, false, candidate);
return base_info.storage ? candidate : format;
}
constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
if (color == std::array<float, 4>{0, 0, 0, 0}) {
return VK_BORDER_COLOR_FLOAT_TRANSPARENT_BLACK;
@@ -107,10 +129,20 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
PixelFormat format) {
VkImageUsageFlags usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT |
VK_IMAGE_USAGE_SAMPLED_BIT;
const auto can_have_storage_bit = [&]() {
if (IsPixelFormatASTC(format) || IsPixelFormatBCn(format) || IsPixelFormatSRGB(format)) {
return false;
}
return info.storage &&
VideoCore::Surface::GetFormatType(format) == SurfaceType::ColorTexture;
};
if (info.attachable) {
switch (VideoCore::Surface::GetFormatType(format)) {
case VideoCore::Surface::SurfaceType::ColorTexture:
usage |= VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT;
if (can_have_storage_bit()) {
usage |= VK_IMAGE_USAGE_STORAGE_BIT;
}
break;
case VideoCore::Surface::SurfaceType::Depth:
case VideoCore::Surface::SurfaceType::Stencil:
@@ -128,9 +160,21 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
return usage;
}
[[nodiscard]] VkImageUsageFlags ImageUsageFlags(const Device& device,
const MaxwellToVK::FormatInfo& info,
PixelFormat format) {
VkImageUsageFlags usage = ImageUsageFlags(info, format);
// ASTC recompression requires STORAGE_BIT for the GPU decoder pass
if (IsPixelFormatASTC(format) && !device.IsOptimalAstcSupported()) {
usage |= VK_IMAGE_USAGE_STORAGE_BIT;
}
return usage;
}
[[nodiscard]] VkImageCreateInfo MakeImageCreateInfo(const Device& device, const ImageInfo& info) {
const PixelFormat base_format = StorageCompatibleBaseFormat(device, info.format);
const auto format_info =
MaxwellToVK::SurfaceFormat(device, FormatType::Optimal, false, info.format);
MaxwellToVK::SurfaceFormat(device, FormatType::Optimal, false, base_format);
VkImageCreateFlags flags{};
if (info.type == ImageType::e2D && info.resources.layers >= 6 &&
info.size.width == info.size.height && !device.HasBrokenCubeImageCompatibility()) {
@@ -155,7 +199,7 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
.arrayLayers = static_cast<u32>(info.resources.layers),
.samples = ConvertSampleCount(info.num_samples),
.tiling = VK_IMAGE_TILING_OPTIMAL,
.usage = ImageUsageFlags(format_info, info.format),
.usage = ImageUsageFlags(device, format_info, base_format),
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
.queueFamilyIndexCount = 0,
.pQueueFamilyIndices = nullptr,
@@ -184,8 +228,45 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
return allocator.CreateImage(image_ci);
}
[[nodiscard]] VkFormat ConvertUintToUnormFormat(VkFormat format) {
// Convert UINT formats to UNORM equivalents for sampling compatibility
// Shaders expect FLOAT component types for samplers, not UINT
switch (format) {
case VK_FORMAT_A8B8G8R8_UINT_PACK32:
return VK_FORMAT_A8B8G8R8_UNORM_PACK32;
case VK_FORMAT_A2B10G10R10_UINT_PACK32:
return VK_FORMAT_A2B10G10R10_UNORM_PACK32;
case VK_FORMAT_R8_UINT:
return VK_FORMAT_R8_UNORM;
case VK_FORMAT_R16_UINT:
return VK_FORMAT_R16_UNORM;
case VK_FORMAT_R8G8_UINT:
return VK_FORMAT_R8G8_UNORM;
case VK_FORMAT_R16G16_UINT:
return VK_FORMAT_R16G16_UNORM;
case VK_FORMAT_R16G16B16A16_UINT:
return VK_FORMAT_R16G16B16A16_UNORM;
default:
return format;
}
}
[[nodiscard]] vk::ImageView MakeStorageView(const vk::Device& device, u32 level, VkImage image,
VkFormat format) {
// GLSL storage images in host shaders commonly use rgba8, which maps to VK_FORMAT_R8G8B8A8_UNORM.
// Some guest formats are represented as A8B8G8R8 in Vulkan; using that directly here triggers
// format mismatch warnings and undefined writes on vkCmdDispatch.
const VkFormat storage_view_format = [&] {
switch (format) {
case VK_FORMAT_A8B8G8R8_UNORM_PACK32:
return VK_FORMAT_R8G8B8A8_UNORM;
case VK_FORMAT_A8B8G8R8_UINT_PACK32:
return VK_FORMAT_R8G8B8A8_UNORM;
default:
return format;
}
}();
static constexpr VkImageViewUsageCreateInfo storage_image_view_usage_create_info{
.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_USAGE_CREATE_INFO,
.pNext = nullptr,
@@ -197,7 +278,7 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
.flags = 0,
.image = image,
.viewType = VK_IMAGE_VIEW_TYPE_2D_ARRAY,
.format = format,
.format = storage_view_format,
.components{
.r = VK_COMPONENT_SWIZZLE_IDENTITY,
.g = VK_COMPONENT_SWIZZLE_IDENTITY,
@@ -524,19 +605,24 @@ struct RangedBarrierRange {
max_layer = (std::max)(max_layer, layers.baseArrayLayer + layers.layerCount);
}
VkImageSubresourceRange SubresourceRange(VkImageAspectFlags aspect_mask) const noexcept {
VkImageSubresourceRange SubresourceRange(VkImageAspectFlags aspect_mask, bool is_3d = false) const noexcept {
u32 layer_count = max_layer - min_layer;
// For 3D images with 2D_ARRAY compatibility, use VK_REMAINING_ARRAY_LAYERS for single layer
if (is_3d && layer_count == 1) {
layer_count = VK_REMAINING_ARRAY_LAYERS;
}
return VkImageSubresourceRange{
.aspectMask = aspect_mask,
.baseMipLevel = min_mip,
.levelCount = max_mip - min_mip,
.baseArrayLayer = min_layer,
.layerCount = max_layer - min_layer,
.layerCount = layer_count,
};
}
};
void CopyBufferToImage(vk::CommandBuffer cmdbuf, VkBuffer src_buffer, VkImage image,
VkImageAspectFlags aspect_mask, bool is_initialized,
std::span<const VkBufferImageCopy> copies) {
std::span<const VkBufferImageCopy> copies, bool is_3d_image = false) {
static constexpr VkAccessFlags WRITE_ACCESS_FLAGS =
VK_ACCESS_SHADER_WRITE_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT |
VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT;
@@ -549,7 +635,7 @@ void CopyBufferToImage(vk::CommandBuffer cmdbuf, VkBuffer src_buffer, VkImage im
for (const auto& region : copies) {
range.AddLayers(region.imageSubresource);
}
const VkImageSubresourceRange subresource_range = range.SubresourceRange(aspect_mask);
const VkImageSubresourceRange subresource_range = range.SubresourceRange(aspect_mask, is_3d_image);
const VkImageMemoryBarrier read_barrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
@@ -692,10 +778,16 @@ void TryTransformSwizzleIfNeeded(PixelFormat format, std::array<SwizzleSource, 4
return VK_FORMAT_R16_SINT;
case Shader::ImageFormat::R32_UINT:
return VK_FORMAT_R32_UINT;
case Shader::ImageFormat::R32_SINT:
return VK_FORMAT_R32_SINT;
case Shader::ImageFormat::R32G32_UINT:
return VK_FORMAT_R32G32_UINT;
case Shader::ImageFormat::R32G32_SINT:
return VK_FORMAT_R32G32_SINT;
case Shader::ImageFormat::R32G32B32A32_UINT:
return VK_FORMAT_R32G32B32A32_UINT;
case Shader::ImageFormat::R32G32B32A32_SINT:
return VK_FORMAT_R32G32B32A32_SINT;
}
ASSERT_MSG(false, "Invalid image format={}", format);
return VK_FORMAT_R32_UINT;
@@ -879,16 +971,29 @@ TextureCacheRuntime::TextureCacheRuntime(const Device& device_, Scheduler& sched
return;
}
for (size_t index_a = 0; index_a < VideoCore::Surface::MaxPixelFormat; index_a++) {
auto add_view_format = [&](VkFormat format) {
auto& formats = view_formats[index_a];
if (std::ranges::find(formats, format) == formats.end()) {
formats.push_back(format);
}
};
const auto image_format = static_cast<PixelFormat>(index_a);
if (IsPixelFormatASTC(image_format) && !device.IsOptimalAstcSupported()) {
view_formats[index_a].push_back(VK_FORMAT_A8B8G8R8_UNORM_PACK32);
add_view_format(VK_FORMAT_A8B8G8R8_UNORM_PACK32);
add_view_format(VK_FORMAT_R8G8B8A8_UNORM);
}
for (size_t index_b = 0; index_b < VideoCore::Surface::MaxPixelFormat; index_b++) {
const auto view_format = static_cast<PixelFormat>(index_b);
if (VideoCore::Surface::IsViewCompatible(image_format, view_format, false, true)) {
const auto view_info =
MaxwellToVK::SurfaceFormat(device, FormatType::Optimal, true, view_format);
view_formats[index_a].push_back(view_info.format);
add_view_format(view_info.format);
if (view_info.format == VK_FORMAT_A8B8G8R8_UNORM_PACK32 ||
view_info.format == VK_FORMAT_A8B8G8R8_UINT_PACK32) {
add_view_format(VK_FORMAT_R8G8B8A8_UNORM);
}
}
}
}
@@ -1023,9 +1128,11 @@ void TextureCacheRuntime::ReinterpretImage(Image& dst, Image& src,
const VkBuffer copy_buffer = GetTemporaryBuffer(total_size);
const VkImage dst_image = dst.Handle();
const VkImage src_image = src.Handle();
const bool dst_is_3d = dst.info.type == ImageType::e3D;
const bool src_is_3d = src.info.type == ImageType::e3D;
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([dst_image, src_image, copy_buffer, src_aspect_mask, dst_aspect_mask,
vk_in_copies, vk_out_copies](vk::CommandBuffer cmdbuf) {
vk_in_copies, vk_out_copies, dst_is_3d, src_is_3d](vk::CommandBuffer cmdbuf) {
RangedBarrierRange dst_range;
RangedBarrierRange src_range;
for (const VkBufferImageCopy& copy : vk_in_copies) {
@@ -1060,7 +1167,7 @@ void TextureCacheRuntime::ReinterpretImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = src_image,
.subresourceRange = src_range.SubresourceRange(src_aspect_mask),
.subresourceRange = src_range.SubresourceRange(src_aspect_mask, src_is_3d),
},
};
const std::array middle_in_barrier{
@@ -1074,7 +1181,7 @@ void TextureCacheRuntime::ReinterpretImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = src_image,
.subresourceRange = src_range.SubresourceRange(src_aspect_mask),
.subresourceRange = src_range.SubresourceRange(src_aspect_mask, src_is_3d),
},
};
const std::array middle_out_barrier{
@@ -1090,7 +1197,7 @@ void TextureCacheRuntime::ReinterpretImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
.subresourceRange = dst_range.SubresourceRange(dst_aspect_mask),
.subresourceRange = dst_range.SubresourceRange(dst_aspect_mask, dst_is_3d),
},
};
const std::array post_barriers{
@@ -1109,7 +1216,7 @@ void TextureCacheRuntime::ReinterpretImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
.subresourceRange = dst_range.SubresourceRange(dst_aspect_mask),
.subresourceRange = dst_range.SubresourceRange(dst_aspect_mask, dst_is_3d),
},
};
const VkPipelineStageFlags src_stages_transfer =
@@ -1497,8 +1604,10 @@ void TextureCacheRuntime::CopyImage(Image& dst, Image& src,
});
const VkImage dst_image = dst.Handle();
const VkImage src_image = src.Handle();
const bool src_is_3d = src.info.type == ImageType::e3D;
const bool dst_is_3d = dst.info.type == ImageType::e3D;
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([dst_image, src_image, aspect_mask, vk_copies](vk::CommandBuffer cmdbuf) {
scheduler.Record([dst_image, src_image, aspect_mask, vk_copies, src_is_3d, dst_is_3d](vk::CommandBuffer cmdbuf) {
RangedBarrierRange dst_range;
RangedBarrierRange src_range;
for (const VkImageCopy& copy : vk_copies) {
@@ -1518,7 +1627,7 @@ void TextureCacheRuntime::CopyImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = src_image,
.subresourceRange = src_range.SubresourceRange(aspect_mask),
.subresourceRange = src_range.SubresourceRange(aspect_mask, src_is_3d),
},
VkImageMemoryBarrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
@@ -1532,7 +1641,7 @@ void TextureCacheRuntime::CopyImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
.subresourceRange = dst_range.SubresourceRange(aspect_mask),
.subresourceRange = dst_range.SubresourceRange(aspect_mask, dst_is_3d),
},
};
const std::array post_barriers{
@@ -1546,7 +1655,7 @@ void TextureCacheRuntime::CopyImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = src_image,
.subresourceRange = src_range.SubresourceRange(aspect_mask),
.subresourceRange = src_range.SubresourceRange(aspect_mask, src_is_3d),
},
VkImageMemoryBarrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
@@ -1563,7 +1672,7 @@ void TextureCacheRuntime::CopyImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
.subresourceRange = dst_range.SubresourceRange(aspect_mask),
.subresourceRange = dst_range.SubresourceRange(aspect_mask, dst_is_3d),
},
};
cmdbuf.PipelineBarrier(
@@ -1611,8 +1720,10 @@ void TextureCacheRuntime::TickFrame() {}
Image::Image(TextureCacheRuntime& runtime_, const ImageInfo& info_, GPUVAddr gpu_addr_,
VAddr cpu_addr_)
: VideoCommon::ImageBase(info_, gpu_addr_, cpu_addr_), scheduler{&runtime_.scheduler},
runtime{&runtime_}, original_image(MakeImage(runtime_.device, runtime_.memory_allocator, info,
runtime->ViewFormats(info.format))),
runtime{&runtime_},
original_image(MakeImage(runtime_.device, runtime_.memory_allocator, info,
runtime->ViewFormats(StorageCompatibleBaseFormat(
runtime_.device, info.format)))),
aspect_mask(ImageAspectMask(info.format)) {
if (IsPixelFormatASTC(info.format) && !runtime->device.IsOptimalAstcSupported()) {
switch (Settings::values.accelerate_astc.GetValue()) {
@@ -1641,14 +1752,13 @@ Image::Image(TextureCacheRuntime& runtime_, const ImageInfo& info_, GPUVAddr gpu
}
current_image = &Image::original_image;
storage_image_views.resize(info.resources.levels);
if (IsPixelFormatASTC(info.format) && !runtime->device.IsOptimalAstcSupported() &&
Settings::values.astc_recompression.GetValue() ==
Settings::AstcRecompression::Uncompressed) {
const auto& device = runtime->device.GetLogical();
for (s32 level = 0; level < info.resources.levels; ++level) {
storage_image_views[level] =
MakeStorageView(device, level, *original_image, VK_FORMAT_A8B8G8R8_UNORM_PACK32);
}
// Transition render targets to GENERAL layout
const auto format_info =
MaxwellToVK::SurfaceFormat(runtime->device, FormatType::Optimal, false, info.format);
const VkImageUsageFlags usage = ImageUsageFlags(runtime->device, format_info, info.format);
if ((usage & (VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT)) != 0) {
runtime->TransitionImageLayout(*this);
}
}
@@ -1745,7 +1855,7 @@ void Image::UploadMemory(VkBuffer buffer, VkDeviceSize offset,
scheduler->Record([src_buffer, temp_vk_image, vk_aspect_mask, vk_copies,
keep = temp_wrapper](vk::CommandBuffer cmdbuf) {
CopyBufferToImage(cmdbuf, src_buffer, temp_vk_image, vk_aspect_mask, false, VideoCommon::FixSmallVectorADL(vk_copies));
CopyBufferToImage(cmdbuf, src_buffer, temp_vk_image, vk_aspect_mask, false, VideoCommon::FixSmallVectorADL(vk_copies), false);
});
// Use MSAACopyPass to convert from non-MSAA to MSAA
@@ -1781,10 +1891,11 @@ void Image::UploadMemory(VkBuffer buffer, VkDeviceSize offset,
const VkImage vk_image = *original_image;
const VkImageAspectFlags vk_aspect_mask = aspect_mask;
const bool was_initialized = std::exchange(initialized, true);
const bool is_3d = info.type == ImageType::e3D;
scheduler->Record([src_buffer, vk_image, vk_aspect_mask, was_initialized,
vk_copies](vk::CommandBuffer cmdbuf) {
CopyBufferToImage(cmdbuf, src_buffer, vk_image, vk_aspect_mask, was_initialized, VideoCommon::FixSmallVectorADL(vk_copies));
vk_copies, is_3d](vk::CommandBuffer cmdbuf) {
CopyBufferToImage(cmdbuf, src_buffer, vk_image, vk_aspect_mask, was_initialized, VideoCommon::FixSmallVectorADL(vk_copies), is_3d);
});
if (is_rescaled) {
@@ -2025,8 +2136,15 @@ void Image::DownloadMemory(const StagingBufferRef& map, std::span<const BufferIm
}
VkImageView Image::StorageImageView(s32 level) noexcept {
// Storage views require the image to have been created with VK_IMAGE_USAGE_STORAGE_BIT
if (!(original_image.UsageFlags() & VK_IMAGE_USAGE_STORAGE_BIT)) {
// Image doesn't support storage usage, return null
return nullptr;
}
auto& view = storage_image_views[level];
if (!view) {
// Ensure image is in VK_IMAGE_LAYOUT_GENERAL before using as storage image
runtime->TransitionImageLayout(*this);
const auto format_info =
MaxwellToVK::SurfaceFormat(runtime->device, FormatType::Optimal, true, info.format);
view = MakeStorageView(runtime->device.GetLogical(), level, *(this->*current_image),
@@ -2059,6 +2177,9 @@ bool Image::ScaleUp(bool ignore) {
scaled_info.size.height = scaled_height;
scaled_image = MakeImage(runtime->device, runtime->memory_allocator, scaled_info,
runtime->ViewFormats(info.format));
const VkImageAspectFlags init_aspect =
aspect_mask != 0 ? aspect_mask : ImageAspectMask(info.format);
runtime->TransitionImageLayout(*scaled_image, init_aspect);
ignore = false;
}
current_image = &Image::scaled_image;
@@ -2178,6 +2299,9 @@ ImageView::ImageView(TextureCacheRuntime& runtime, const VideoCommon::ImageViewI
samples(ConvertSampleCount(image.info.num_samples)) {
using Shader::TextureType;
// Ensure image is transitioned to GENERAL layout before creating views
runtime.TransitionImageLayout(image);
const VkImageAspectFlags aspect_mask = ImageViewAspectMask(info);
std::array<SwizzleSource, 4> swizzle{
SwizzleSource::R,
@@ -2193,7 +2317,13 @@ ImageView::ImageView(TextureCacheRuntime& runtime, const VideoCommon::ImageViewI
std::ranges::transform(swizzle, swizzle.begin(), ConvertGreenRed);
}
}
const auto format_info = MaxwellToVK::SurfaceFormat(*device, FormatType::Optimal, true, format);
auto format_info = MaxwellToVK::SurfaceFormat(*device, FormatType::Optimal, true, format);
// Convert UINT formats to UNORM for sampling compatibility when not a render target
if (!info.IsRenderTarget()) {
format_info.format = ConvertUintToUnormFormat(format_info.format);
}
if (ImageUsageFlags(format_info, format) != image.UsageFlags()) {
LOG_WARNING(Render_Vulkan,
"Image view format {} has different usage flags than image format {}", format,
@@ -2202,7 +2332,7 @@ ImageView::ImageView(TextureCacheRuntime& runtime, const VideoCommon::ImageViewI
const VkImageViewUsageCreateInfo image_view_usage{
.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_USAGE_CREATE_INFO,
.pNext = nullptr,
.usage = ImageUsageFlags(format_info, format),
.usage = ImageUsageFlags(format_info, format) & image.UsageFlags(),
};
const VkImageViewCreateInfo create_info{
.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO,
@@ -2283,6 +2413,7 @@ ImageView::ImageView(TextureCacheRuntime& runtime, const VideoCommon::NullImageV
null_image = MakeImage(*device, runtime.memory_allocator, info, {});
image_handle = *null_image;
runtime.TransitionImageLayout(*null_image, VK_IMAGE_ASPECT_COLOR_BIT);
for (u32 i = 0; i < Shader::NUM_TEXTURE_TYPES; i++) {
image_views[i] = MakeView(VK_FORMAT_A8B8G8R8_UNORM_PACK32, VK_IMAGE_ASPECT_COLOR_BIT);
}
@@ -2328,13 +2459,20 @@ VkImageView ImageView::ColorView() {
VkImageView ImageView::StorageView(Shader::TextureType texture_type,
Shader::ImageFormat image_format) {
if (image_handle) {
if (!storage_views) {
storage_views.emplace();
}
if (image_format == Shader::ImageFormat::Typeless) {
return Handle(texture_type);
auto& view{storage_views->typeless[size_t(texture_type)]};
if (!view) {
const auto& format_info =
MaxwellToVK::SurfaceFormat(*device, FormatType::Optimal, false, format);
view = MakeView(format_info.format, VK_IMAGE_ASPECT_COLOR_BIT);
}
return *view;
}
const bool is_signed = image_format == Shader::ImageFormat::R8_SINT
|| image_format == Shader::ImageFormat::R16_SINT;
if (!storage_views)
storage_views.emplace();
auto& views{is_signed ? storage_views->signeds : storage_views->unsigneds};
auto& view{views[size_t(texture_type)]};
if (!view)
@@ -2561,41 +2699,36 @@ void TextureCacheRuntime::AccelerateImageUpload(
}
void TextureCacheRuntime::TransitionImageLayout(Image& image) {
if (!image.ExchangeInitialization()) {
VkImageMemoryBarrier barrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
.pNext = nullptr,
.srcAccessMask = VK_ACCESS_NONE,
.dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT |
VK_ACCESS_COLOR_ATTACHMENT_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT |
VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT,
.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED,
.newLayout = VK_IMAGE_LAYOUT_GENERAL,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = image.Handle(),
.subresourceRange{
.aspectMask = image.AspectMask(),
.baseMipLevel = 0,
.levelCount = VK_REMAINING_MIP_LEVELS,
.baseArrayLayer = 0,
.layerCount = VK_REMAINING_ARRAY_LAYERS,
},
};
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([barrier](vk::CommandBuffer cmdbuf) {
// After layout transition, image may be used in shaders or as attachment
const VkPipelineStageFlags dst_stages_layout =
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT |
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT |
VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT |
VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT |
VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT |
VK_PIPELINE_STAGE_TRANSFER_BIT;
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT,
dst_stages_layout, 0, barrier);
});
if (image.ExchangeInitialization()) {
return;
}
TransitionImageLayout(image.Handle(), image.AspectMask());
}
void TextureCacheRuntime::TransitionImageLayout(VkImage image, VkImageAspectFlags aspect_mask) {
VkImageMemoryBarrier barrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
.pNext = nullptr,
.srcAccessMask = VK_ACCESS_NONE,
.dstAccessMask = VK_ACCESS_MEMORY_READ_BIT | VK_ACCESS_MEMORY_WRITE_BIT,
.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED,
.newLayout = VK_IMAGE_LAYOUT_GENERAL,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = image,
.subresourceRange{
.aspectMask = aspect_mask,
.baseMipLevel = 0,
.levelCount = VK_REMAINING_MIP_LEVELS,
.baseArrayLayer = 0,
.layerCount = VK_REMAINING_ARRAY_LAYERS,
},
};
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([barrier](vk::CommandBuffer cmdbuf) {
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT,
VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, barrier);
});
}
} // namespace Vulkan
@@ -97,6 +97,7 @@ public:
void InsertUploadMemoryBarrier() {}
void TransitionImageLayout(Image& image);
void TransitionImageLayout(VkImage image, VkImageAspectFlags aspect_mask);
bool HasBrokenTextureViewFormats() const noexcept {
// No known Vulkan driver has broken image views
@@ -371,6 +372,7 @@ private:
struct StorageViews {
std::array<vk::ImageView, Shader::NUM_TEXTURE_TYPES> signeds;
std::array<vk::ImageView, Shader::NUM_TEXTURE_TYPES> unsigneds;
std::array<vk::ImageView, Shader::NUM_TEXTURE_TYPES> typeless;
};
[[nodiscard]] vk::ImageView MakeView(VkFormat vk_format, VkImageAspectFlags aspect_mask);
@@ -317,6 +317,11 @@ public:
return properties.properties.limits.minStorageBufferOffsetAlignment;
}
/// Returns texel buffer offset alignment requirement.
VkDeviceSize GetTexelBufferOffsetAlignment() const {
return properties.properties.limits.minTexelBufferOffsetAlignment;
}
/// Returns the maximum range for storage buffers.
VkDeviceSize GetMaxStorageBufferRange() const {
return properties.properties.limits.maxStorageBufferRange;
@@ -127,6 +127,7 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
X(vkCmdPipelineBarrier);
X(vkCmdPushConstants);
X(vkCmdPushDescriptorSetWithTemplateKHR);
X(vkCmdResetQueryPool);
X(vkCmdSetBlendConstants);
X(vkCmdSetDepthBias);
X(vkCmdSetDepthBias2EXT);
@@ -228,6 +228,7 @@ struct DeviceDispatch : InstanceDispatch {
PFN_vkCmdPushConstants vkCmdPushConstants{};
PFN_vkCmdPushDescriptorSetWithTemplateKHR vkCmdPushDescriptorSetWithTemplateKHR{};
PFN_vkCmdResolveImage vkCmdResolveImage{};
PFN_vkCmdResetQueryPool vkCmdResetQueryPool{};
PFN_vkCmdSetBlendConstants vkCmdSetBlendConstants{};
PFN_vkCmdSetCullModeEXT vkCmdSetCullModeEXT{};
PFN_vkCmdSetDepthBias vkCmdSetDepthBias{};
@@ -1164,6 +1165,10 @@ public:
dld->vkCmdBeginQuery(handle, query_pool, query, flags);
}
void ResetQueryPool(VkQueryPool query_pool, u32 first, u32 count) const noexcept {
dld->vkCmdResetQueryPool(handle, query_pool, first, count);
}
void EndQuery(VkQueryPool query_pool, u32 query) const noexcept {
dld->vkCmdEndQuery(handle, query_pool, query);
}
+5 -1
View File
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: 2016 Citra Emulator Project
@@ -57,6 +57,10 @@ void ConfigureDebug::SetConfiguration() {
#endif
// Immutable after starting
ui->serial_battery_edit->setEnabled(runtime_lock);
ui->serial_battery_edit->setText(QString::fromStdString(Settings::values.serial_battery.GetValue()));
ui->serial_board_edit->setEnabled(runtime_lock);
ui->serial_board_edit->setText(QString::fromStdString(Settings::values.serial_unit.GetValue()));
ui->homebrew_args_edit->setEnabled(runtime_lock);
ui->homebrew_args_edit->setText(QString::fromStdString(Settings::values.program_args.GetValue()));
ui->toggle_console->setEnabled(runtime_lock);