Compare commits

...

17 Commits

Author SHA1 Message Date
xbzk 6b36035a1b fix annoying ninja if constexpr warning 2026-02-15 04:01:13 -03:00
xbzk 56f0701b5a VUID-vkCmdPipelineBarrier-dstAccessMask-02816 2026-02-15 00:40:39 -03:00
xbzk 9ab3e9f482 VUID-vkCmdBeginRenderPass-initialLayout-00900 2026-02-15 00:40:39 -03:00
xbzk c372c88813 VUID-vkCmdDrawIndexedIndirect-imageLayout-00344 2026-02-15 00:40:39 -03:00
xbzk f46321da03 VK_FORMAT_R8G8B8A8_UNORM 2026-02-14 23:31:56 -03:00
xbzk 5fd780f721 layerCount_VK_REMAINING_ARRAY_LAYERS 2026-02-14 23:04:28 -03:00
xbzk d03ca7181a VUID-vkCmdBeginQuery-None-00807 2026-02-14 23:03:42 -03:00
xbzk a08114bd5d minTexelBufferOffsetAlignment 2026-02-14 23:03:42 -03:00
xbzk aefdd92834 VUID-VkWriteDescriptorSet-descriptorType-00339 2026-02-14 23:03:42 -03:00
xbzk 3249462561 VUID-VkWriteDescriptorSet-descriptorType-00336+00339 2026-02-14 23:03:42 -03:00
xbzk 60fd8380fb VUID-vkCmdDraw-None-09600 2026-02-14 23:03:20 -03:00
xbzk 068c272e27 VUID-VkImageViewCreateInfo-pNext-02662 2026-02-14 23:02:33 -03:00
xbzk 3ee14ed99a VUID-vkCmdDrawIndexed-format-07753 (FLOAT/SINT/UINT mistmatches) 2026-02-14 23:02:33 -03:00
xbzk 42451f0c3d versionwise workgroup vars 2026-02-14 23:02:33 -03:00
xbzk 1ffa381d74 vulkan vkQueue dispute fix 2026-02-14 23:02:33 -03:00
xbzk a72f444753 spirv logs cleanup 2 2026-02-14 23:02:33 -03:00
xbzk 2ba40a452f spirv log cleanup 1 2026-02-14 23:02:32 -03:00
32 changed files with 594 additions and 154 deletions
+2 -1
View File
@@ -25,7 +25,8 @@ void AssertFailSoftImpl();
#define ASSERT_MSG(_a_, ...) \
([&]() YUZU_NO_INLINE { \
if (!(_a_)) [[unlikely]] { \
auto&& _a_eval_ = (_a_); \
if (!(_a_eval_)) [[unlikely]] { \
LOG_CRITICAL(Debug, __FILE__ ": assert " __VA_ARGS__); \
AssertFailSoftImpl(); \
} \
@@ -206,10 +206,16 @@ std::string_view FormatStorage(ImageFormat format) {
return "S16";
case ImageFormat::R32_UINT:
return "U32";
case ImageFormat::R32_SINT:
return "S32";
case ImageFormat::R32G32_UINT:
return "U32X2";
case ImageFormat::R32G32_SINT:
return "S32X2";
case ImageFormat::R32G32B32A32_UINT:
return "U32X4";
case ImageFormat::R32G32B32A32_SINT:
return "S32X4";
}
throw InvalidArgument("Invalid image format {}", format);
}
@@ -148,10 +148,16 @@ std::string_view ImageFormatString(ImageFormat format) {
return ",r16i";
case ImageFormat::R32_UINT:
return ",r32ui";
case ImageFormat::R32_SINT:
return ",r32i";
case ImageFormat::R32G32_UINT:
return ",rg32ui";
case ImageFormat::R32G32_SINT:
return ",rg32i";
case ImageFormat::R32G32B32A32_UINT:
return ",rgba32ui";
case ImageFormat::R32G32B32A32_SINT:
return ",rgba32i";
default:
throw NotImplementedException("Image format: {}", format);
}
@@ -142,15 +142,22 @@ Id GetCbuf(EmitContext& ctx, Id result_type, Id UniformDefinitions::*member_ptr,
const auto is_float = UniformDefinitions::IsFloat(member_ptr);
const auto num_elements = UniformDefinitions::NumElements(member_ptr);
const std::array zero_vec{
is_float ? ctx.Const(0.0f) : ctx.Const(0u),
is_float ? ctx.Const(0.0f) : ctx.Const(0u),
is_float ? ctx.Const(0.0f) : ctx.Const(0u),
is_float ? ctx.Const(0.0f) : ctx.Const(0u),
};
const Id zero_element = is_float ? ctx.Const(0.0f) : ctx.Const(0u);
const Id cond = ctx.OpULessThanEqual(ctx.TypeBool(), buffer_offset, ctx.Const(0xFFFFu));
const Id zero = ctx.OpCompositeConstruct(result_type, std::span(zero_vec.data(), num_elements));
return ctx.OpSelect(result_type, cond, val, zero);
// OpSelect with vector result requires vector condition, scalar uses scalar directly
if (num_elements > 1) {
const std::array zero_vec{zero_element, zero_element, zero_element, zero_element};
const Id zero = ctx.OpCompositeConstruct(result_type, std::span(zero_vec.data(), num_elements));
const Id bool_vector_type = ctx.TypeVector(ctx.U1, num_elements);
const std::array cond_vec{cond, cond, cond, cond};
const Id vector_cond = ctx.OpCompositeConstruct(bool_vector_type, std::span(cond_vec.data(), num_elements));
return ctx.OpSelect(result_type, vector_cond, val, zero);
} else {
return ctx.OpSelect(result_type, cond, val, zero_element);
}
}
Id GetCbufU32(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset) {
@@ -205,7 +205,7 @@ Id TextureImage(EmitContext& ctx, IR::TextureInstInfo info, const IR::Value& ind
if (def.count > 1) {
throw NotImplementedException("Indirect texture sample");
}
return ctx.OpLoad(ctx.image_buffer_type, def.id);
return ctx.OpLoad(def.image_type, def.id);
} else {
const TextureDefinition& def{ctx.textures.at(info.descriptor_index)};
if (def.count > 1) {
@@ -215,16 +215,22 @@ Id TextureImage(EmitContext& ctx, IR::TextureInstInfo info, const IR::Value& ind
}
}
std::pair<Id, bool> Image(EmitContext& ctx, const IR::Value& index, IR::TextureInstInfo info) {
struct ImageInfo {
Id image;
bool is_integer;
bool is_signed;
};
ImageInfo Image(EmitContext& ctx, const IR::Value& index, IR::TextureInstInfo info) {
if (!index.IsImmediate() || index.U32() != 0) {
throw NotImplementedException("Indirect image indexing");
}
if (info.type == TextureType::Buffer) {
const ImageBufferDefinition def{ctx.image_buffers.at(info.descriptor_index)};
return {ctx.OpLoad(def.image_type, def.id), def.is_integer};
return {ctx.OpLoad(def.image_type, def.id), def.is_integer, def.is_signed};
} else {
const ImageDefinition def{ctx.images.at(info.descriptor_index)};
return {ctx.OpLoad(def.image_type, def.id), def.is_integer};
return {ctx.OpLoad(def.image_type, def.id), def.is_integer, def.is_signed};
}
}
@@ -550,8 +556,25 @@ Id EmitImageFetch(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id c
lod = Id{};
}
const ImageOperands operands(lod, ms);
return Emit(&EmitContext::OpImageSparseFetch, &EmitContext::OpImageFetch, ctx, inst, ctx.F32[4],
TextureImage(ctx, info, index), coords, operands.MaskOptional(), operands.Span());
bool is_integer = false;
bool is_signed = false;
Id result_type{ctx.F32[4]};
if (info.type == TextureType::Buffer) {
const TextureBufferDefinition& def{ctx.texture_buffers.at(info.descriptor_index)};
is_integer = def.is_integer;
is_signed = def.is_signed;
if (is_integer) {
result_type = is_signed ? ctx.S32[4] : ctx.U32[4];
}
}
Id fetched = Emit(&EmitContext::OpImageSparseFetch, &EmitContext::OpImageFetch, ctx, inst,
result_type, TextureImage(ctx, info, index), coords,
operands.MaskOptional(), operands.Span());
if (is_integer) {
// IR expects F32x4 from ImageFetch; bitcast integer results to float vector.
fetched = ctx.OpBitcast(ctx.F32[4], fetched);
}
return fetched;
}
Id EmitImageQueryDimensions(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id lod,
@@ -612,21 +635,25 @@ Id EmitImageRead(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id co
LOG_WARNING(Shader_SPIRV, "Typeless image read not supported by host");
return ctx.ConstantNull(ctx.U32[4]);
}
const auto [image, is_integer] = Image(ctx, index, info);
const Id result_type{is_integer ? ctx.U32[4] : ctx.F32[4]};
const auto [image, is_integer, is_signed] = Image(ctx, index, info);
const Id result_type{is_integer ? (is_signed ? ctx.S32[4] : ctx.U32[4]) : ctx.F32[4]};
Id color{Emit(&EmitContext::OpImageSparseRead, &EmitContext::OpImageRead, ctx, inst,
result_type, image, coords, std::nullopt, std::span<const Id>{})};
if (!is_integer) {
color = ctx.OpBitcast(ctx.U32[4], color);
} else if (is_signed) {
color = ctx.OpBitcast(ctx.U32[4], color);
}
return color;
}
void EmitImageWrite(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords, Id color) {
const auto info{inst->Flags<IR::TextureInstInfo>()};
const auto [image, is_integer] = Image(ctx, index, info);
const auto [image, is_integer, is_signed] = Image(ctx, index, info);
if (!is_integer) {
color = ctx.OpBitcast(ctx.F32[4], color);
} else if (is_signed) {
color = ctx.OpBitcast(ctx.S32[4], color);
}
ctx.OpImageWrite(image, coords, color);
}
@@ -7,13 +7,18 @@
namespace Shader::Backend::SPIRV {
namespace {
Id Image(EmitContext& ctx, IR::TextureInstInfo info) {
struct ImageInfo {
Id id;
bool is_signed;
};
ImageInfo Image(EmitContext& ctx, IR::TextureInstInfo info) {
if (info.type == TextureType::Buffer) {
const ImageBufferDefinition def{ctx.image_buffers.at(info.descriptor_index)};
return def.id;
return {def.id, def.is_signed};
} else {
const ImageDefinition def{ctx.images.at(info.descriptor_index)};
return def.id;
return {def.id, def.is_signed};
}
}
@@ -24,42 +29,66 @@ std::pair<Id, Id> AtomicArgs(EmitContext& ctx) {
}
Id ImageAtomicU32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords, Id value,
Id (Sirit::Module::*atomic_func)(Id, Id, Id, Id, Id)) {
Id (Sirit::Module::*atomic_func)(Id, Id, Id, Id, Id), bool value_signed) {
if (!index.IsImmediate() || index.U32() != 0) {
// TODO: handle layers
throw NotImplementedException("Image indexing");
}
const auto info{inst->Flags<IR::TextureInstInfo>()};
const Id image{Image(ctx, info)};
const Id pointer{ctx.OpImageTexelPointer(ctx.image_u32, image, coords, ctx.Const(0U))};
const auto image_info{Image(ctx, info)};
Id pointer_type{image_info.is_signed ? ctx.image_s32 : ctx.image_u32};
if (!Sirit::ValidId(pointer_type)) {
const Id element_type{image_info.is_signed ? ctx.S32[1] : ctx.U32[1]};
pointer_type = ctx.TypePointer(spv::StorageClass::Image, element_type);
}
const Id image{image_info.id};
const Id pointer{ctx.OpImageTexelPointer(pointer_type, image, coords, ctx.Const(0U))};
const auto [scope, semantics]{AtomicArgs(ctx)};
return (ctx.*atomic_func)(ctx.U32[1], pointer, scope, semantics, value);
const Id result_type{image_info.is_signed ? ctx.S32[1] : ctx.U32[1]};
// Ensure value type matches result_type's pointee type
Id cast_value{value};
if (image_info.is_signed) {
// Result type is signed s32, ensure value is also s32
cast_value = ctx.OpBitcast(ctx.S32[1], value);
} else {
// Result type is unsigned u32, ensure value is also u32
cast_value = ctx.OpBitcast(ctx.U32[1], value);
}
Id result{(ctx.*atomic_func)(result_type, pointer, scope, semantics, cast_value)};
// Convert result back to u32 for IR compatibility
if (image_info.is_signed) {
result = ctx.OpBitcast(ctx.U32[1], result);
}
return result;
}
} // Anonymous namespace
Id EmitImageAtomicIAdd32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicIAdd);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicIAdd, false);
}
Id EmitImageAtomicSMin32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicSMin);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicSMin, true);
}
Id EmitImageAtomicUMin32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicUMin);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicUMin, false);
}
Id EmitImageAtomicSMax32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicSMax);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicSMax, true);
}
Id EmitImageAtomicUMax32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicUMax);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicUMax, false);
}
Id EmitImageAtomicInc32(EmitContext&, IR::Inst*, const IR::Value&, Id, Id) {
@@ -74,22 +103,22 @@ Id EmitImageAtomicDec32(EmitContext&, IR::Inst*, const IR::Value&, Id, Id) {
Id EmitImageAtomicAnd32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicAnd);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicAnd, false);
}
Id EmitImageAtomicOr32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicOr);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicOr, false);
}
Id EmitImageAtomicXor32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicXor);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicXor, false);
}
Id EmitImageAtomicExchange32(EmitContext& ctx, IR::Inst* inst, const IR::Value& index, Id coords,
Id value) {
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicExchange);
return ImageAtomicU32(ctx, inst, index, coords, value, &Sirit::Module::OpAtomicExchange, false);
}
Id EmitBindlessImageAtomicIAdd32(EmitContext&) {
@@ -69,10 +69,16 @@ spv::ImageFormat GetImageFormat(ImageFormat format) {
return spv::ImageFormat::R16i;
case ImageFormat::R32_UINT:
return spv::ImageFormat::R32ui;
case ImageFormat::R32_SINT:
return spv::ImageFormat::R32i;
case ImageFormat::R32G32_UINT:
return spv::ImageFormat::Rg32ui;
case ImageFormat::R32G32_SINT:
return spv::ImageFormat::Rg32i;
case ImageFormat::R32G32B32A32_UINT:
return spv::ImageFormat::Rgba32ui;
case ImageFormat::R32G32B32A32_SINT:
return spv::ImageFormat::Rgba32i;
}
throw InvalidArgument("Invalid image format {}", format);
}
@@ -613,7 +619,10 @@ void EmitContext::DefineSharedMemory(const IR::Program& program) {
const Id element_pointer{TypePointer(spv::StorageClass::Workgroup, element_type)};
const Id variable{AddGlobalVariable(pointer, spv::StorageClass::Workgroup)};
Decorate(variable, spv::Decoration::Aliased);
interfaces.push_back(variable);
// Workgroup variables in EntryPoint interfaces are only supported in SPIR-V 1.4+
if (profile.supported_spirv >= 0x00010400) {
interfaces.push_back(variable);
}
return std::make_tuple(variable, element_pointer, pointer);
}};
@@ -642,7 +651,10 @@ void EmitContext::DefineSharedMemory(const IR::Program& program) {
shared_u32 = TypePointer(spv::StorageClass::Workgroup, U32[1]);
shared_memory_u32 = AddGlobalVariable(shared_memory_u32_type, spv::StorageClass::Workgroup);
interfaces.push_back(shared_memory_u32);
// Workgroup variables in EntryPoint interfaces are only supported in SPIR-V 1.4+
if (profile.supported_spirv >= 0x00010400) {
interfaces.push_back(shared_memory_u32);
}
const Id func_type{TypeFunction(void_id, U32[1], U32[1])};
const auto make_function{[&](u32 mask, u32 size) {
@@ -1305,20 +1317,25 @@ void EmitContext::DefineTextureBuffers(const Info& info, u32& binding) {
return;
}
const spv::ImageFormat format{spv::ImageFormat::Unknown};
image_buffer_type = TypeImage(F32[1], spv::Dim::Buffer, 0U, false, false, 1, format);
const Id type{TypePointer(spv::StorageClass::UniformConstant, image_buffer_type)};
texture_buffers.reserve(info.texture_buffer_descriptors.size());
for (const TextureBufferDescriptor& desc : info.texture_buffer_descriptors) {
if (desc.count != 1) {
throw NotImplementedException("Array of texture buffers");
}
// Use the correct sampled type based on the descriptor's data format
const Id sampled_type{desc.is_integer ? (desc.is_signed ? S32[1] : U32[1]) : F32[1]};
image_buffer_type = TypeImage(sampled_type, spv::Dim::Buffer, 0U, false, false, 1, format);
const Id type{TypePointer(spv::StorageClass::UniformConstant, image_buffer_type)};
const Id id{AddGlobalVariable(type, spv::StorageClass::UniformConstant)};
Decorate(id, spv::Decoration::Binding, binding);
Decorate(id, spv::Decoration::DescriptorSet, 0U);
Name(id, NameOf(stage, desc, "texbuf"));
texture_buffers.push_back({
.id = id,
.image_type = image_buffer_type,
.is_integer = desc.is_integer,
.is_signed = desc.is_signed,
.count = desc.count,
});
if (profile.supported_spirv >= 0x00010400) {
@@ -1335,19 +1352,23 @@ void EmitContext::DefineImageBuffers(const Info& info, u32& binding) {
throw NotImplementedException("Array of image buffers");
}
const spv::ImageFormat format{GetImageFormat(desc.format)};
const Id sampled_type{desc.is_integer ? U32[1] : F32[1]};
const Id sampled_type{desc.is_integer ? (desc.is_signed ? S32[1] : U32[1]) : F32[1]};
const Id image_type{
TypeImage(sampled_type, spv::Dim::Buffer, false, false, false, 2, format)};
const Id pointer_type{TypePointer(spv::StorageClass::UniformConstant, image_type)};
const Id id{AddGlobalVariable(pointer_type, spv::StorageClass::UniformConstant)};
Decorate(id, spv::Decoration::Binding, binding);
Decorate(id, spv::Decoration::DescriptorSet, 0U);
if (format == spv::ImageFormat::Unknown) {
Decorate(id, spv::Decoration::NonReadable);
}
Name(id, NameOf(stage, desc, "imgbuf"));
image_buffers.push_back({
.id = id,
.image_type = image_type,
.count = desc.count,
.is_integer = desc.is_integer,
.is_signed = desc.is_signed,
});
if (profile.supported_spirv >= 0x00010400) {
interfaces.push_back(id);
@@ -1384,6 +1405,9 @@ void EmitContext::DefineTextures(const Info& info, u32& binding, u32& scaling_in
if (info.uses_atomic_image_u32) {
image_u32 = TypePointer(spv::StorageClass::Image, U32[1]);
}
if (info.uses_atomic_s32_min || info.uses_atomic_s32_max) {
image_s32 = TypePointer(spv::StorageClass::Image, S32[1]);
}
}
void EmitContext::DefineImages(const Info& info, u32& binding, u32& scaling_index) {
@@ -1392,18 +1416,23 @@ void EmitContext::DefineImages(const Info& info, u32& binding, u32& scaling_inde
if (desc.count != 1) {
throw NotImplementedException("Array of images");
}
const Id sampled_type{desc.is_integer ? U32[1] : F32[1]};
const Id sampled_type{desc.is_integer ? (desc.is_signed ? S32[1] : U32[1]) : F32[1]};
const Id image_type{ImageType(*this, desc, sampled_type)};
const Id pointer_type{TypePointer(spv::StorageClass::UniformConstant, image_type)};
const Id id{AddGlobalVariable(pointer_type, spv::StorageClass::UniformConstant)};
Decorate(id, spv::Decoration::Binding, binding);
Decorate(id, spv::Decoration::DescriptorSet, 0U);
const spv::ImageFormat format{GetImageFormat(desc.format)};
if (format == spv::ImageFormat::Unknown) {
Decorate(id, spv::Decoration::NonReadable);
}
Name(id, NameOf(stage, desc, "img"));
images.push_back({
.id = id,
.image_type = image_type,
.count = desc.count,
.is_integer = desc.is_integer,
.is_signed = desc.is_signed,
});
if (profile.supported_spirv >= 0x00010400) {
interfaces.push_back(id);
@@ -45,6 +45,9 @@ struct TextureDefinition {
struct TextureBufferDefinition {
Id id;
Id image_type; // Stores the correct buffer image type (F32 or U32 based on descriptor)
bool is_integer;
bool is_signed;
u32 count;
};
@@ -53,6 +56,7 @@ struct ImageBufferDefinition {
Id image_type;
u32 count;
bool is_integer;
bool is_signed;
};
struct ImageDefinition {
@@ -60,6 +64,7 @@ struct ImageDefinition {
Id image_type;
u32 count;
bool is_integer;
bool is_signed;
};
struct UniformDefinitions {
@@ -250,6 +255,7 @@ public:
Id image_buffer_type{};
Id image_u32{};
Id image_s32{};
std::array<UniformDefinitions, Info::MAX_CBUFS> cbufs{};
std::array<StorageDefinitions, Info::MAX_SSBOS> ssbos{};
+74 -2
View File
@@ -204,6 +204,65 @@ static inline bool IsTexturePixelFormatIntegerCached(Environment& env,
return env.IsTexturePixelFormatInteger(GetTextureHandleCached(env, cbuf));
}
static inline bool IsTexturePixelFormatSignedCached(Environment& env,
const ConstBufferAddr& cbuf) {
switch (ReadTexturePixelFormatCached(env, cbuf)) {
case TexturePixelFormat::A8B8G8R8_SINT:
case TexturePixelFormat::R8_SINT:
case TexturePixelFormat::R8G8_SINT:
case TexturePixelFormat::R16_SINT:
case TexturePixelFormat::R16G16_SINT:
case TexturePixelFormat::R16G16B16A16_SINT:
case TexturePixelFormat::R32_SINT:
case TexturePixelFormat::R32G32_SINT:
case TexturePixelFormat::R32G32B32A32_SINT:
return true;
default:
return false;
}
}
static inline bool IsImageFormatSigned(ImageFormat format) {
switch (format) {
case ImageFormat::R8_SINT:
case ImageFormat::R16_SINT:
case ImageFormat::R32_SINT:
case ImageFormat::R32G32_SINT:
case ImageFormat::R32G32B32A32_SINT:
return true;
default:
return false;
}
}
static inline std::optional<ImageFormat> BufferImageFormatFromPixelFormat(
TexturePixelFormat pixel_format) {
switch (pixel_format) {
case TexturePixelFormat::R8_UINT:
return ImageFormat::R8_UINT;
case TexturePixelFormat::R8_SINT:
return ImageFormat::R8_SINT;
case TexturePixelFormat::R16_UINT:
return ImageFormat::R16_UINT;
case TexturePixelFormat::R16_SINT:
return ImageFormat::R16_SINT;
case TexturePixelFormat::R32_UINT:
return ImageFormat::R32_UINT;
case TexturePixelFormat::R32_SINT:
return ImageFormat::R32_SINT;
case TexturePixelFormat::R32G32_UINT:
return ImageFormat::R32G32_UINT;
case TexturePixelFormat::R32G32_SINT:
return ImageFormat::R32G32_SINT;
case TexturePixelFormat::R32G32B32A32_UINT:
return ImageFormat::R32G32B32A32_UINT;
case TexturePixelFormat::R32G32B32A32_SINT:
return ImageFormat::R32G32B32A32_SINT;
default:
return std::nullopt;
}
}
std::optional<ConstBufferAddr> Track(const IR::Value& value, Environment& env);
static inline std::optional<ConstBufferAddr> TrackCached(const IR::Value& v, Environment& env) {
@@ -652,12 +711,21 @@ void TexturePass(Environment& env, IR::Program& program, const HostTranslateInfo
const bool is_written{inst->GetOpcode() != IR::Opcode::ImageRead};
const bool is_read{inst->GetOpcode() != IR::Opcode::ImageWrite};
const bool is_integer{IsTexturePixelFormatIntegerCached(env, cbuf)};
ImageFormat image_format = flags.image_format;
if (flags.type == TextureType::Buffer) {
const auto pixel_format = ReadTexturePixelFormatCached(env, cbuf);
if (const auto mapped = BufferImageFormatFromPixelFormat(pixel_format)) {
image_format = *mapped;
}
}
const bool is_signed{IsImageFormatSigned(image_format)};
if (flags.type == TextureType::Buffer) {
index = descriptors.Add(ImageBufferDescriptor{
.format = flags.image_format,
.format = image_format,
.is_written = is_written,
.is_read = is_read,
.is_integer = is_integer,
.is_signed = is_signed,
.cbuf_index = cbuf.index,
.cbuf_offset = cbuf.offset,
.count = cbuf.count,
@@ -666,22 +734,26 @@ void TexturePass(Environment& env, IR::Program& program, const HostTranslateInfo
} else {
index = descriptors.Add(ImageDescriptor{
.type = flags.type,
.format = flags.image_format,
.format = image_format,
.is_written = is_written,
.is_read = is_read,
.is_integer = is_integer,
.is_signed = is_signed,
.cbuf_index = cbuf.index,
.cbuf_offset = cbuf.offset,
.count = cbuf.count,
.size_shift = DESCRIPTOR_SIZE_SHIFT,
});
}
flags.image_format.Assign(image_format);
break;
}
default:
if (flags.type == TextureType::Buffer) {
index = descriptors.Add(TextureBufferDescriptor{
.has_secondary = cbuf.has_secondary,
.is_integer = IsTexturePixelFormatIntegerCached(env, cbuf),
.is_signed = IsTexturePixelFormatSignedCached(env, cbuf),
.cbuf_index = cbuf.index,
.cbuf_offset = cbuf.offset,
.shift_left = cbuf.shift_left,
+7
View File
@@ -150,8 +150,11 @@ enum class ImageFormat : u32 {
R16_UINT,
R16_SINT,
R32_UINT,
R32_SINT,
R32G32_UINT,
R32G32_SINT,
R32G32B32A32_UINT,
R32G32B32A32_SINT,
};
enum class Interpolation {
@@ -178,6 +181,8 @@ struct StorageBufferDescriptor {
struct TextureBufferDescriptor {
bool has_secondary;
bool is_integer; // True if data is SINT/UINT (from R_type in TIC), false if FLOAT
bool is_signed; // True if integer data is signed
u32 cbuf_index;
u32 cbuf_offset;
u32 shift_left;
@@ -196,6 +201,7 @@ struct ImageBufferDescriptor {
bool is_written;
bool is_read;
bool is_integer;
bool is_signed;
u32 cbuf_index;
u32 cbuf_offset;
u32 count;
@@ -229,6 +235,7 @@ struct ImageDescriptor {
bool is_written;
bool is_read;
bool is_integer;
bool is_signed;
u32 cbuf_index;
u32 cbuf_offset;
u32 count;
+19 -9
View File
@@ -458,14 +458,14 @@ void BufferCache<P>::UnbindGraphicsTextureBuffers(size_t stage) {
template <class P>
void BufferCache<P>::BindGraphicsTextureBuffer(size_t stage, size_t tbo_index, GPUVAddr gpu_addr,
u32 size, PixelFormat format, bool is_written,
bool is_image) {
bool is_image, bool is_integer, bool is_signed) {
channel_state->enabled_texture_buffers[stage] |= 1U << tbo_index;
channel_state->written_texture_buffers[stage] |= (is_written ? 1U : 0U) << tbo_index;
if constexpr (SEPARATE_IMAGE_BUFFERS_BINDINGS) {
channel_state->image_texture_buffers[stage] |= (is_image ? 1U : 0U) << tbo_index;
}
channel_state->texture_buffers[stage][tbo_index] =
GetTextureBufferBinding(gpu_addr, size, format);
GetTextureBufferBinding(gpu_addr, size, format, is_integer, is_signed);
}
template <class P>
@@ -532,7 +532,8 @@ void BufferCache<P>::UnbindComputeTextureBuffers() {
template <class P>
void BufferCache<P>::BindComputeTextureBuffer(size_t tbo_index, GPUVAddr gpu_addr, u32 size,
PixelFormat format, bool is_written, bool is_image) {
PixelFormat format, bool is_written, bool is_image,
bool is_integer, bool is_signed) {
if (tbo_index >= channel_state->compute_texture_buffers.size()) [[unlikely]] {
LOG_ERROR(HW_GPU, "Texture buffer index {} exceeds maximum texture buffer count",
tbo_index);
@@ -544,7 +545,7 @@ void BufferCache<P>::BindComputeTextureBuffer(size_t tbo_index, GPUVAddr gpu_add
channel_state->image_compute_texture_buffers |= (is_image ? 1U : 0U) << tbo_index;
}
channel_state->compute_texture_buffers[tbo_index] =
GetTextureBufferBinding(gpu_addr, size, format);
GetTextureBufferBinding(gpu_addr, size, format, is_integer, is_signed);
}
template <class P>
@@ -955,15 +956,17 @@ void BufferCache<P>::BindHostGraphicsTextureBuffers(size_t stage) {
const u32 offset = buffer.Offset(binding.device_addr);
const PixelFormat format = binding.format;
const bool is_integer = binding.is_integer;
const bool is_signed = binding.is_signed;
buffer.MarkUsage(offset, size);
if constexpr (SEPARATE_IMAGE_BUFFERS_BINDINGS) {
if (((channel_state->image_texture_buffers[stage] >> index) & 1) != 0) {
runtime.BindImageBuffer(buffer, offset, size, format);
} else {
runtime.BindTextureBuffer(buffer, offset, size, format);
runtime.BindTextureBuffer(buffer, offset, size, format, is_integer, is_signed);
}
} else {
runtime.BindTextureBuffer(buffer, offset, size, format);
runtime.BindTextureBuffer(buffer, offset, size, format, is_integer, is_signed);
}
});
}
@@ -1090,15 +1093,17 @@ void BufferCache<P>::BindHostComputeTextureBuffers() {
const u32 offset = buffer.Offset(binding.device_addr);
const PixelFormat format = binding.format;
const bool is_integer = binding.is_integer;
const bool is_signed = binding.is_signed;
buffer.MarkUsage(offset, size);
if constexpr (SEPARATE_IMAGE_BUFFERS_BINDINGS) {
if (((channel_state->image_compute_texture_buffers >> index) & 1) != 0) {
runtime.BindImageBuffer(buffer, offset, size, format);
} else {
runtime.BindTextureBuffer(buffer, offset, size, format);
runtime.BindTextureBuffer(buffer, offset, size, format, is_integer, is_signed);
}
} else {
runtime.BindTextureBuffer(buffer, offset, size, format);
runtime.BindTextureBuffer(buffer, offset, size, format, is_integer, is_signed);
}
});
}
@@ -1833,7 +1838,8 @@ Binding BufferCache<P>::StorageBufferBinding(GPUVAddr ssbo_addr, u32 cbuf_index,
template <class P>
TextureBufferBinding BufferCache<P>::GetTextureBufferBinding(GPUVAddr gpu_addr, u32 size,
PixelFormat format) {
PixelFormat format, bool is_integer,
bool is_signed) {
const std::optional<DAddr> device_addr = gpu_memory->GpuToCpuAddress(gpu_addr);
TextureBufferBinding binding;
if (!device_addr || size == 0) {
@@ -1841,11 +1847,15 @@ TextureBufferBinding BufferCache<P>::GetTextureBufferBinding(GPUVAddr gpu_addr,
binding.size = 0;
binding.buffer_id = NULL_BUFFER_ID;
binding.format = PixelFormat::Invalid;
binding.is_integer = false;
binding.is_signed = false;
} else {
binding.device_addr = *device_addr;
binding.size = size;
binding.buffer_id = BufferId{};
binding.format = format;
binding.is_integer = is_integer;
binding.is_signed = is_signed;
}
return binding;
}
@@ -84,6 +84,8 @@ struct Binding {
struct TextureBufferBinding : Binding {
PixelFormat format;
bool is_integer{}; // True if data is SINT/UINT, false if FLOAT
bool is_signed{}; // True if integer data is signed
};
static constexpr Binding NULL_BINDING{
@@ -251,7 +253,8 @@ public:
void UnbindGraphicsTextureBuffers(size_t stage);
void BindGraphicsTextureBuffer(size_t stage, size_t tbo_index, GPUVAddr gpu_addr, u32 size,
PixelFormat format, bool is_written, bool is_image);
PixelFormat format, bool is_written, bool is_image,
bool is_integer, bool is_signed);
void UnbindComputeStorageBuffers();
@@ -261,7 +264,7 @@ public:
void UnbindComputeTextureBuffers();
void BindComputeTextureBuffer(size_t tbo_index, GPUVAddr gpu_addr, u32 size, PixelFormat format,
bool is_written, bool is_image);
bool is_written, bool is_image, bool is_integer, bool is_signed);
[[nodiscard]] std::pair<Buffer*, u32> ObtainBuffer(GPUVAddr gpu_addr, u32 size,
ObtainBufferSynchronize sync_info,
@@ -445,7 +448,8 @@ private:
bool is_written) const;
[[nodiscard]] TextureBufferBinding GetTextureBufferBinding(GPUVAddr gpu_addr, u32 size,
PixelFormat format);
PixelFormat format, bool is_integer,
bool is_signed);
[[nodiscard]] std::span<const u8> ImmediateBufferWithData(DAddr device_addr, size_t size);
@@ -88,7 +88,7 @@ void Buffer::MakeResident(GLenum access) noexcept {
glMakeNamedBufferResidentNV(buffer.handle, access);
}
GLuint Buffer::View(u32 offset, u32 size, PixelFormat format) {
GLuint Buffer::View(u32 offset, u32 size, PixelFormat format, bool is_integer, bool is_signed) {
const auto it{std::ranges::find_if(views, [offset, size, format](const BufferView& view) {
return offset == view.offset && size == view.size && format == view.format;
})};
@@ -370,8 +370,8 @@ void BufferCacheRuntime::BindTransformFeedbackBuffers(VideoCommon::HostBindings<
}
void BufferCacheRuntime::BindTextureBuffer(Buffer& buffer, u32 offset, u32 size,
PixelFormat format) {
*texture_handles++ = buffer.View(offset, size, format);
PixelFormat format, bool is_integer, bool is_signed) {
*texture_handles++ = buffer.View(offset, size, format, is_integer, is_signed);
}
void BufferCacheRuntime::BindImageBuffer(Buffer& buffer, u32 offset, u32 size, PixelFormat format) {
@@ -34,7 +34,8 @@ public:
void MarkUsage(u64 offset, u64 size) {}
[[nodiscard]] GLuint View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format);
[[nodiscard]] GLuint View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format,
bool is_integer = false, bool is_signed = false);
[[nodiscard]] GLuint64EXT HostGpuAddr() const noexcept {
return address;
@@ -118,7 +119,7 @@ public:
void BindTransformFeedbackBuffers(VideoCommon::HostBindings<Buffer>& bindings);
void BindTextureBuffer(Buffer& buffer, u32 offset, u32 size,
VideoCore::Surface::PixelFormat format);
VideoCore::Surface::PixelFormat format, bool is_integer, bool is_signed);
void BindImageBuffer(Buffer& buffer, u32 offset, u32 size,
VideoCore::Surface::PixelFormat format);
@@ -177,7 +177,8 @@ void ComputePipeline::Configure() {
ImageView& image_view{texture_cache.GetImageView(views[texbuf_index].id)};
buffer_cache.BindComputeTextureBuffer(texbuf_index, image_view.GpuAddr(),
image_view.BufferSize(), image_view.format,
is_written, is_image);
is_written, is_image, desc.is_integer,
desc.is_signed);
++texbuf_index;
}
}};
@@ -397,7 +397,8 @@ bool GraphicsPipeline::ConfigureImpl(bool is_indexed) {
ImageView& image_view{texture_cache.GetImageView(texture_buffer_it->id)};
buffer_cache.BindGraphicsTextureBuffer(stage, index, image_view.GpuAddr(),
image_view.BufferSize(), image_view.format,
is_written, is_image);
is_written, is_image, desc.is_integer,
desc.is_signed);
++index;
++texture_buffer_it;
}
@@ -433,10 +433,16 @@ OGLTexture MakeImage(const VideoCommon::ImageInfo& info, GLenum gl_internal_form
return GL_R16I;
case Shader::ImageFormat::R32_UINT:
return GL_R32UI;
case Shader::ImageFormat::R32_SINT:
return GL_R32I;
case Shader::ImageFormat::R32G32_UINT:
return GL_RG32UI;
case Shader::ImageFormat::R32G32_SINT:
return GL_RG32I;
case Shader::ImageFormat::R32G32B32A32_UINT:
return GL_RGBA32UI;
case Shader::ImageFormat::R32G32B32A32_SINT:
return GL_RGBA32I;
}
ASSERT_MSG(false, "Invalid image format={}", format);
return GL_R32UI;
@@ -177,6 +177,10 @@ try
RendererVulkan::~RendererVulkan() {
scheduler.RegisterOnSubmit([] {});
scheduler.WaitWorker();
// vkDeviceWaitIdle MUST be called only after all queue submissions are complete
// to avoid threading errors on VkQueue simultaneous access
std::scoped_lock lock{scheduler.submit_mutex};
void(device.GetLogical().WaitIdle());
}
@@ -30,6 +30,9 @@ BlitScreen::~BlitScreen() = default;
void BlitScreen::WaitIdle() {
present_manager.WaitPresent();
scheduler.Finish();
std::scoped_lock lock{scheduler.submit_mutex};
// vkDeviceWaitIdle MUST be protected by submit_mutex to prevent racing with queue submissions
// from the worker thread. This ensures no simultaneous access to VkQueue.
device.GetLogical().WaitIdle();
}
@@ -83,7 +83,7 @@ vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allo
} // Anonymous namespace
Buffer::Buffer(BufferCacheRuntime& runtime, VideoCommon::NullBufferParams null_params)
: VideoCommon::BufferBase(null_params), tracker{4096} {
: VideoCommon::BufferBase(null_params), runtime{&runtime}, tracker{4096} {
if (runtime.device.HasNullDescriptor()) {
return;
}
@@ -93,14 +93,53 @@ Buffer::Buffer(BufferCacheRuntime& runtime, VideoCommon::NullBufferParams null_p
}
Buffer::Buffer(BufferCacheRuntime& runtime, DAddr cpu_addr_, u64 size_bytes_)
: VideoCommon::BufferBase(cpu_addr_, size_bytes_), device{&runtime.device},
: VideoCommon::BufferBase(cpu_addr_, size_bytes_), runtime{&runtime}, device{&runtime.device},
buffer{CreateBuffer(*device, runtime.memory_allocator, SizeBytes())}, tracker{SizeBytes()} {
if (runtime.device.HasDebuggingToolAttached()) {
buffer.SetObjectNameEXT(fmt::format("Buffer 0x{:x}", CpuAddr()).c_str());
}
}
VkBufferView Buffer::View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format) {
VkFormat SelectTexelBufferFormat(VkFormat float_format, bool is_integer, bool is_signed) {
// If the buffer stores integer data but Vulkan reports float format,
// we need to map to appropriate integer formats for type compatibility
if (!is_integer) {
// Non-integer buffer, use the original float format
return float_format;
}
// Integer buffer: map float formats to signed/unsigned equivalents
if (is_signed) {
// Signed integer
switch (float_format) {
case VK_FORMAT_R32_SFLOAT:
return VK_FORMAT_R32_SINT;
case VK_FORMAT_R32G32_SFLOAT:
return VK_FORMAT_R32G32_SINT;
case VK_FORMAT_R32G32B32A32_SFLOAT:
return VK_FORMAT_R32G32B32A32_SINT;
default:
// For non-float formats, use as-is
return float_format;
}
} else {
// Unsigned integer
switch (float_format) {
case VK_FORMAT_R32_SFLOAT:
return VK_FORMAT_R32_UINT;
case VK_FORMAT_R32G32_SFLOAT:
return VK_FORMAT_R32G32_UINT;
case VK_FORMAT_R32G32B32A32_SFLOAT:
return VK_FORMAT_R32G32B32A32_UINT;
default:
// For non-float formats, use as-is
return float_format;
}
}
}
VkBufferView Buffer::View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format,
bool is_integer, bool is_signed) {
if (!device) {
// Null buffer supported, return a null descriptor
return VK_NULL_HANDLE;
@@ -109,25 +148,44 @@ VkBufferView Buffer::View(u32 offset, u32 size, VideoCore::Surface::PixelFormat
offset = 0;
size = 0;
}
const VkDeviceSize alignment = device->GetTexelBufferOffsetAlignment();
const bool needs_alignment = alignment != 0 && (offset % alignment) != 0;
const auto it{std::ranges::find_if(views, [offset, size, format](const BufferView& view) {
return offset == view.offset && size == view.size && format == view.format;
})};
if (it != views.end()) {
return *it->handle;
}
vk::Buffer backing_buffer{};
VkBuffer view_buffer = *buffer;
u32 view_offset = offset;
if (needs_alignment && runtime && size != 0) {
backing_buffer = CreateBuffer(*device, runtime->memory_allocator, size);
VideoCommon::BufferCopy copy{
.src_offset = offset,
.dst_offset = 0,
.size = size,
};
runtime->CopyBuffer(*backing_buffer, *buffer, std::span{&copy, 1}, true);
view_buffer = *backing_buffer;
view_offset = 0;
}
views.push_back({
.offset = offset,
.size = size,
.format = format,
.is_integer = is_integer,
.is_signed = is_signed,
.handle = device->GetLogical().CreateBufferView({
.sType = VK_STRUCTURE_TYPE_BUFFER_VIEW_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.buffer = *buffer,
.buffer = view_buffer,
.format = MaxwellToVK::SurfaceFormat(*device, FormatType::Buffer, false, format).format,
.offset = offset,
.offset = view_offset,
.range = size,
}),
.backing_buffer = std::move(backing_buffer),
});
return *views.back().handle;
}
@@ -494,6 +552,7 @@ void BufferCacheRuntime::PostCopyBarrier() {
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([](vk::CommandBuffer cmdbuf) {
const VkPipelineStageFlags dst_stages =
VK_PIPELINE_STAGE_DRAW_INDIRECT_BIT |
VK_PIPELINE_STAGE_VERTEX_INPUT_BIT |
VK_PIPELINE_STAGE_VERTEX_SHADER_BIT |
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT |
@@ -33,7 +33,8 @@ public:
explicit Buffer(BufferCacheRuntime&, VideoCommon::NullBufferParams null_params);
explicit Buffer(BufferCacheRuntime& runtime, VAddr cpu_addr_, u64 size_bytes_);
[[nodiscard]] VkBufferView View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format);
[[nodiscard]] VkBufferView View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format,
bool is_integer = false, bool is_signed = false);
[[nodiscard]] VkBuffer Handle() const noexcept {
return *buffer;
@@ -60,9 +61,13 @@ private:
u32 offset;
u32 size;
VideoCore::Surface::PixelFormat format;
bool is_integer;
bool is_signed;
vk::BufferView handle;
vk::Buffer backing_buffer;
};
BufferCacheRuntime* runtime{};
const Device* device{};
vk::Buffer buffer;
std::vector<BufferView> views;
@@ -149,8 +154,8 @@ public:
}
void BindTextureBuffer(Buffer& buffer, u32 offset, u32 size,
VideoCore::Surface::PixelFormat format) {
guest_descriptor_queue.AddTexelBuffer(buffer.View(offset, size, format));
VideoCore::Surface::PixelFormat format, bool is_integer, bool is_signed) {
guest_descriptor_queue.AddTexelBuffer(buffer.View(offset, size, format, is_integer, is_signed));
}
bool ShouldLimitDynamicStorageBuffers() const {
@@ -916,7 +916,7 @@ void BlockLinearUnswizzle3DPass::UnswizzleChunk(
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
.subresourceRange = {aspect, 0, 1, 0, 1},
.subresourceRange{aspect, 0, VK_REMAINING_MIP_LEVELS, 0, VK_REMAINING_ARRAY_LAYERS},
};
// Single barrier handles both buffer and image
@@ -950,7 +950,7 @@ void BlockLinearUnswizzle3DPass::UnswizzleChunk(
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
.subresourceRange = {aspect, 0, 1, 0, 1},
.subresourceRange{aspect, 0, VK_REMAINING_MIP_LEVELS, 0, VK_REMAINING_ARRAY_LAYERS},
};
cmdbuf.PipelineBarrier(
@@ -194,7 +194,8 @@ void ComputePipeline::Configure(Tegra::Engines::KeplerCompute& kepler_compute,
ImageView& image_view = texture_cache.GetImageView(views[index].id);
buffer_cache.BindComputeTextureBuffer(index, image_view.GpuAddr(),
image_view.BufferSize(), image_view.format,
is_written, is_image);
is_written, is_image, desc.is_integer,
desc.is_signed);
++index;
}
}};
@@ -422,7 +422,8 @@ bool GraphicsPipeline::ConfigureImpl(bool is_indexed) {
ImageView& image_view{texture_cache.GetImageView(texture_buffer_it->id)};
buffer_cache.BindGraphicsTextureBuffer(stage, index, image_view.GpuAddr(),
image_view.BufferSize(), image_view.format,
is_written, is_image);
is_written, is_image, desc.is_integer,
desc.is_signed);
++index;
++texture_buffer_it;
}
@@ -157,10 +157,10 @@ public:
ReserveHostQuery();
scheduler.Record([query_pool = current_query_pool,
query_index = current_bank_slot](vk::CommandBuffer cmdbuf) {
query_index = current_bank_slot](vk::CommandBuffer cmdbuf) {
const bool use_precise = Settings::IsGPULevelHigh();
cmdbuf.BeginQuery(query_pool, static_cast<u32>(query_index),
use_precise ? VK_QUERY_CONTROL_PRECISE_BIT : 0);
use_precise ? VK_QUERY_CONTROL_PRECISE_BIT : 0);
});
has_started = true;
@@ -454,6 +454,11 @@ private:
void ReserveHostQuery() {
size_t new_slot = ReserveBankSlot();
scheduler.RecordWithUploadBuffer([query_pool = current_query_pool,
query_index = current_bank_slot](vk::CommandBuffer,
vk::CommandBuffer upload_cmdbuf) {
upload_cmdbuf.ResetQueryPool(query_pool, static_cast<u32>(query_index), 1);
});
current_bank->AddReference(1);
num_slots_used++;
if (current_query) {
@@ -52,7 +52,8 @@ using VideoCore::Surface::SurfaceType;
const bool has_stencil = surface_type == SurfaceType::DepthStencil ||
surface_type == SurfaceType::Stencil;
// Use optimal layouts for attachments - this allows drivers to optimize tiling and access patterns
// Attachments are tracked as GENERAL outside render passes; render-pass begin performs
// the transition into attachment-optimal layout for the subpass.
const VkImageLayout attachment_layout = is_depth_stencil
? VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL
: VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL;
@@ -67,7 +68,7 @@ using VideoCore::Surface::SurfaceType;
: VK_ATTACHMENT_LOAD_OP_DONT_CARE,
.stencilStoreOp = has_stencil ? VK_ATTACHMENT_STORE_OP_STORE
: VK_ATTACHMENT_STORE_OP_DONT_CARE,
.initialLayout = attachment_layout,
.initialLayout = VK_IMAGE_LAYOUT_GENERAL,
.finalLayout = attachment_layout,
};
}
@@ -90,7 +91,7 @@ VkRenderPass RenderPassCache::Get(const RenderPassKey& key) {
const bool is_valid{format != PixelFormat::Invalid};
references[index] = VkAttachmentReference{
.attachment = is_valid ? num_colors : VK_ATTACHMENT_UNUSED,
.layout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL,
.layout = VK_IMAGE_LAYOUT_GENERAL,
};
if (is_valid) {
descriptions.push_back(AttachmentDescription(*device, format, key.samples, false));
@@ -103,7 +104,7 @@ VkRenderPass RenderPassCache::Get(const RenderPassKey& key) {
if (key.depth_format != PixelFormat::Invalid) {
depth_reference = VkAttachmentReference{
.attachment = num_colors,
.layout = VK_IMAGE_LAYOUT_DEPTH_STENCIL_ATTACHMENT_OPTIMAL,
.layout = VK_IMAGE_LAYOUT_GENERAL,
};
descriptions.push_back(AttachmentDescription(*device, key.depth_format, key.samples, true));
}
@@ -365,20 +365,22 @@ void Scheduler::EndRenderPass()
VkImageLayout new_layout;
if (is_color) {
// Color attachments can be read as textures or used as attachments again
// Keep GENERAL to match descriptor image layouts used across the renderer.
src_access = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT;
this_stage = VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT;
new_layout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
new_layout = VK_IMAGE_LAYOUT_GENERAL;
dst_access = VK_ACCESS_SHADER_READ_BIT
| VK_ACCESS_SHADER_WRITE_BIT
| VK_ACCESS_COLOR_ATTACHMENT_READ_BIT
| VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT;
} else if (is_depth_stencil) {
// Depth attachments can be read as textures or used as attachments again
// Keep GENERAL to match descriptor image layouts used across the renderer.
src_access = VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT;
this_stage = VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT
| VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT;
new_layout = VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
new_layout = VK_IMAGE_LAYOUT_GENERAL;
dst_access = VK_ACCESS_SHADER_READ_BIT
| VK_ACCESS_SHADER_WRITE_BIT
| VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT
| VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT;
} else {
@@ -47,10 +47,32 @@ using VideoCommon::SubresourceRange;
using VideoCore::Surface::BytesPerBlock;
using VideoCore::Surface::HasAlpha;
using VideoCore::Surface::IsPixelFormatASTC;
using VideoCore::Surface::IsPixelFormatBCn;
using VideoCore::Surface::IsPixelFormatSRGB;
using VideoCore::Surface::IsPixelFormatInteger;
using VideoCore::Surface::SurfaceType;
namespace {
PixelFormat StorageCompatibleBaseFormat(const Device& device, PixelFormat format) {
if (!IsPixelFormatSRGB(format)) {
return format;
}
PixelFormat candidate = format;
switch (format) {
case PixelFormat::A8B8G8R8_SRGB:
candidate = PixelFormat::A8B8G8R8_UNORM;
break;
case PixelFormat::B8G8R8A8_SRGB:
candidate = PixelFormat::B8G8R8A8_UNORM;
break;
default:
return format;
}
const auto base_info =
MaxwellToVK::SurfaceFormat(device, FormatType::Optimal, false, candidate);
return base_info.storage ? candidate : format;
}
constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
if (color == std::array<float, 4>{0, 0, 0, 0}) {
return VK_BORDER_COLOR_FLOAT_TRANSPARENT_BLACK;
@@ -107,10 +129,20 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
PixelFormat format) {
VkImageUsageFlags usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT |
VK_IMAGE_USAGE_SAMPLED_BIT;
const auto can_have_storage_bit = [&]() {
if (IsPixelFormatASTC(format) || IsPixelFormatBCn(format) || IsPixelFormatSRGB(format)) {
return false;
}
return info.storage &&
VideoCore::Surface::GetFormatType(format) == SurfaceType::ColorTexture;
};
if (info.attachable) {
switch (VideoCore::Surface::GetFormatType(format)) {
case VideoCore::Surface::SurfaceType::ColorTexture:
usage |= VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT;
if (can_have_storage_bit()) {
usage |= VK_IMAGE_USAGE_STORAGE_BIT;
}
break;
case VideoCore::Surface::SurfaceType::Depth:
case VideoCore::Surface::SurfaceType::Stencil:
@@ -128,9 +160,21 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
return usage;
}
[[nodiscard]] VkImageUsageFlags ImageUsageFlags(const Device& device,
const MaxwellToVK::FormatInfo& info,
PixelFormat format) {
VkImageUsageFlags usage = ImageUsageFlags(info, format);
// ASTC recompression requires STORAGE_BIT for the GPU decoder pass
if (IsPixelFormatASTC(format) && !device.IsOptimalAstcSupported()) {
usage |= VK_IMAGE_USAGE_STORAGE_BIT;
}
return usage;
}
[[nodiscard]] VkImageCreateInfo MakeImageCreateInfo(const Device& device, const ImageInfo& info) {
const PixelFormat base_format = StorageCompatibleBaseFormat(device, info.format);
const auto format_info =
MaxwellToVK::SurfaceFormat(device, FormatType::Optimal, false, info.format);
MaxwellToVK::SurfaceFormat(device, FormatType::Optimal, false, base_format);
VkImageCreateFlags flags{};
if (info.type == ImageType::e2D && info.resources.layers >= 6 &&
info.size.width == info.size.height && !device.HasBrokenCubeImageCompatibility()) {
@@ -155,7 +199,7 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
.arrayLayers = static_cast<u32>(info.resources.layers),
.samples = ConvertSampleCount(info.num_samples),
.tiling = VK_IMAGE_TILING_OPTIMAL,
.usage = ImageUsageFlags(format_info, info.format),
.usage = ImageUsageFlags(device, format_info, base_format),
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
.queueFamilyIndexCount = 0,
.pQueueFamilyIndices = nullptr,
@@ -184,8 +228,45 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
return allocator.CreateImage(image_ci);
}
[[nodiscard]] VkFormat ConvertUintToUnormFormat(VkFormat format) {
// Convert UINT formats to UNORM equivalents for sampling compatibility
// Shaders expect FLOAT component types for samplers, not UINT
switch (format) {
case VK_FORMAT_A8B8G8R8_UINT_PACK32:
return VK_FORMAT_A8B8G8R8_UNORM_PACK32;
case VK_FORMAT_A2B10G10R10_UINT_PACK32:
return VK_FORMAT_A2B10G10R10_UNORM_PACK32;
case VK_FORMAT_R8_UINT:
return VK_FORMAT_R8_UNORM;
case VK_FORMAT_R16_UINT:
return VK_FORMAT_R16_UNORM;
case VK_FORMAT_R8G8_UINT:
return VK_FORMAT_R8G8_UNORM;
case VK_FORMAT_R16G16_UINT:
return VK_FORMAT_R16G16_UNORM;
case VK_FORMAT_R16G16B16A16_UINT:
return VK_FORMAT_R16G16B16A16_UNORM;
default:
return format;
}
}
[[nodiscard]] vk::ImageView MakeStorageView(const vk::Device& device, u32 level, VkImage image,
VkFormat format) {
// GLSL storage images in host shaders commonly use rgba8, which maps to VK_FORMAT_R8G8B8A8_UNORM.
// Some guest formats are represented as A8B8G8R8 in Vulkan; using that directly here triggers
// format mismatch warnings and undefined writes on vkCmdDispatch.
const VkFormat storage_view_format = [&] {
switch (format) {
case VK_FORMAT_A8B8G8R8_UNORM_PACK32:
return VK_FORMAT_R8G8B8A8_UNORM;
case VK_FORMAT_A8B8G8R8_UINT_PACK32:
return VK_FORMAT_R8G8B8A8_UNORM;
default:
return format;
}
}();
static constexpr VkImageViewUsageCreateInfo storage_image_view_usage_create_info{
.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_USAGE_CREATE_INFO,
.pNext = nullptr,
@@ -197,7 +278,7 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
.flags = 0,
.image = image,
.viewType = VK_IMAGE_VIEW_TYPE_2D_ARRAY,
.format = format,
.format = storage_view_format,
.components{
.r = VK_COMPONENT_SWIZZLE_IDENTITY,
.g = VK_COMPONENT_SWIZZLE_IDENTITY,
@@ -524,19 +605,24 @@ struct RangedBarrierRange {
max_layer = (std::max)(max_layer, layers.baseArrayLayer + layers.layerCount);
}
VkImageSubresourceRange SubresourceRange(VkImageAspectFlags aspect_mask) const noexcept {
VkImageSubresourceRange SubresourceRange(VkImageAspectFlags aspect_mask, bool is_3d = false) const noexcept {
u32 layer_count = max_layer - min_layer;
// For 3D images with 2D_ARRAY compatibility, use VK_REMAINING_ARRAY_LAYERS for single layer
if (is_3d && layer_count == 1) {
layer_count = VK_REMAINING_ARRAY_LAYERS;
}
return VkImageSubresourceRange{
.aspectMask = aspect_mask,
.baseMipLevel = min_mip,
.levelCount = max_mip - min_mip,
.baseArrayLayer = min_layer,
.layerCount = max_layer - min_layer,
.layerCount = layer_count,
};
}
};
void CopyBufferToImage(vk::CommandBuffer cmdbuf, VkBuffer src_buffer, VkImage image,
VkImageAspectFlags aspect_mask, bool is_initialized,
std::span<const VkBufferImageCopy> copies) {
std::span<const VkBufferImageCopy> copies, bool is_3d_image = false) {
static constexpr VkAccessFlags WRITE_ACCESS_FLAGS =
VK_ACCESS_SHADER_WRITE_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT |
VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT;
@@ -549,7 +635,7 @@ void CopyBufferToImage(vk::CommandBuffer cmdbuf, VkBuffer src_buffer, VkImage im
for (const auto& region : copies) {
range.AddLayers(region.imageSubresource);
}
const VkImageSubresourceRange subresource_range = range.SubresourceRange(aspect_mask);
const VkImageSubresourceRange subresource_range = range.SubresourceRange(aspect_mask, is_3d_image);
const VkImageMemoryBarrier read_barrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
@@ -692,10 +778,16 @@ void TryTransformSwizzleIfNeeded(PixelFormat format, std::array<SwizzleSource, 4
return VK_FORMAT_R16_SINT;
case Shader::ImageFormat::R32_UINT:
return VK_FORMAT_R32_UINT;
case Shader::ImageFormat::R32_SINT:
return VK_FORMAT_R32_SINT;
case Shader::ImageFormat::R32G32_UINT:
return VK_FORMAT_R32G32_UINT;
case Shader::ImageFormat::R32G32_SINT:
return VK_FORMAT_R32G32_SINT;
case Shader::ImageFormat::R32G32B32A32_UINT:
return VK_FORMAT_R32G32B32A32_UINT;
case Shader::ImageFormat::R32G32B32A32_SINT:
return VK_FORMAT_R32G32B32A32_SINT;
}
ASSERT_MSG(false, "Invalid image format={}", format);
return VK_FORMAT_R32_UINT;
@@ -879,16 +971,29 @@ TextureCacheRuntime::TextureCacheRuntime(const Device& device_, Scheduler& sched
return;
}
for (size_t index_a = 0; index_a < VideoCore::Surface::MaxPixelFormat; index_a++) {
auto add_view_format = [&](VkFormat format) {
auto& formats = view_formats[index_a];
if (std::ranges::find(formats, format) == formats.end()) {
formats.push_back(format);
}
};
const auto image_format = static_cast<PixelFormat>(index_a);
if (IsPixelFormatASTC(image_format) && !device.IsOptimalAstcSupported()) {
view_formats[index_a].push_back(VK_FORMAT_A8B8G8R8_UNORM_PACK32);
add_view_format(VK_FORMAT_A8B8G8R8_UNORM_PACK32);
add_view_format(VK_FORMAT_R8G8B8A8_UNORM);
}
for (size_t index_b = 0; index_b < VideoCore::Surface::MaxPixelFormat; index_b++) {
const auto view_format = static_cast<PixelFormat>(index_b);
if (VideoCore::Surface::IsViewCompatible(image_format, view_format, false, true)) {
const auto view_info =
MaxwellToVK::SurfaceFormat(device, FormatType::Optimal, true, view_format);
view_formats[index_a].push_back(view_info.format);
add_view_format(view_info.format);
if (view_info.format == VK_FORMAT_A8B8G8R8_UNORM_PACK32 ||
view_info.format == VK_FORMAT_A8B8G8R8_UINT_PACK32) {
add_view_format(VK_FORMAT_R8G8B8A8_UNORM);
}
}
}
}
@@ -1023,9 +1128,11 @@ void TextureCacheRuntime::ReinterpretImage(Image& dst, Image& src,
const VkBuffer copy_buffer = GetTemporaryBuffer(total_size);
const VkImage dst_image = dst.Handle();
const VkImage src_image = src.Handle();
const bool dst_is_3d = dst.info.type == ImageType::e3D;
const bool src_is_3d = src.info.type == ImageType::e3D;
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([dst_image, src_image, copy_buffer, src_aspect_mask, dst_aspect_mask,
vk_in_copies, vk_out_copies](vk::CommandBuffer cmdbuf) {
vk_in_copies, vk_out_copies, dst_is_3d, src_is_3d](vk::CommandBuffer cmdbuf) {
RangedBarrierRange dst_range;
RangedBarrierRange src_range;
for (const VkBufferImageCopy& copy : vk_in_copies) {
@@ -1060,7 +1167,7 @@ void TextureCacheRuntime::ReinterpretImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = src_image,
.subresourceRange = src_range.SubresourceRange(src_aspect_mask),
.subresourceRange = src_range.SubresourceRange(src_aspect_mask, src_is_3d),
},
};
const std::array middle_in_barrier{
@@ -1074,7 +1181,7 @@ void TextureCacheRuntime::ReinterpretImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = src_image,
.subresourceRange = src_range.SubresourceRange(src_aspect_mask),
.subresourceRange = src_range.SubresourceRange(src_aspect_mask, src_is_3d),
},
};
const std::array middle_out_barrier{
@@ -1090,7 +1197,7 @@ void TextureCacheRuntime::ReinterpretImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
.subresourceRange = dst_range.SubresourceRange(dst_aspect_mask),
.subresourceRange = dst_range.SubresourceRange(dst_aspect_mask, dst_is_3d),
},
};
const std::array post_barriers{
@@ -1109,7 +1216,7 @@ void TextureCacheRuntime::ReinterpretImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
.subresourceRange = dst_range.SubresourceRange(dst_aspect_mask),
.subresourceRange = dst_range.SubresourceRange(dst_aspect_mask, dst_is_3d),
},
};
const VkPipelineStageFlags src_stages_transfer =
@@ -1497,8 +1604,10 @@ void TextureCacheRuntime::CopyImage(Image& dst, Image& src,
});
const VkImage dst_image = dst.Handle();
const VkImage src_image = src.Handle();
const bool src_is_3d = src.info.type == ImageType::e3D;
const bool dst_is_3d = dst.info.type == ImageType::e3D;
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([dst_image, src_image, aspect_mask, vk_copies](vk::CommandBuffer cmdbuf) {
scheduler.Record([dst_image, src_image, aspect_mask, vk_copies, src_is_3d, dst_is_3d](vk::CommandBuffer cmdbuf) {
RangedBarrierRange dst_range;
RangedBarrierRange src_range;
for (const VkImageCopy& copy : vk_copies) {
@@ -1518,7 +1627,7 @@ void TextureCacheRuntime::CopyImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = src_image,
.subresourceRange = src_range.SubresourceRange(aspect_mask),
.subresourceRange = src_range.SubresourceRange(aspect_mask, src_is_3d),
},
VkImageMemoryBarrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
@@ -1532,7 +1641,7 @@ void TextureCacheRuntime::CopyImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
.subresourceRange = dst_range.SubresourceRange(aspect_mask),
.subresourceRange = dst_range.SubresourceRange(aspect_mask, dst_is_3d),
},
};
const std::array post_barriers{
@@ -1546,7 +1655,7 @@ void TextureCacheRuntime::CopyImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = src_image,
.subresourceRange = src_range.SubresourceRange(aspect_mask),
.subresourceRange = src_range.SubresourceRange(aspect_mask, src_is_3d),
},
VkImageMemoryBarrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
@@ -1563,7 +1672,7 @@ void TextureCacheRuntime::CopyImage(Image& dst, Image& src,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
.subresourceRange = dst_range.SubresourceRange(aspect_mask),
.subresourceRange = dst_range.SubresourceRange(aspect_mask, dst_is_3d),
},
};
cmdbuf.PipelineBarrier(
@@ -1611,8 +1720,10 @@ void TextureCacheRuntime::TickFrame() {}
Image::Image(TextureCacheRuntime& runtime_, const ImageInfo& info_, GPUVAddr gpu_addr_,
VAddr cpu_addr_)
: VideoCommon::ImageBase(info_, gpu_addr_, cpu_addr_), scheduler{&runtime_.scheduler},
runtime{&runtime_}, original_image(MakeImage(runtime_.device, runtime_.memory_allocator, info,
runtime->ViewFormats(info.format))),
runtime{&runtime_},
original_image(MakeImage(runtime_.device, runtime_.memory_allocator, info,
runtime->ViewFormats(StorageCompatibleBaseFormat(
runtime_.device, info.format)))),
aspect_mask(ImageAspectMask(info.format)) {
if (IsPixelFormatASTC(info.format) && !runtime->device.IsOptimalAstcSupported()) {
switch (Settings::values.accelerate_astc.GetValue()) {
@@ -1641,14 +1752,13 @@ Image::Image(TextureCacheRuntime& runtime_, const ImageInfo& info_, GPUVAddr gpu
}
current_image = &Image::original_image;
storage_image_views.resize(info.resources.levels);
if (IsPixelFormatASTC(info.format) && !runtime->device.IsOptimalAstcSupported() &&
Settings::values.astc_recompression.GetValue() ==
Settings::AstcRecompression::Uncompressed) {
const auto& device = runtime->device.GetLogical();
for (s32 level = 0; level < info.resources.levels; ++level) {
storage_image_views[level] =
MakeStorageView(device, level, *original_image, VK_FORMAT_A8B8G8R8_UNORM_PACK32);
}
// Transition render targets to GENERAL layout
const auto format_info =
MaxwellToVK::SurfaceFormat(runtime->device, FormatType::Optimal, false, info.format);
const VkImageUsageFlags usage = ImageUsageFlags(runtime->device, format_info, info.format);
if ((usage & (VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT | VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT)) != 0) {
runtime->TransitionImageLayout(*this);
}
}
@@ -1745,7 +1855,7 @@ void Image::UploadMemory(VkBuffer buffer, VkDeviceSize offset,
scheduler->Record([src_buffer, temp_vk_image, vk_aspect_mask, vk_copies,
keep = temp_wrapper](vk::CommandBuffer cmdbuf) {
CopyBufferToImage(cmdbuf, src_buffer, temp_vk_image, vk_aspect_mask, false, VideoCommon::FixSmallVectorADL(vk_copies));
CopyBufferToImage(cmdbuf, src_buffer, temp_vk_image, vk_aspect_mask, false, VideoCommon::FixSmallVectorADL(vk_copies), false);
});
// Use MSAACopyPass to convert from non-MSAA to MSAA
@@ -1781,10 +1891,11 @@ void Image::UploadMemory(VkBuffer buffer, VkDeviceSize offset,
const VkImage vk_image = *original_image;
const VkImageAspectFlags vk_aspect_mask = aspect_mask;
const bool was_initialized = std::exchange(initialized, true);
const bool is_3d = info.type == ImageType::e3D;
scheduler->Record([src_buffer, vk_image, vk_aspect_mask, was_initialized,
vk_copies](vk::CommandBuffer cmdbuf) {
CopyBufferToImage(cmdbuf, src_buffer, vk_image, vk_aspect_mask, was_initialized, VideoCommon::FixSmallVectorADL(vk_copies));
vk_copies, is_3d](vk::CommandBuffer cmdbuf) {
CopyBufferToImage(cmdbuf, src_buffer, vk_image, vk_aspect_mask, was_initialized, VideoCommon::FixSmallVectorADL(vk_copies), is_3d);
});
if (is_rescaled) {
@@ -2025,8 +2136,15 @@ void Image::DownloadMemory(const StagingBufferRef& map, std::span<const BufferIm
}
VkImageView Image::StorageImageView(s32 level) noexcept {
// Storage views require the image to have been created with VK_IMAGE_USAGE_STORAGE_BIT
if (!(original_image.UsageFlags() & VK_IMAGE_USAGE_STORAGE_BIT)) {
// Image doesn't support storage usage, return null
return nullptr;
}
auto& view = storage_image_views[level];
if (!view) {
// Ensure image is in VK_IMAGE_LAYOUT_GENERAL before using as storage image
runtime->TransitionImageLayout(*this);
const auto format_info =
MaxwellToVK::SurfaceFormat(runtime->device, FormatType::Optimal, true, info.format);
view = MakeStorageView(runtime->device.GetLogical(), level, *(this->*current_image),
@@ -2059,6 +2177,9 @@ bool Image::ScaleUp(bool ignore) {
scaled_info.size.height = scaled_height;
scaled_image = MakeImage(runtime->device, runtime->memory_allocator, scaled_info,
runtime->ViewFormats(info.format));
const VkImageAspectFlags init_aspect =
aspect_mask != 0 ? aspect_mask : ImageAspectMask(info.format);
runtime->TransitionImageLayout(*scaled_image, init_aspect);
ignore = false;
}
current_image = &Image::scaled_image;
@@ -2178,6 +2299,9 @@ ImageView::ImageView(TextureCacheRuntime& runtime, const VideoCommon::ImageViewI
samples(ConvertSampleCount(image.info.num_samples)) {
using Shader::TextureType;
// Ensure image is transitioned to GENERAL layout before creating views
runtime.TransitionImageLayout(image);
const VkImageAspectFlags aspect_mask = ImageViewAspectMask(info);
std::array<SwizzleSource, 4> swizzle{
SwizzleSource::R,
@@ -2193,7 +2317,13 @@ ImageView::ImageView(TextureCacheRuntime& runtime, const VideoCommon::ImageViewI
std::ranges::transform(swizzle, swizzle.begin(), ConvertGreenRed);
}
}
const auto format_info = MaxwellToVK::SurfaceFormat(*device, FormatType::Optimal, true, format);
auto format_info = MaxwellToVK::SurfaceFormat(*device, FormatType::Optimal, true, format);
// Convert UINT formats to UNORM for sampling compatibility when not a render target
if (!info.IsRenderTarget()) {
format_info.format = ConvertUintToUnormFormat(format_info.format);
}
if (ImageUsageFlags(format_info, format) != image.UsageFlags()) {
LOG_WARNING(Render_Vulkan,
"Image view format {} has different usage flags than image format {}", format,
@@ -2202,7 +2332,7 @@ ImageView::ImageView(TextureCacheRuntime& runtime, const VideoCommon::ImageViewI
const VkImageViewUsageCreateInfo image_view_usage{
.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_USAGE_CREATE_INFO,
.pNext = nullptr,
.usage = ImageUsageFlags(format_info, format),
.usage = ImageUsageFlags(format_info, format) & image.UsageFlags(),
};
const VkImageViewCreateInfo create_info{
.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO,
@@ -2283,6 +2413,7 @@ ImageView::ImageView(TextureCacheRuntime& runtime, const VideoCommon::NullImageV
null_image = MakeImage(*device, runtime.memory_allocator, info, {});
image_handle = *null_image;
runtime.TransitionImageLayout(*null_image, VK_IMAGE_ASPECT_COLOR_BIT);
for (u32 i = 0; i < Shader::NUM_TEXTURE_TYPES; i++) {
image_views[i] = MakeView(VK_FORMAT_A8B8G8R8_UNORM_PACK32, VK_IMAGE_ASPECT_COLOR_BIT);
}
@@ -2328,13 +2459,20 @@ VkImageView ImageView::ColorView() {
VkImageView ImageView::StorageView(Shader::TextureType texture_type,
Shader::ImageFormat image_format) {
if (image_handle) {
if (!storage_views) {
storage_views.emplace();
}
if (image_format == Shader::ImageFormat::Typeless) {
return Handle(texture_type);
auto& view{storage_views->typeless[size_t(texture_type)]};
if (!view) {
const auto& format_info =
MaxwellToVK::SurfaceFormat(*device, FormatType::Optimal, false, format);
view = MakeView(format_info.format, VK_IMAGE_ASPECT_COLOR_BIT);
}
return *view;
}
const bool is_signed = image_format == Shader::ImageFormat::R8_SINT
|| image_format == Shader::ImageFormat::R16_SINT;
if (!storage_views)
storage_views.emplace();
auto& views{is_signed ? storage_views->signeds : storage_views->unsigneds};
auto& view{views[size_t(texture_type)]};
if (!view)
@@ -2561,41 +2699,36 @@ void TextureCacheRuntime::AccelerateImageUpload(
}
void TextureCacheRuntime::TransitionImageLayout(Image& image) {
if (!image.ExchangeInitialization()) {
VkImageMemoryBarrier barrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
.pNext = nullptr,
.srcAccessMask = VK_ACCESS_NONE,
.dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT |
VK_ACCESS_COLOR_ATTACHMENT_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT |
VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_READ_BIT | VK_ACCESS_DEPTH_STENCIL_ATTACHMENT_WRITE_BIT,
.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED,
.newLayout = VK_IMAGE_LAYOUT_GENERAL,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = image.Handle(),
.subresourceRange{
.aspectMask = image.AspectMask(),
.baseMipLevel = 0,
.levelCount = VK_REMAINING_MIP_LEVELS,
.baseArrayLayer = 0,
.layerCount = VK_REMAINING_ARRAY_LAYERS,
},
};
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([barrier](vk::CommandBuffer cmdbuf) {
// After layout transition, image may be used in shaders or as attachment
const VkPipelineStageFlags dst_stages_layout =
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT |
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT |
VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT |
VK_PIPELINE_STAGE_EARLY_FRAGMENT_TESTS_BIT |
VK_PIPELINE_STAGE_LATE_FRAGMENT_TESTS_BIT |
VK_PIPELINE_STAGE_TRANSFER_BIT;
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT,
dst_stages_layout, 0, barrier);
});
if (image.ExchangeInitialization()) {
return;
}
TransitionImageLayout(image.Handle(), image.AspectMask());
}
void TextureCacheRuntime::TransitionImageLayout(VkImage image, VkImageAspectFlags aspect_mask) {
VkImageMemoryBarrier barrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
.pNext = nullptr,
.srcAccessMask = VK_ACCESS_NONE,
.dstAccessMask = VK_ACCESS_MEMORY_READ_BIT | VK_ACCESS_MEMORY_WRITE_BIT,
.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED,
.newLayout = VK_IMAGE_LAYOUT_GENERAL,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = image,
.subresourceRange{
.aspectMask = aspect_mask,
.baseMipLevel = 0,
.levelCount = VK_REMAINING_MIP_LEVELS,
.baseArrayLayer = 0,
.layerCount = VK_REMAINING_ARRAY_LAYERS,
},
};
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([barrier](vk::CommandBuffer cmdbuf) {
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT,
VK_PIPELINE_STAGE_ALL_COMMANDS_BIT, 0, barrier);
});
}
} // namespace Vulkan
@@ -97,6 +97,7 @@ public:
void InsertUploadMemoryBarrier() {}
void TransitionImageLayout(Image& image);
void TransitionImageLayout(VkImage image, VkImageAspectFlags aspect_mask);
bool HasBrokenTextureViewFormats() const noexcept {
// No known Vulkan driver has broken image views
@@ -371,6 +372,7 @@ private:
struct StorageViews {
std::array<vk::ImageView, Shader::NUM_TEXTURE_TYPES> signeds;
std::array<vk::ImageView, Shader::NUM_TEXTURE_TYPES> unsigneds;
std::array<vk::ImageView, Shader::NUM_TEXTURE_TYPES> typeless;
};
[[nodiscard]] vk::ImageView MakeView(VkFormat vk_format, VkImageAspectFlags aspect_mask);
@@ -317,6 +317,11 @@ public:
return properties.properties.limits.minStorageBufferOffsetAlignment;
}
/// Returns texel buffer offset alignment requirement.
VkDeviceSize GetTexelBufferOffsetAlignment() const {
return properties.properties.limits.minTexelBufferOffsetAlignment;
}
/// Returns the maximum range for storage buffers.
VkDeviceSize GetMaxStorageBufferRange() const {
return properties.properties.limits.maxStorageBufferRange;
@@ -127,6 +127,7 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
X(vkCmdPipelineBarrier);
X(vkCmdPushConstants);
X(vkCmdPushDescriptorSetWithTemplateKHR);
X(vkCmdResetQueryPool);
X(vkCmdSetBlendConstants);
X(vkCmdSetDepthBias);
X(vkCmdSetDepthBias2EXT);
@@ -228,6 +228,7 @@ struct DeviceDispatch : InstanceDispatch {
PFN_vkCmdPushConstants vkCmdPushConstants{};
PFN_vkCmdPushDescriptorSetWithTemplateKHR vkCmdPushDescriptorSetWithTemplateKHR{};
PFN_vkCmdResolveImage vkCmdResolveImage{};
PFN_vkCmdResetQueryPool vkCmdResetQueryPool{};
PFN_vkCmdSetBlendConstants vkCmdSetBlendConstants{};
PFN_vkCmdSetCullModeEXT vkCmdSetCullModeEXT{};
PFN_vkCmdSetDepthBias vkCmdSetDepthBias{};
@@ -1164,6 +1165,10 @@ public:
dld->vkCmdBeginQuery(handle, query_pool, query, flags);
}
void ResetQueryPool(VkQueryPool query_pool, u32 first, u32 count) const noexcept {
dld->vkCmdResetQueryPool(handle, query_pool, first, count);
}
void EndQuery(VkQueryPool query_pool, u32 query) const noexcept {
dld->vkCmdEndQuery(handle, query_pool, query);
}