diff --git a/src/video_core/renderer_vulkan/vk_texture_cache.cpp b/src/video_core/renderer_vulkan/vk_texture_cache.cpp index bf8efaaf09..caff739187 100644 --- a/src/video_core/renderer_vulkan/vk_texture_cache.cpp +++ b/src/video_core/renderer_vulkan/vk_texture_cache.cpp @@ -57,6 +57,7 @@ using VideoCore::Surface::SurfaceType; namespace { constexpr bool ENABLE_MSAA_TILER_RESOLVE = true; constexpr bool ENABLE_MSAA_RESOLVE_CONSUME = true; +constexpr bool ENABLE_ACCELERATED_UNSWIZZLE = false; constexpr bool ENABLE_MSAA_COLOR_DISCARD = true; constexpr bool ENABLE_MSAA_DEPTH_STENCIL_DISCARD = true; @@ -193,6 +194,9 @@ constexpr VkBorderColor ConvertBorderColor(const std::array& color) { } [[nodiscard]] bool IsUnswizzleAcceleratedFormat(const Device& device, PixelFormat format) { + if (!ENABLE_ACCELERATED_UNSWIZZLE) { + return false; + } if (IsPixelFormatASTC(format) || VideoCore::Surface::IsPixelFormatBCn(format)) { return false; } @@ -1018,7 +1022,7 @@ TextureCacheRuntime::TextureCacheRuntime(const Device& device_, Scheduler& sched astc_decoder_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool, compute_pass_descriptor_queue, memory_allocator); } - if (device.IsKhrImageFormatListSupported()) { + if (ENABLE_ACCELERATED_UNSWIZZLE && device.IsKhrImageFormatListSupported()) { bl_unswizzle_2d_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool, compute_pass_descriptor_queue); bl_unswizzle_3d_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool, diff --git a/src/video_core/texture_cache/texture_cache.h b/src/video_core/texture_cache/texture_cache.h index 5b2cafc89d..861d9f63d0 100644 --- a/src/video_core/texture_cache/texture_cache.h +++ b/src/video_core/texture_cache/texture_cache.h @@ -1117,6 +1117,12 @@ void TextureCache

::UploadImageContents(Image& image, StagingBuffer& staging) Tegra::Memory::GpuGuestMemory swizzle_data( *gpu_memory, gpu_addr, image.guest_size_bytes, &swizzle_data_buffer); if (True(image.flags & ImageFlagBits::Converted)) { + if (CanConvertFromGuest(image.info)) { + const auto copies = + FixSmallVectorADL(ConvertImageFromGuest(swizzle_data, image.info, mapped_span)); + image.UploadMemory(staging, copies); + return; + } unswizzle_data_buffer.resize_destructive(image.unswizzled_size_bytes); auto copies = FixSmallVectorADL(UnswizzleImage(*gpu_memory, gpu_addr, image.info, swizzle_data, unswizzle_data_buffer)); ConvertImage(unswizzle_data_buffer, image.info, mapped_span, copies); @@ -1302,10 +1308,27 @@ void TextureCache

::QueueAsyncDecode(Image& image, ImageId image_id) { decode->image_id = image_id; async_decodes.push_back(std::move(decode)); + const size_t out_size = MapSizeBytes(image); + if (CanConvertFromGuest(image.info)) { + decode_ptr->input_data.resize_destructive(image.guest_size_bytes); + gpu_memory->ReadBlockUnsafe(image.gpu_addr, decode_ptr->input_data.data(), + image.guest_size_bytes); + + texture_decode_worker.QueueWork([out_size, info = image.info, async_decode = decode_ptr] { + async_decode->decoded_data.resize_destructive(out_size); + auto copies = + ConvertImageFromGuest(async_decode->input_data, info, async_decode->decoded_data); + + std::unique_lock lock{async_decode->mutex}; + async_decode->copies = std::move(copies); + async_decode->complete = true; + }); + return; + } + std::vector local_unswizzle_data_buffer(image.unswizzled_size_bytes, 0); Tegra::Memory::GpuGuestMemory swizzle_data(*gpu_memory, image.gpu_addr, image.guest_size_bytes, &swizzle_data_buffer); auto copies = UnswizzleImage(*gpu_memory, image.gpu_addr, image.info, swizzle_data, local_unswizzle_data_buffer); - const size_t out_size = MapSizeBytes(image); auto func = [out_size, copies, info = image.info, input = std::move(local_unswizzle_data_buffer), diff --git a/src/video_core/texture_cache/texture_cache_base.h b/src/video_core/texture_cache/texture_cache_base.h index 517afa4aa8..8fbd7f61bc 100644 --- a/src/video_core/texture_cache/texture_cache_base.h +++ b/src/video_core/texture_cache/texture_cache_base.h @@ -63,6 +63,7 @@ struct ImageViewInOut { struct AsyncDecodeContext { ImageId image_id; + Common::ScratchBuffer input_data; Common::ScratchBuffer decoded_data; boost::container::small_vector copies; std::mutex mutex; diff --git a/src/video_core/texture_cache/util.cpp b/src/video_core/texture_cache/util.cpp index 25cf6e33b1..8d275ff8eb 100644 --- a/src/video_core/texture_cache/util.cpp +++ b/src/video_core/texture_cache/util.cpp @@ -1,4 +1,4 @@ -// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project +// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project // SPDX-License-Identifier: GPL-3.0-or-later // SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project @@ -994,6 +994,78 @@ void ConvertImage(std::span input, const ImageInfo& info, std::span ConvertImageFromGuest( + std::span input, const ImageInfo& info, std::span output) { + const u32 bpp_log2 = BytesPerBlockLog2(info.format); + const Extent2D tile_size = DefaultBlockSize(info.format); + const Extent3D size = info.size; + const LevelInfo level_info = MakeLevelInfo(info); + const s32 num_layers = info.resources.layers; + const s32 num_levels = info.resources.levels; + const std::array level_sizes = CalculateLevelSizes(level_info, num_levels); + const Extent2D gob = GobSize(bpp_log2, info.block.height, info.tile_width_spacing); + const u32 layer_size = CalculateLevelBytes(level_sizes, num_levels); + const u32 layer_stride = AlignLayerSize(layer_size, size, level_info.block, tile_size.height, + info.tile_width_spacing); + const u32 out_bytes_per_texel = BytesPerBlock(PixelFormat::A8B8G8R8_UNORM); + size_t guest_offset = 0; + u32 output_offset = 0; + boost::container::small_vector copies(num_levels); + + for (s32 level = 0; level < num_levels; ++level) { + const Extent3D level_size = AdjustMipSize(size, level); + const Extent3D num_tiles = AdjustTileSize(level_size, tile_size); + const Extent3D block = + AdjustMipBlockSize(num_tiles, level_info.block, level, level_info.num_levels); + const u32 stride_alignment = StrideAlignment(num_tiles, info.block, gob, bpp_log2); + const u32 stride = Common::AlignUpLog2(num_tiles.width, stride_alignment) << bpp_log2; + const u32 gobs_in_x = Common::DivCeilLog2(stride, GOB_SIZE_X_SHIFT); + const u32 gob_block_size = gobs_in_x << (GOB_SIZE_SHIFT + block.height + block.depth); + const u32 level_bytes = level_size.width * level_size.height * level_size.depth * + num_layers * out_bytes_per_texel; + copies[level] = BufferImageCopy{ + .buffer_offset = output_offset, + .buffer_size = level_bytes, + .buffer_row_length = level_size.width, + .buffer_image_height = level_size.height, + .image_subresource = + { + .base_level = level, + .base_layer = 0, + .num_layers = num_layers, + }, + .image_offset = {0, 0, 0}, + .image_extent = level_size, + }; + const Tegra::Texture::ASTC::BlockLinearLayout layout{ + .layer_stride = layer_stride, + .slice_size = + Common::DivCeilLog2(num_tiles.height, block.height + GOB_SIZE_Y_SHIFT) * + gob_block_size, + .block_size = gob_block_size, + .x_shift = GOB_SIZE_SHIFT + block.height + block.depth, + .gob_height = block.height, + .gob_height_mask = (1U << block.height) - 1, + .gob_depth = block.depth, + .gob_depth_mask = (1U << block.depth) - 1, + }; + Tegra::Texture::ASTC::DecompressBlockLinear( + input.subspan(guest_offset), level_size.width, level_size.height, level_size.depth, + num_layers, tile_size.width, tile_size.height, layout, output.subspan(output_offset)); + output_offset += level_bytes; + guest_offset += level_sizes[level]; + } + return copies; +} + boost::container::small_vector FullDownloadCopies(const ImageInfo& info) { const Extent3D size = info.size; const u32 bytes_per_block = BytesPerBlock(info.format); diff --git a/src/video_core/texture_cache/util.h b/src/video_core/texture_cache/util.h index 3e8bb00032..1bc4b53582 100644 --- a/src/video_core/texture_cache/util.h +++ b/src/video_core/texture_cache/util.h @@ -1,4 +1,4 @@ -// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project +// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project // SPDX-License-Identifier: GPL-3.0-or-later // SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project @@ -72,6 +72,11 @@ struct OverlapResult { void ConvertImage(std::span input, const ImageInfo& info, std::span output, std::span copies); +[[nodiscard]] bool CanConvertFromGuest(const ImageInfo& info); + +[[nodiscard]] boost::container::small_vector ConvertImageFromGuest( + std::span input, const ImageInfo& info, std::span output); + [[nodiscard]] boost::container::small_vector FullDownloadCopies( const ImageInfo& info); diff --git a/src/video_core/textures/astc.cpp b/src/video_core/textures/astc.cpp index 0b6d06c49a..678e52a242 100644 --- a/src/video_core/textures/astc.cpp +++ b/src/video_core/textures/astc.cpp @@ -20,6 +20,7 @@ #include "common/common_types.h" #include #include "video_core/textures/astc.h" +#include "video_core/textures/decoders.h" #include "video_core/textures/workers.h" class InputBitStream { @@ -1671,7 +1672,7 @@ static void ComputeEndpoints(Pixel& ep1, Pixel& ep2, const u32*& colorValues, } static void FillVoidExtentLDR(InputBitStream& strm, std::span outBuf, u32 blockWidth, - u32 blockHeight) { + u32 blockHeight, u32 stride) { // Don't actually care about the void extent, just read the bits... for (s32 i = 0; i < 4; ++i) { strm.ReadBits<13>(); @@ -1688,7 +1689,7 @@ static void FillVoidExtentLDR(InputBitStream& strm, std::span outBuf, u32 b for (u32 j = 0; j < blockHeight; j++) { for (u32 i = 0; i < blockWidth; i++) { - outBuf[j * blockWidth + i] = rgba; + outBuf[j * stride + i] = rgba; } } } @@ -1726,52 +1727,52 @@ static u16 HalfToClampedByte(u16 half_bits) { return static_cast(clamped * 255.0f + 0.5f); } -static void FillError(std::span outBuf, u32 blockWidth, u32 blockHeight) { +static void FillError(std::span outBuf, u32 blockWidth, u32 blockHeight, u32 stride) { for (u32 j = 0; j < blockHeight; j++) { for (u32 i = 0; i < blockWidth; i++) { - outBuf[j * blockWidth + i] = 0x00000000; + outBuf[j * stride + i] = 0x00000000; } } } static void DecompressBlock(std::span inBuf, const u32 blockWidth, - const u32 blockHeight, std::span outBuf) { + const u32 blockHeight, std::span outBuf, u32 stride) { InputBitStream strm(inBuf); TexelWeightParams weightParams = DecodeBlockInfo(strm); // Was there an error? if (weightParams.m_bError) { assert(false && "Invalid block mode"); - FillError(outBuf, blockWidth, blockHeight); + FillError(outBuf, blockWidth, blockHeight, stride); return; } if (weightParams.m_bVoidExtentLDR) { - FillVoidExtentLDR(strm, outBuf, blockWidth, blockHeight); + FillVoidExtentLDR(strm, outBuf, blockWidth, blockHeight, stride); return; } if (weightParams.m_bVoidExtentHDR) { assert(false && "HDR void extent blocks are unsupported!"); - FillError(outBuf, blockWidth, blockHeight); + FillError(outBuf, blockWidth, blockHeight, stride); return; } if (weightParams.m_Width > blockWidth) { assert(false && "Texel weight grid width should be smaller than block width"); - FillError(outBuf, blockWidth, blockHeight); + FillError(outBuf, blockWidth, blockHeight, stride); return; } if (weightParams.m_Height > blockHeight) { assert(false && "Texel weight grid height should be smaller than block height"); - FillError(outBuf, blockWidth, blockHeight); + FillError(outBuf, blockWidth, blockHeight, stride); return; } if (weightParams.GetNumWeightValues() > 64) { assert(false && "Too many weights in the weight grid"); - FillError(outBuf, blockWidth, blockHeight); + FillError(outBuf, blockWidth, blockHeight, stride); return; } @@ -1781,7 +1782,7 @@ static void DecompressBlock(std::span inBuf, const u32 blockWidth, if (nPartitions == 4 && weightParams.m_bDualPlane) { assert(false && "Dual plane mode is incompatible with four partition blocks"); - FillError(outBuf, blockWidth, blockHeight); + FillError(outBuf, blockWidth, blockHeight, stride); return; } @@ -1813,7 +1814,7 @@ static void DecompressBlock(std::span inBuf, const u32 blockWidth, u32 nWeightBits = weightParams.GetPackedBitSize(); if (nWeightBits < 24 || nWeightBits > 96) { assert(false && "Invalid weight bit count"); - FillError(outBuf, blockWidth, blockHeight); + FillError(outBuf, blockWidth, blockHeight, stride); return; } s32 remainingBits = 128 - nWeightBits - static_cast(strm.GetBitsRead()); @@ -1995,41 +1996,54 @@ static void DecompressBlock(std::span inBuf, const u32 blockWidth, } } - outBuf[j * blockWidth + i] = p.Pack(); + outBuf[j * stride + i] = p.Pack(); } } +static void DecodeAndStoreBlock(std::span data, size_t src_offset, u32 x, u32 y, + u32 width, u32 height, u32 block_width, u32 block_height, + std::span out_slice) { + const u32 decomp_width = (std::min)(block_width, width - x); + const u32 decomp_height = (std::min)(block_height, height - y); + u32* const dst = out_slice.data() + size_t{y} * width + x; + if (src_offset + 16 > data.size()) { + for (u32 h = 0; h < decomp_height; ++h) { + std::memset(dst + size_t{h} * width, 0, decomp_width * 4); + } + return; + } + const std::span block_ptr{data.subspan(src_offset, 16)}; + if (decomp_width == block_width && decomp_height == block_height) { + const size_t touched = size_t{block_height - 1} * width + block_width; + DecompressBlock(block_ptr, block_width, block_height, std::span{dst, touched}, width); + return; + } + std::array staging; + DecompressBlock(block_ptr, block_width, block_height, staging, block_width); + for (u32 h = 0; h < decomp_height; ++h) { + std::memcpy(dst + size_t{h} * width, staging.data() + h * block_width, decomp_width * 4); + } +} + void Decompress(std::span data, uint32_t width, uint32_t height, uint32_t depth, uint32_t block_width, uint32_t block_height, std::span output) { const u32 rows = Common::DivideUp(height, block_height); const u32 cols = Common::DivideUp(width, block_width); Common::ThreadWorker& workers{GetThreadWorkers()}; + const std::span out{reinterpret_cast(output.data()), output.size() / 4}; + const size_t slice_texels = size_t{height} * width; for (u32 z = 0; z < depth; ++z) { - const u32 depth_offset = z * height * width * 4; + const std::span out_slice = out.subspan(z * slice_texels); for (u32 y_index = 0; y_index < rows; ++y_index) { - auto decompress_stride = [data, width, height, block_width, block_height, output, rows, - cols, z, depth_offset, y_index] { + auto decompress_stride = [data, width, height, block_width, block_height, out_slice, + rows, cols, z, y_index] { const u32 y = y_index * block_height; for (u32 x_index = 0; x_index < cols; ++x_index) { const u32 block_index = (z * rows * cols) + (y_index * cols) + x_index; - const u32 x = x_index * block_width; - - const std::span blockPtr{data.subspan(block_index * 16, 16)}; - - // Blocks can be at most 12x12 - std::array uncompData; - DecompressBlock(blockPtr, block_width, block_height, uncompData); - - u32 decompWidth = (std::min)(block_width, width - x); - u32 decompHeight = (std::min)(block_height, height - y); - - const std::span outRow = output.subspan(depth_offset + (y * width + x) * 4); - for (u32 h = 0; h < decompHeight; ++h) { - std::memcpy(outRow.data() + h * width * 4, - uncompData.data() + h * block_width, decompWidth * 4); - } + DecodeAndStoreBlock(data, size_t{block_index} * 16, x_index * block_width, y, + width, height, block_width, block_height, out_slice); } }; workers.QueueWork(std::move(decompress_stride)); @@ -2038,4 +2052,50 @@ void Decompress(std::span data, uint32_t width, uint32_t height, } } +void DecompressBlockLinear(std::span data, uint32_t width, uint32_t height, + uint32_t depth, uint32_t layers, uint32_t block_width, + uint32_t block_height, const BlockLinearLayout& layout, + std::span output) { + const u32 rows = Common::DivideUp(height, block_height); + const u32 cols = Common::DivideUp(width, block_width); + const size_t slice_texels = size_t{height} * width; + + Common::ThreadWorker& workers{GetThreadWorkers()}; + const std::span out{reinterpret_cast(output.data()), output.size() / 4}; + + for (u32 layer = 0; layer < layers; ++layer) { + const size_t layer_offset = size_t{layer} * layout.layer_stride; + for (u32 z = 0; z < depth; ++z) { + const size_t slice_offset = + layer_offset + (z >> layout.gob_depth) * size_t{layout.slice_size} + + ((z & layout.gob_depth_mask) << (GOB_SIZE_SHIFT + layout.gob_height)); + const std::span out_slice = + out.subspan((size_t{layer} * depth + z) * slice_texels); + for (u32 y_index = 0; y_index < rows; ++y_index) { + auto decompress_stride = [data, width, height, block_width, block_height, out_slice, + cols, layout, slice_offset, y_index] { + const u32 y = y_index * block_height; + const u32 gob_y = y_index >> GOB_SIZE_Y_SHIFT; + const size_t offset_y = + (gob_y >> layout.gob_height) * size_t{layout.block_size} + + ((gob_y & layout.gob_height_mask) << GOB_SIZE_SHIFT); + const u32 swizzled_y = ((y_index & 1) << 4) | ((y_index & 6) << 5); + for (u32 x_index = 0; x_index < cols; ++x_index) { + const u32 byte_x = x_index * 16; + const size_t offset_x = size_t{byte_x >> GOB_SIZE_X_SHIFT} + << layout.x_shift; + const u32 swizzled_x = ((x_index & 1) << 5) | ((x_index & 2) << 7); + const size_t src_offset = slice_offset + offset_y + offset_x + + (swizzled_x | swizzled_y); + DecodeAndStoreBlock(data, src_offset, x_index * block_width, y, width, + height, block_width, block_height, out_slice); + } + }; + workers.QueueWork(std::move(decompress_stride)); + } + } + } + workers.WaitForRequests(); +} + } // namespace Tegra::Texture::ASTC diff --git a/src/video_core/textures/astc.h b/src/video_core/textures/astc.h index afd3933c3e..5d59da7496 100644 --- a/src/video_core/textures/astc.h +++ b/src/video_core/textures/astc.h @@ -1,3 +1,6 @@ +// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project +// SPDX-License-Identifier: GPL-3.0-or-later + // SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project // SPDX-License-Identifier: GPL-2.0-or-later @@ -5,7 +8,23 @@ namespace Tegra::Texture::ASTC { +struct BlockLinearLayout { + uint32_t layer_stride; + uint32_t slice_size; + uint32_t block_size; + uint32_t x_shift; + uint32_t gob_height; + uint32_t gob_height_mask; + uint32_t gob_depth; + uint32_t gob_depth_mask; +}; + void Decompress(std::span data, uint32_t width, uint32_t height, uint32_t depth, uint32_t block_width, uint32_t block_height, std::span output); +void DecompressBlockLinear(std::span data, uint32_t width, uint32_t height, + uint32_t depth, uint32_t layers, uint32_t block_width, + uint32_t block_height, const BlockLinearLayout& layout, + std::span output); + } // namespace Tegra::Texture::ASTC diff --git a/src/video_core/vulkan_common/vulkan_memory_allocator.cpp b/src/video_core/vulkan_common/vulkan_memory_allocator.cpp index f4721543ae..4d2d5b1317 100644 --- a/src/video_core/vulkan_common/vulkan_memory_allocator.cpp +++ b/src/video_core/vulkan_common/vulkan_memory_allocator.cpp @@ -30,26 +30,6 @@ namespace { // Helpers translating MemoryUsage to flags/usage - [[maybe_unused]] VkMemoryPropertyFlags MemoryUsagePropertyFlags(MemoryUsage usage) { - switch (usage) { - case MemoryUsage::DeviceLocal: - return VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT; - case MemoryUsage::Upload: - return VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | - VK_MEMORY_PROPERTY_HOST_COHERENT_BIT; - case MemoryUsage::Download: - return VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | - VK_MEMORY_PROPERTY_HOST_COHERENT_BIT | - VK_MEMORY_PROPERTY_HOST_CACHED_BIT; - case MemoryUsage::Stream: - return VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT | - VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | - VK_MEMORY_PROPERTY_HOST_COHERENT_BIT; - } - ASSERT_MSG(false, "Invalid memory usage={}", usage); - return VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT; - } - [[nodiscard]] VkMemoryPropertyFlags MemoryUsagePreferredVmaFlags(MemoryUsage usage) { if (usage == MemoryUsage::Download) { return VK_MEMORY_PROPERTY_HOST_CACHED_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;