Keep working on ASTC

This commit is contained in:
CamilleLaVey
2026-09-16 16:19:52 -04:00
parent 66a25a3d79
commit 9bedcf6202
8 changed files with 221 additions and 57 deletions
@@ -57,6 +57,7 @@ using VideoCore::Surface::SurfaceType;
namespace {
constexpr bool ENABLE_MSAA_TILER_RESOLVE = true;
constexpr bool ENABLE_MSAA_RESOLVE_CONSUME = true;
constexpr bool ENABLE_ACCELERATED_UNSWIZZLE = false;
constexpr bool ENABLE_MSAA_COLOR_DISCARD = true;
constexpr bool ENABLE_MSAA_DEPTH_STENCIL_DISCARD = true;
@@ -193,6 +194,9 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
}
[[nodiscard]] bool IsUnswizzleAcceleratedFormat(const Device& device, PixelFormat format) {
if (!ENABLE_ACCELERATED_UNSWIZZLE) {
return false;
}
if (IsPixelFormatASTC(format) || VideoCore::Surface::IsPixelFormatBCn(format)) {
return false;
}
@@ -1018,7 +1022,7 @@ TextureCacheRuntime::TextureCacheRuntime(const Device& device_, Scheduler& sched
astc_decoder_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool,
compute_pass_descriptor_queue, memory_allocator);
}
if (device.IsKhrImageFormatListSupported()) {
if (ENABLE_ACCELERATED_UNSWIZZLE && device.IsKhrImageFormatListSupported()) {
bl_unswizzle_2d_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool,
compute_pass_descriptor_queue);
bl_unswizzle_3d_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool,
+24 -1
View File
@@ -1117,6 +1117,12 @@ void TextureCache<P>::UploadImageContents(Image& image, StagingBuffer& staging)
Tegra::Memory::GpuGuestMemory<u8, Tegra::Memory::GuestMemoryFlags::UnsafeRead> swizzle_data(
*gpu_memory, gpu_addr, image.guest_size_bytes, &swizzle_data_buffer);
if (True(image.flags & ImageFlagBits::Converted)) {
if (CanConvertFromGuest(image.info)) {
const auto copies =
FixSmallVectorADL(ConvertImageFromGuest(swizzle_data, image.info, mapped_span));
image.UploadMemory(staging, copies);
return;
}
unswizzle_data_buffer.resize_destructive(image.unswizzled_size_bytes);
auto copies = FixSmallVectorADL(UnswizzleImage(*gpu_memory, gpu_addr, image.info, swizzle_data, unswizzle_data_buffer));
ConvertImage(unswizzle_data_buffer, image.info, mapped_span, copies);
@@ -1302,10 +1308,27 @@ void TextureCache<P>::QueueAsyncDecode(Image& image, ImageId image_id) {
decode->image_id = image_id;
async_decodes.push_back(std::move(decode));
const size_t out_size = MapSizeBytes(image);
if (CanConvertFromGuest(image.info)) {
decode_ptr->input_data.resize_destructive(image.guest_size_bytes);
gpu_memory->ReadBlockUnsafe(image.gpu_addr, decode_ptr->input_data.data(),
image.guest_size_bytes);
texture_decode_worker.QueueWork([out_size, info = image.info, async_decode = decode_ptr] {
async_decode->decoded_data.resize_destructive(out_size);
auto copies =
ConvertImageFromGuest(async_decode->input_data, info, async_decode->decoded_data);
std::unique_lock lock{async_decode->mutex};
async_decode->copies = std::move(copies);
async_decode->complete = true;
});
return;
}
std::vector<u8> local_unswizzle_data_buffer(image.unswizzled_size_bytes, 0);
Tegra::Memory::GpuGuestMemory<u8, Tegra::Memory::GuestMemoryFlags::UnsafeRead> swizzle_data(*gpu_memory, image.gpu_addr, image.guest_size_bytes, &swizzle_data_buffer);
auto copies = UnswizzleImage(*gpu_memory, image.gpu_addr, image.info, swizzle_data, local_unswizzle_data_buffer);
const size_t out_size = MapSizeBytes(image);
auto func = [out_size, copies, info = image.info,
input = std::move(local_unswizzle_data_buffer),
@@ -63,6 +63,7 @@ struct ImageViewInOut {
struct AsyncDecodeContext {
ImageId image_id;
Common::ScratchBuffer<u8> input_data;
Common::ScratchBuffer<u8> decoded_data;
boost::container::small_vector<BufferImageCopy, 16> copies;
std::mutex mutex;
+73 -1
View File
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
@@ -994,6 +994,78 @@ void ConvertImage(std::span<const u8> input, const ImageInfo& info, std::span<u8
}
}
bool CanConvertFromGuest(const ImageInfo& info) {
if (!IsPixelFormatASTC(info.format) || info.type == ImageType::Linear) {
return false;
}
return Settings::values.astc_recompression.GetValue() ==
Settings::AstcRecompression::Uncompressed;
}
boost::container::small_vector<BufferImageCopy, 16> ConvertImageFromGuest(
std::span<const u8> input, const ImageInfo& info, std::span<u8> output) {
const u32 bpp_log2 = BytesPerBlockLog2(info.format);
const Extent2D tile_size = DefaultBlockSize(info.format);
const Extent3D size = info.size;
const LevelInfo level_info = MakeLevelInfo(info);
const s32 num_layers = info.resources.layers;
const s32 num_levels = info.resources.levels;
const std::array level_sizes = CalculateLevelSizes(level_info, num_levels);
const Extent2D gob = GobSize(bpp_log2, info.block.height, info.tile_width_spacing);
const u32 layer_size = CalculateLevelBytes(level_sizes, num_levels);
const u32 layer_stride = AlignLayerSize(layer_size, size, level_info.block, tile_size.height,
info.tile_width_spacing);
const u32 out_bytes_per_texel = BytesPerBlock(PixelFormat::A8B8G8R8_UNORM);
size_t guest_offset = 0;
u32 output_offset = 0;
boost::container::small_vector<BufferImageCopy, 16> copies(num_levels);
for (s32 level = 0; level < num_levels; ++level) {
const Extent3D level_size = AdjustMipSize(size, level);
const Extent3D num_tiles = AdjustTileSize(level_size, tile_size);
const Extent3D block =
AdjustMipBlockSize(num_tiles, level_info.block, level, level_info.num_levels);
const u32 stride_alignment = StrideAlignment(num_tiles, info.block, gob, bpp_log2);
const u32 stride = Common::AlignUpLog2(num_tiles.width, stride_alignment) << bpp_log2;
const u32 gobs_in_x = Common::DivCeilLog2(stride, GOB_SIZE_X_SHIFT);
const u32 gob_block_size = gobs_in_x << (GOB_SIZE_SHIFT + block.height + block.depth);
const u32 level_bytes = level_size.width * level_size.height * level_size.depth *
num_layers * out_bytes_per_texel;
copies[level] = BufferImageCopy{
.buffer_offset = output_offset,
.buffer_size = level_bytes,
.buffer_row_length = level_size.width,
.buffer_image_height = level_size.height,
.image_subresource =
{
.base_level = level,
.base_layer = 0,
.num_layers = num_layers,
},
.image_offset = {0, 0, 0},
.image_extent = level_size,
};
const Tegra::Texture::ASTC::BlockLinearLayout layout{
.layer_stride = layer_stride,
.slice_size =
Common::DivCeilLog2(num_tiles.height, block.height + GOB_SIZE_Y_SHIFT) *
gob_block_size,
.block_size = gob_block_size,
.x_shift = GOB_SIZE_SHIFT + block.height + block.depth,
.gob_height = block.height,
.gob_height_mask = (1U << block.height) - 1,
.gob_depth = block.depth,
.gob_depth_mask = (1U << block.depth) - 1,
};
Tegra::Texture::ASTC::DecompressBlockLinear(
input.subspan(guest_offset), level_size.width, level_size.height, level_size.depth,
num_layers, tile_size.width, tile_size.height, layout, output.subspan(output_offset));
output_offset += level_bytes;
guest_offset += level_sizes[level];
}
return copies;
}
boost::container::small_vector<BufferImageCopy, 16> FullDownloadCopies(const ImageInfo& info) {
const Extent3D size = info.size;
const u32 bytes_per_block = BytesPerBlock(info.format);
+6 -1
View File
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
@@ -72,6 +72,11 @@ struct OverlapResult {
void ConvertImage(std::span<const u8> input, const ImageInfo& info, std::span<u8> output,
std::span<BufferImageCopy> copies);
[[nodiscard]] bool CanConvertFromGuest(const ImageInfo& info);
[[nodiscard]] boost::container::small_vector<BufferImageCopy, 16> ConvertImageFromGuest(
std::span<const u8> input, const ImageInfo& info, std::span<u8> output);
[[nodiscard]] boost::container::small_vector<BufferImageCopy, 16> FullDownloadCopies(
const ImageInfo& info);
+93 -33
View File
@@ -20,6 +20,7 @@
#include "common/common_types.h"
#include <ranges>
#include "video_core/textures/astc.h"
#include "video_core/textures/decoders.h"
#include "video_core/textures/workers.h"
class InputBitStream {
@@ -1671,7 +1672,7 @@ static void ComputeEndpoints(Pixel& ep1, Pixel& ep2, const u32*& colorValues,
}
static void FillVoidExtentLDR(InputBitStream& strm, std::span<u32> outBuf, u32 blockWidth,
u32 blockHeight) {
u32 blockHeight, u32 stride) {
// Don't actually care about the void extent, just read the bits...
for (s32 i = 0; i < 4; ++i) {
strm.ReadBits<13>();
@@ -1688,7 +1689,7 @@ static void FillVoidExtentLDR(InputBitStream& strm, std::span<u32> outBuf, u32 b
for (u32 j = 0; j < blockHeight; j++) {
for (u32 i = 0; i < blockWidth; i++) {
outBuf[j * blockWidth + i] = rgba;
outBuf[j * stride + i] = rgba;
}
}
}
@@ -1726,52 +1727,52 @@ static u16 HalfToClampedByte(u16 half_bits) {
return static_cast<u16>(clamped * 255.0f + 0.5f);
}
static void FillError(std::span<u32> outBuf, u32 blockWidth, u32 blockHeight) {
static void FillError(std::span<u32> outBuf, u32 blockWidth, u32 blockHeight, u32 stride) {
for (u32 j = 0; j < blockHeight; j++) {
for (u32 i = 0; i < blockWidth; i++) {
outBuf[j * blockWidth + i] = 0x00000000;
outBuf[j * stride + i] = 0x00000000;
}
}
}
static void DecompressBlock(std::span<const u8, 16> inBuf, const u32 blockWidth,
const u32 blockHeight, std::span<u32, 12 * 12> outBuf) {
const u32 blockHeight, std::span<u32> outBuf, u32 stride) {
InputBitStream strm(inBuf);
TexelWeightParams weightParams = DecodeBlockInfo(strm);
// Was there an error?
if (weightParams.m_bError) {
assert(false && "Invalid block mode");
FillError(outBuf, blockWidth, blockHeight);
FillError(outBuf, blockWidth, blockHeight, stride);
return;
}
if (weightParams.m_bVoidExtentLDR) {
FillVoidExtentLDR(strm, outBuf, blockWidth, blockHeight);
FillVoidExtentLDR(strm, outBuf, blockWidth, blockHeight, stride);
return;
}
if (weightParams.m_bVoidExtentHDR) {
assert(false && "HDR void extent blocks are unsupported!");
FillError(outBuf, blockWidth, blockHeight);
FillError(outBuf, blockWidth, blockHeight, stride);
return;
}
if (weightParams.m_Width > blockWidth) {
assert(false && "Texel weight grid width should be smaller than block width");
FillError(outBuf, blockWidth, blockHeight);
FillError(outBuf, blockWidth, blockHeight, stride);
return;
}
if (weightParams.m_Height > blockHeight) {
assert(false && "Texel weight grid height should be smaller than block height");
FillError(outBuf, blockWidth, blockHeight);
FillError(outBuf, blockWidth, blockHeight, stride);
return;
}
if (weightParams.GetNumWeightValues() > 64) {
assert(false && "Too many weights in the weight grid");
FillError(outBuf, blockWidth, blockHeight);
FillError(outBuf, blockWidth, blockHeight, stride);
return;
}
@@ -1781,7 +1782,7 @@ static void DecompressBlock(std::span<const u8, 16> inBuf, const u32 blockWidth,
if (nPartitions == 4 && weightParams.m_bDualPlane) {
assert(false && "Dual plane mode is incompatible with four partition blocks");
FillError(outBuf, blockWidth, blockHeight);
FillError(outBuf, blockWidth, blockHeight, stride);
return;
}
@@ -1813,7 +1814,7 @@ static void DecompressBlock(std::span<const u8, 16> inBuf, const u32 blockWidth,
u32 nWeightBits = weightParams.GetPackedBitSize();
if (nWeightBits < 24 || nWeightBits > 96) {
assert(false && "Invalid weight bit count");
FillError(outBuf, blockWidth, blockHeight);
FillError(outBuf, blockWidth, blockHeight, stride);
return;
}
s32 remainingBits = 128 - nWeightBits - static_cast<int>(strm.GetBitsRead());
@@ -1995,41 +1996,54 @@ static void DecompressBlock(std::span<const u8, 16> inBuf, const u32 blockWidth,
}
}
outBuf[j * blockWidth + i] = p.Pack();
outBuf[j * stride + i] = p.Pack();
}
}
static void DecodeAndStoreBlock(std::span<const uint8_t> data, size_t src_offset, u32 x, u32 y,
u32 width, u32 height, u32 block_width, u32 block_height,
std::span<u32> out_slice) {
const u32 decomp_width = (std::min)(block_width, width - x);
const u32 decomp_height = (std::min)(block_height, height - y);
u32* const dst = out_slice.data() + size_t{y} * width + x;
if (src_offset + 16 > data.size()) {
for (u32 h = 0; h < decomp_height; ++h) {
std::memset(dst + size_t{h} * width, 0, decomp_width * 4);
}
return;
}
const std::span<const u8, 16> block_ptr{data.subspan(src_offset, 16)};
if (decomp_width == block_width && decomp_height == block_height) {
const size_t touched = size_t{block_height - 1} * width + block_width;
DecompressBlock(block_ptr, block_width, block_height, std::span<u32>{dst, touched}, width);
return;
}
std::array<u32, 12 * 12> staging;
DecompressBlock(block_ptr, block_width, block_height, staging, block_width);
for (u32 h = 0; h < decomp_height; ++h) {
std::memcpy(dst + size_t{h} * width, staging.data() + h * block_width, decomp_width * 4);
}
}
void Decompress(std::span<const uint8_t> data, uint32_t width, uint32_t height, uint32_t depth,
uint32_t block_width, uint32_t block_height, std::span<uint8_t> output) {
const u32 rows = Common::DivideUp(height, block_height);
const u32 cols = Common::DivideUp(width, block_width);
Common::ThreadWorker& workers{GetThreadWorkers()};
const std::span<u32> out{reinterpret_cast<u32*>(output.data()), output.size() / 4};
const size_t slice_texels = size_t{height} * width;
for (u32 z = 0; z < depth; ++z) {
const u32 depth_offset = z * height * width * 4;
const std::span<u32> out_slice = out.subspan(z * slice_texels);
for (u32 y_index = 0; y_index < rows; ++y_index) {
auto decompress_stride = [data, width, height, block_width, block_height, output, rows,
cols, z, depth_offset, y_index] {
auto decompress_stride = [data, width, height, block_width, block_height, out_slice,
rows, cols, z, y_index] {
const u32 y = y_index * block_height;
for (u32 x_index = 0; x_index < cols; ++x_index) {
const u32 block_index = (z * rows * cols) + (y_index * cols) + x_index;
const u32 x = x_index * block_width;
const std::span<const u8, 16> blockPtr{data.subspan(block_index * 16, 16)};
// Blocks can be at most 12x12
std::array<u32, 12 * 12> uncompData;
DecompressBlock(blockPtr, block_width, block_height, uncompData);
u32 decompWidth = (std::min)(block_width, width - x);
u32 decompHeight = (std::min)(block_height, height - y);
const std::span<u8> outRow = output.subspan(depth_offset + (y * width + x) * 4);
for (u32 h = 0; h < decompHeight; ++h) {
std::memcpy(outRow.data() + h * width * 4,
uncompData.data() + h * block_width, decompWidth * 4);
}
DecodeAndStoreBlock(data, size_t{block_index} * 16, x_index * block_width, y,
width, height, block_width, block_height, out_slice);
}
};
workers.QueueWork(std::move(decompress_stride));
@@ -2038,4 +2052,50 @@ void Decompress(std::span<const uint8_t> data, uint32_t width, uint32_t height,
}
}
void DecompressBlockLinear(std::span<const uint8_t> data, uint32_t width, uint32_t height,
uint32_t depth, uint32_t layers, uint32_t block_width,
uint32_t block_height, const BlockLinearLayout& layout,
std::span<uint8_t> output) {
const u32 rows = Common::DivideUp(height, block_height);
const u32 cols = Common::DivideUp(width, block_width);
const size_t slice_texels = size_t{height} * width;
Common::ThreadWorker& workers{GetThreadWorkers()};
const std::span<u32> out{reinterpret_cast<u32*>(output.data()), output.size() / 4};
for (u32 layer = 0; layer < layers; ++layer) {
const size_t layer_offset = size_t{layer} * layout.layer_stride;
for (u32 z = 0; z < depth; ++z) {
const size_t slice_offset =
layer_offset + (z >> layout.gob_depth) * size_t{layout.slice_size} +
((z & layout.gob_depth_mask) << (GOB_SIZE_SHIFT + layout.gob_height));
const std::span<u32> out_slice =
out.subspan((size_t{layer} * depth + z) * slice_texels);
for (u32 y_index = 0; y_index < rows; ++y_index) {
auto decompress_stride = [data, width, height, block_width, block_height, out_slice,
cols, layout, slice_offset, y_index] {
const u32 y = y_index * block_height;
const u32 gob_y = y_index >> GOB_SIZE_Y_SHIFT;
const size_t offset_y =
(gob_y >> layout.gob_height) * size_t{layout.block_size} +
((gob_y & layout.gob_height_mask) << GOB_SIZE_SHIFT);
const u32 swizzled_y = ((y_index & 1) << 4) | ((y_index & 6) << 5);
for (u32 x_index = 0; x_index < cols; ++x_index) {
const u32 byte_x = x_index * 16;
const size_t offset_x = size_t{byte_x >> GOB_SIZE_X_SHIFT}
<< layout.x_shift;
const u32 swizzled_x = ((x_index & 1) << 5) | ((x_index & 2) << 7);
const size_t src_offset = slice_offset + offset_y + offset_x +
(swizzled_x | swizzled_y);
DecodeAndStoreBlock(data, src_offset, x_index * block_width, y, width,
height, block_width, block_height, out_slice);
}
};
workers.QueueWork(std::move(decompress_stride));
}
}
}
workers.WaitForRequests();
}
} // namespace Tegra::Texture::ASTC
+19
View File
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later
@@ -5,7 +8,23 @@
namespace Tegra::Texture::ASTC {
struct BlockLinearLayout {
uint32_t layer_stride;
uint32_t slice_size;
uint32_t block_size;
uint32_t x_shift;
uint32_t gob_height;
uint32_t gob_height_mask;
uint32_t gob_depth;
uint32_t gob_depth_mask;
};
void Decompress(std::span<const uint8_t> data, uint32_t width, uint32_t height, uint32_t depth,
uint32_t block_width, uint32_t block_height, std::span<uint8_t> output);
void DecompressBlockLinear(std::span<const uint8_t> data, uint32_t width, uint32_t height,
uint32_t depth, uint32_t layers, uint32_t block_width,
uint32_t block_height, const BlockLinearLayout& layout,
std::span<uint8_t> output);
} // namespace Tegra::Texture::ASTC
@@ -30,26 +30,6 @@ namespace {
// Helpers translating MemoryUsage to flags/usage
[[maybe_unused]] VkMemoryPropertyFlags MemoryUsagePropertyFlags(MemoryUsage usage) {
switch (usage) {
case MemoryUsage::DeviceLocal:
return VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT;
case MemoryUsage::Upload:
return VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT |
VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
case MemoryUsage::Download:
return VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT |
VK_MEMORY_PROPERTY_HOST_COHERENT_BIT |
VK_MEMORY_PROPERTY_HOST_CACHED_BIT;
case MemoryUsage::Stream:
return VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT |
VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT |
VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
}
ASSERT_MSG(false, "Invalid memory usage={}", usage);
return VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;
}
[[nodiscard]] VkMemoryPropertyFlags MemoryUsagePreferredVmaFlags(MemoryUsage usage) {
if (usage == MemoryUsage::Download) {
return VK_MEMORY_PROPERTY_HOST_CACHED_BIT | VK_MEMORY_PROPERTY_HOST_COHERENT_BIT;