Compare commits

..

8 Commits

Author SHA1 Message Date
lizzie 5e230b4257 license 2026-08-29 08:57:57 +02:00
lizzie e72f528bd8 fx 2026-08-29 08:57:57 +02:00
lizzie 99c77fe4cc x5 faster anchordIndex for BC7 2026-08-29 08:57:57 +02:00
lizzie 6b7aff1a70 better anchorIndex 2026-08-29 08:57:57 +02:00
lizzie 5f6017467b better table set 2026-08-29 08:57:57 +02:00
lizzie 3ab734bc61 oops bcn format 2026-08-29 08:57:57 +02:00
lizzie bcb739c9c0 fix ffs 2026-08-29 08:57:57 +02:00
lizzie 29e2d09ebf [bcn] ternary compose and simplify subsetIndex & anchordIndex 2026-08-29 08:57:57 +02:00
7 changed files with 386 additions and 670 deletions
+329 -544
View File
File diff suppressed because it is too large Load Diff
+12 -32
View File
@@ -1,43 +1,23 @@
// SPDX-License-Identifier: MPL-2.0 // SPDX-License-Identifier: MPL-2.0
// Copyright © 2022 Skyline Team and Contributors (https://github.com/skyline-emu/) // Copyright 2022 Skyline Team and Contributors (https://github.com/skyline-emu/)
#pragma once #pragma once
#include <cstdint> #include <cstdint>
namespace bcn { namespace bcn {
/** /// @brief Decodes a BC1 encoded image to R8G8B8A8
* @brief Decodes a BC1 encoded image to R8G8B8A8 void DecodeBc1(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height, bool isSigned);
*/ /// @brief Decodes a BC2 encoded image to R8G8B8A8
void DecodeBc1(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height); void DecodeBc2(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height, bool isSigned);
//// @brief Decodes a BC3 encoded image to R8G8B8A8
/** void DecodeBc3(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height, bool isSigned);
* @brief Decodes a BC2 encoded image to R8G8B8A8 /// @brief Decodes a BC4 encoded image to R8
*/
void DecodeBc2(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height);
/**
* @brief Decodes a BC3 encoded image to R8G8B8A8
*/
void DecodeBc3(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height);
/**
* @brief Decodes a BC4 encoded image to R8
*/
void DecodeBc4(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height, bool isSigned); void DecodeBc4(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height, bool isSigned);
//// @brief Decodes a BC5 encoded image to R8G8
/**
* @brief Decodes a BC5 encoded image to R8G8
*/
void DecodeBc5(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height, bool isSigned); void DecodeBc5(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height, bool isSigned);
//// @brief Decodes a BC6 encoded image to R16G16B16A16
/**
* @brief Decodes a BC6 encoded image to R16G16B16A16
*/
void DecodeBc6(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height, bool isSigned); void DecodeBc6(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height, bool isSigned);
/// @brief Decodes a BC7 encoded image to R8G8B8A8
/** void DecodeBc7(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height, bool isSigned);
* @brief Decodes a BC7 encoded image to R8G8B8A8
*/
void DecodeBc7(const uint8_t *src, uint8_t *dst, size_t x, size_t y, size_t width, size_t height);
} }
+8 -9
View File
@@ -38,12 +38,10 @@
#include "common/bounded_threadsafe_queue.h" #include "common/bounded_threadsafe_queue.h"
namespace Common::Log { namespace Common::Log {
namespace {
/// @brief A log entry. Log entries are store in a structured format to permit more varied output /// @brief A log entry. Log entries are store in a structured format to permit more varied output
/// formatting on different frontends, as well as facilitating filtering and aggregation. /// formatting on different frontends, as well as facilitating filtering and aggregation.
struct Entry { struct Entry {
std::string_view thread_name;
char const* message = nullptr; char const* message = nullptr;
size_t message_len = 0; size_t message_len = 0;
std::chrono::microseconds timestamp; std::chrono::microseconds timestamp;
@@ -51,9 +49,11 @@ struct Entry {
Level log_level{}; Level log_level{};
const char* filename = nullptr; const char* filename = nullptr;
const char* function = nullptr; const char* function = nullptr;
unsigned int line_num = 0; uint32_t line_num = 0;
}; };
namespace {
/// @brief Returns the name of the passed log class as a C-string. Subclasses are separated by periods /// @brief Returns the name of the passed log class as a C-string. Subclasses are separated by periods
/// instead of underscores as in the enumeration. /// instead of underscores as in the enumeration.
/// @note GetClassName is a macro defined by Windows.h, grrr... /// @note GetClassName is a macro defined by Windows.h, grrr...
@@ -90,7 +90,7 @@ std::string FormatLogMessage(const Entry& entry) noexcept {
auto const time_fractional = uint32_t(entry.timestamp.count() % 1000000); auto const time_fractional = uint32_t(entry.timestamp.count() % 1000000);
auto const class_name = GetLogClassName(entry.log_class); auto const class_name = GetLogClassName(entry.log_class);
auto const level_name = GetLevelName(entry.log_level); auto const level_name = GetLevelName(entry.log_level);
return fmt::format("[{:4d}.{:06d}] {} <{}> ({}) {}:{}:{}: {}\n", time_seconds, time_fractional, class_name, level_name, entry.thread_name.data(), entry.filename, entry.line_num, entry.function, entry.message); return fmt::format("[{:4d}.{:06d}] {} <{}> {}:{}:{}: {}\n", time_seconds, time_fractional, class_name, level_name, entry.filename, entry.line_num, entry.function, entry.message);
} }
template <typename It> template <typename It>
@@ -156,7 +156,7 @@ struct Backend {
}; };
/// @brief Formatting specifier (to use with printf) of the equivalent fmt::format() expression /// @brief Formatting specifier (to use with printf) of the equivalent fmt::format() expression
#define CCB_PRINTF_FMT "[%4d.%06d] %s <%s> (eden:%s) %s:%u:%s: %s" #define CCB_PRINTF_FMT "[%4d.%06d] %s <%s> %s:%u:%s: %s"
/// @brief Instead of using fmt::format() just use the system's formatting capabilities directly /// @brief Instead of using fmt::format() just use the system's formatting capabilities directly
struct DirectFormatArgs { struct DirectFormatArgs {
@@ -199,7 +199,7 @@ struct ColorConsoleBackend final : public Backend {
}()); }());
SetConsoleTextAttribute(console_handle, color); SetConsoleTextAttribute(console_handle, color);
auto const df = GetDirectFormatArgs(entry); auto const df = GetDirectFormatArgs(entry);
std::fprintf(stdout, CCB_PRINTF_FMT "\n", df.time_seconds, df.time_fractional, df.class_name, df.level_name, entry.thread_name.data(), entry.filename, entry.line_num, entry.function, entry.message); std::fprintf(stdout, CCB_PRINTF_FMT "\n", df.time_seconds, df.time_fractional, df.class_name, df.level_name, entry.filename, entry.line_num, entry.function, entry.message);
} }
} }
void Flush() noexcept override {} void Flush() noexcept override {}
@@ -225,7 +225,7 @@ struct ColorConsoleBackend final : public Backend {
// more restrictive, because take for example this simple prelude: // more restrictive, because take for example this simple prelude:
// [ 50.872256] Config <Info> common/settings.cpp:142:LogSettings: // [ 50.872256] Config <Info> common/settings.cpp:142:LogSettings:
char buffer[256]; char buffer[256];
auto result = fmt::format_to_n(buffer, sizeof(buffer) - 1, "\x1b{}[{:4d}.{:06d}] {} <{}> {}:{}:{}: ", color_str, df.time_seconds, df.time_fractional, df.class_name, df.level_name, entry.thread_name.data(), entry.filename, entry.line_num, entry.function, entry.message); auto result = fmt::format_to_n(buffer, sizeof(buffer) - 1, "\x1b{}[{:4d}.{:06d}] {} <{}> {}:{}:{}: ", color_str, df.time_seconds, df.time_fractional, df.class_name, df.level_name, entry.filename, entry.line_num, entry.function, entry.message);
std::fwrite(buffer, 1, (std::min)(sizeof(buffer) - 1, result.size), stdout); std::fwrite(buffer, 1, (std::min)(sizeof(buffer) - 1, result.size), stdout);
std::fwrite(entry.message, 1, entry.message_len, stdout); std::fwrite(entry.message, 1, entry.message_len, stdout);
std::fwrite("\x1b[0m\n", 1, sizeof("\x1b[0m\n"), stdout); std::fwrite("\x1b[0m\n", 1, sizeof("\x1b[0m\n"), stdout);
@@ -329,7 +329,7 @@ struct LogcatBackend : public Backend {
} }
}(); }();
auto const df = GetDirectFormatArgs(entry); auto const df = GetDirectFormatArgs(entry);
__android_log_print(android_log_priority, "YuzuNative", CCB_PRINTF_FMT, df.time_seconds, df.time_fractional, df.class_name, df.level_name, entry.thread_name.data(), entry.filename, entry.line_num, entry.function, entry.message); __android_log_print(android_log_priority, "YuzuNative", CCB_PRINTF_FMT, df.time_seconds, df.time_fractional, df.class_name, df.level_name, entry.filename, entry.line_num, entry.function, entry.message);
} }
void Flush() noexcept override {} void Flush() noexcept override {}
}; };
@@ -431,7 +431,6 @@ void FmtLogMessageImpl(Class log_class, Level log_level, const char* filename, u
buffer[(std::min)(result.size, sizeof(buffer) - 1)] = '\0'; buffer[(std::min)(result.size, sizeof(buffer) - 1)] = '\0';
logging_instance->ForEachBackend([=](Backend& backend) { logging_instance->ForEachBackend([=](Backend& backend) {
backend.Write(Entry{ backend.Write(Entry{
.thread_name = Common::GetCurrentThreadName(),
.message = buffer, .message = buffer,
.message_len = (std::min)(result.size, sizeof(buffer) - 1), .message_len = (std::min)(result.size, sizeof(buffer) - 1),
.timestamp = std::chrono::duration_cast<std::chrono::microseconds>(std::chrono::steady_clock::now() - logging_instance->time_origin), .timestamp = std::chrono::duration_cast<std::chrono::microseconds>(std::chrono::steady_clock::now() - logging_instance->time_origin),
+1 -13
View File
@@ -449,13 +449,6 @@ void RememberCurrentThreadNice(pid_t tid, s32 nice_value) {
namespace Common { namespace Common {
// The use of TLS is justified as it is faster than using pthread_* functions
// and generally will be better long term... yeah %fs/%gs reloads aren't great
// but it's better than doing a potential call-stack-fuckery...
thread_local struct {
std::string name{};
} per_thread_data = {};
void SetCurrentThreadPriority(ThreadPriority new_priority) { void SetCurrentThreadPriority(ThreadPriority new_priority) {
#ifdef _WIN32 #ifdef _WIN32
int windows_priority = [&]() { int windows_priority = [&]() {
@@ -510,7 +503,7 @@ void SetCurrentThreadPriority(ThreadPriority new_priority) {
#endif #endif
} }
void SetCurrentThreadName(const char* name) noexcept { void SetCurrentThreadName(const char* name) {
#ifdef _MSC_VER #ifdef _MSC_VER
// Sets the debugger-visible name of the current thread. // Sets the debugger-visible name of the current thread.
if (auto pf = (decltype(&SetThreadDescription))(void*)GetProcAddress(GetModuleHandle(TEXT("KernelBase.dll")), "SetThreadDescription"); pf) if (auto pf = (decltype(&SetThreadDescription))(void*)GetProcAddress(GetModuleHandle(TEXT("KernelBase.dll")), "SetThreadDescription"); pf)
@@ -544,11 +537,6 @@ void SetCurrentThreadName(const char* name) noexcept {
#else #else
pthread_setname_np(pthread_self(), name); pthread_setname_np(pthread_self(), name);
#endif #endif
per_thread_data.name = std::string{name};
}
std::string_view GetCurrentThreadName() noexcept {
return per_thread_data.name;
} }
void SetCurrentThreadToPerformanceCores() { void SetCurrentThreadToPerformanceCores() {
+1 -4
View File
@@ -106,10 +106,7 @@ enum class ThreadPlacement : u32 {
}; };
void SetCurrentThreadPriority(ThreadPriority new_priority); void SetCurrentThreadPriority(ThreadPriority new_priority);
void SetCurrentThreadName(const char* name);
void SetCurrentThreadName(const char* name) noexcept;
std::string_view GetCurrentThreadName() noexcept;
void SetCurrentThreadToPerformanceCores(); void SetCurrentThreadToPerformanceCores();
void SetCurrentThreadToEfficiencyCores(); void SetCurrentThreadToEfficiencyCores();
void SetCurrentThreadToBackgroundWork(); void SetCurrentThreadToBackgroundWork();
+25 -51
View File
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project // SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later // SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
@@ -9,6 +9,7 @@
#include <span> #include <span>
#include <bc_decoder.h> #include <bc_decoder.h>
#include "common/assert.h"
#include "common/common_types.h" #include "common/common_types.h"
#include "video_core/texture_cache/decode_bc.h" #include "video_core/texture_cache/decode_bc.h"
@@ -22,11 +23,8 @@ using VideoCore::Surface::PixelFormat;
constexpr bool IsSigned(PixelFormat pixel_format) { constexpr bool IsSigned(PixelFormat pixel_format) {
switch (pixel_format) { switch (pixel_format) {
case PixelFormat::BC4_SNORM: case PixelFormat::BC4_SNORM:
case PixelFormat::BC4_UNORM:
case PixelFormat::BC5_SNORM: case PixelFormat::BC5_SNORM:
case PixelFormat::BC5_UNORM:
case PixelFormat::BC6H_SFLOAT: case PixelFormat::BC6H_SFLOAT:
case PixelFormat::BC6H_UFLOAT:
return true; return true;
default: default:
return false; return false;
@@ -62,9 +60,28 @@ u32 ConvertedBytesPerBlock(VideoCore::Surface::PixelFormat pixel_format) {
} }
} }
template <auto decompress, PixelFormat pixel_format> void DecompressBCn(std::span<const u8> input, std::span<u8> output, BufferImageCopy& copy, VideoCore::Surface::PixelFormat pixel_format) {
void DecompressBlocks(std::span<const u8> input, std::span<u8> output, BufferImageCopy& copy, auto const f = [pixel_format]{
bool is_signed = false) { switch (pixel_format) {
case PixelFormat::BC1_RGBA_UNORM:
case PixelFormat::BC1_RGBA_SRGB: return &bcn::DecodeBc1;
case PixelFormat::BC2_UNORM:
case PixelFormat::BC2_SRGB: return &bcn::DecodeBc2;
case PixelFormat::BC3_UNORM:
case PixelFormat::BC3_SRGB: return &bcn::DecodeBc3;
case PixelFormat::BC4_SNORM:
case PixelFormat::BC4_UNORM: return &bcn::DecodeBc4;
case PixelFormat::BC5_SNORM:
case PixelFormat::BC5_UNORM: return &bcn::DecodeBc5;
case PixelFormat::BC6H_SFLOAT:
case PixelFormat::BC6H_UFLOAT: return &bcn::DecodeBc6;
case PixelFormat::BC7_SRGB:
case PixelFormat::BC7_UNORM: return &bcn::DecodeBc7;
default:
UNREACHABLE_MSG("Unimplemented BCn decompression {}", pixel_format);
return &bcn::DecodeBc1;
}
}();
const u32 out_bpp = ConvertedBytesPerBlock(pixel_format); const u32 out_bpp = ConvertedBytesPerBlock(pixel_format);
const u32 block_size = BlockSize(pixel_format); const u32 block_size = BlockSize(pixel_format);
const u32 width = copy.image_extent.width; const u32 width = copy.image_extent.width;
@@ -82,11 +99,7 @@ void DecompressBlocks(std::span<const u8> input, std::span<u8> output, BufferIma
for (u32 x = 0; x < width; x += block_width) { for (u32 x = 0; x < width; x += block_width) {
const u8* src = input.data() + src_offset; const u8* src = input.data() + src_offset;
u8* const dst = output.data() + dst_offset; u8* const dst = output.data() + dst_offset;
if constexpr (IsSigned(pixel_format)) { f(src, dst, x, y, width, height, IsSigned(pixel_format));
decompress(src, dst, x, y, width, height, is_signed);
} else {
decompress(src, dst, x, y, width, height);
}
src_offset += block_size; src_offset += block_size;
dst_offset += block_width * out_bpp; dst_offset += block_width * out_bpp;
} }
@@ -96,43 +109,4 @@ void DecompressBlocks(std::span<const u8> input, std::span<u8> output, BufferIma
} }
} }
void DecompressBCn(std::span<const u8> input, std::span<u8> output, BufferImageCopy& copy,
VideoCore::Surface::PixelFormat pixel_format) {
switch (pixel_format) {
case PixelFormat::BC1_RGBA_UNORM:
case PixelFormat::BC1_RGBA_SRGB:
DecompressBlocks<bcn::DecodeBc1, PixelFormat::BC1_RGBA_UNORM>(input, output, copy);
break;
case PixelFormat::BC2_UNORM:
case PixelFormat::BC2_SRGB:
DecompressBlocks<bcn::DecodeBc2, PixelFormat::BC2_UNORM>(input, output, copy);
break;
case PixelFormat::BC3_UNORM:
case PixelFormat::BC3_SRGB:
DecompressBlocks<bcn::DecodeBc3, PixelFormat::BC3_UNORM>(input, output, copy);
break;
case PixelFormat::BC4_SNORM:
case PixelFormat::BC4_UNORM:
DecompressBlocks<bcn::DecodeBc4, PixelFormat::BC4_UNORM>(
input, output, copy, pixel_format == PixelFormat::BC4_SNORM);
break;
case PixelFormat::BC5_SNORM:
case PixelFormat::BC5_UNORM:
DecompressBlocks<bcn::DecodeBc5, PixelFormat::BC5_UNORM>(
input, output, copy, pixel_format == PixelFormat::BC5_SNORM);
break;
case PixelFormat::BC6H_SFLOAT:
case PixelFormat::BC6H_UFLOAT:
DecompressBlocks<bcn::DecodeBc6, PixelFormat::BC6H_UFLOAT>(
input, output, copy, pixel_format == PixelFormat::BC6H_SFLOAT);
break;
case PixelFormat::BC7_SRGB:
case PixelFormat::BC7_UNORM:
DecompressBlocks<bcn::DecodeBc7, PixelFormat::BC7_UNORM>(input, output, copy);
break;
default:
LOG_WARNING(HW_GPU, "Unimplemented BCn decompression {}", pixel_format);
}
}
} // namespace VideoCommon } // namespace VideoCommon
+10 -17
View File
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project // SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later // SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
@@ -922,8 +922,7 @@ boost::container::small_vector<BufferImageCopy, 16> UnswizzleImage(Tegra::Memory
return copies; return copies;
} }
void ConvertImage(std::span<const u8> input, const ImageInfo& info, std::span<u8> output, void ConvertImage(std::span<const u8> input, const ImageInfo& info, std::span<u8> output, std::span<BufferImageCopy> copies) {
std::span<BufferImageCopy> copies) {
u32 output_offset = 0; u32 output_offset = 0;
Common::ScratchBuffer<u8> decode_scratch; Common::ScratchBuffer<u8> decode_scratch;
@@ -955,10 +954,10 @@ void ConvertImage(std::span<const u8> input, const ImageInfo& info, std::span<u8
} else if (astc) { } else if (astc) {
// BC1 uses 0.5 bytes per texel // BC1 uses 0.5 bytes per texel
// BC3 uses 1 byte per texel // BC3 uses 1 byte per texel
const auto compress = recompression_setting == Settings::AstcRecompression::Bc1 auto const compress = recompression_setting == Settings::AstcRecompression::Bc1
? Tegra::Texture::BCN::CompressBC1 ? Tegra::Texture::BCN::CompressBC1
: Tegra::Texture::BCN::CompressBC3; : Tegra::Texture::BCN::CompressBC3;
const auto bpp_div = recompression_setting == Settings::AstcRecompression::Bc1 ? 2 : 1; const auto bpp_div = compress == Tegra::Texture::BCN::CompressBC1 ? 2 : 1;
const u32 plane_dim = copy.image_extent.width * copy.image_extent.height; const u32 plane_dim = copy.image_extent.width * copy.image_extent.height;
const u32 level_size = plane_dim * copy.image_extent.depth * const u32 level_size = plane_dim * copy.image_extent.depth *
@@ -975,18 +974,12 @@ void ConvertImage(std::span<const u8> input, const ImageInfo& info, std::span<u8
copy.image_subresource.num_layers * copy.image_extent.depth, copy.image_subresource.num_layers * copy.image_extent.depth,
output.subspan(output_offset)); output.subspan(output_offset));
const u32 aligned_plane_dim = Common::AlignUp(copy.image_extent.width, 4) * const u32 aligned_plane_dim = Common::AlignUp(copy.image_extent.width, 4) * Common::AlignUp(copy.image_extent.height, 4);
Common::AlignUp(copy.image_extent.height, 4); copy.buffer_size = (aligned_plane_dim * copy.image_extent.depth * copy.image_subresource.num_layers) / bpp_div;
output_offset += u32(copy.buffer_size);
copy.buffer_size =
(aligned_plane_dim * copy.image_extent.depth * copy.image_subresource.num_layers) /
bpp_div;
output_offset += static_cast<u32>(copy.buffer_size);
} else { } else {
DecompressBCn(input_offset, output.subspan(output_offset), copy, info.format); DecompressBCn(input_offset, output.subspan(output_offset), copy, info.format);
output_offset += copy.image_extent.width * copy.image_extent.height * output_offset += copy.image_extent.width * copy.image_extent.height * copy.image_subresource.num_layers * ConvertedBytesPerBlock(info.format);
copy.image_subresource.num_layers *
ConvertedBytesPerBlock(info.format);
} }
copy.buffer_row_length = mip_size.width; copy.buffer_row_length = mip_size.width;