mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-10-07 14:26:36 +00:00
Some cleanups
This commit is contained in:
@@ -22,7 +22,6 @@ set(SHADER_FILES
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/blit_depth_msaa.frag
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/blit_depth_stencil_msaa.frag
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/block_linear_unswizzle_3d.comp
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/block_linear_unswizzle_3d_bcn.comp
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/convert_abgr8_to_d24s8.frag
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/convert_abgr8_to_d32f.frag
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/convert_d32f_to_abgr8.frag
|
||||
|
||||
@@ -165,7 +165,26 @@ uint FastReplicateTo8(uint value, uint num_bits) {
|
||||
}
|
||||
|
||||
uint FastReplicateTo6(uint value, uint num_bits) {
|
||||
return ReplicateBits(value, num_bits, 6);
|
||||
if (num_bits == 0) {
|
||||
return 0;
|
||||
}
|
||||
if (num_bits >= 6) {
|
||||
return value;
|
||||
}
|
||||
const uint v = value & ((1u << num_bits) - 1u);
|
||||
if (num_bits == 1) {
|
||||
return v * 63u;
|
||||
}
|
||||
if (num_bits == 2) {
|
||||
return v * 21u;
|
||||
}
|
||||
if (num_bits == 3) {
|
||||
return v * 9u;
|
||||
}
|
||||
if (num_bits == 4) {
|
||||
return (v << 2u) | (v >> 2u);
|
||||
}
|
||||
return (v << 1u) | (v >> 4u);
|
||||
}
|
||||
|
||||
uint Div3Floor(uint v) {
|
||||
@@ -958,35 +977,71 @@ void ComputeEndpoints(out uvec4 ep1, out uvec4 ep2, uint color_endpoint_mode, ui
|
||||
}
|
||||
|
||||
uint UnquantizeTexelWeight(EncodingData val) {
|
||||
uint encoding = Encoding(val), bitlen = NumBits(val), bitval = BitValue(val);
|
||||
if (encoding == JUST_BITS) {
|
||||
return (bitlen >= 1 && bitlen <= 5)
|
||||
? uint(floor(0.5f + float(bitval) * 64.0f / float((1 << bitlen) - 1)))
|
||||
: FastReplicateTo6(bitval, bitlen);
|
||||
} else if (encoding == TRIT || encoding == QUINT) {
|
||||
uint B = 0, C = 0, D = 0;
|
||||
uint b_mask = (0x3100 >> (bitlen * 4)) & 0xf;
|
||||
uint b = (bitval >> 1) & b_mask;
|
||||
const uint encoding = Encoding(val);
|
||||
const uint bitlen = NumBits(val);
|
||||
const uint bitval = BitValue(val);
|
||||
const uint A = ReplicateBitTo7((bitval & 1));
|
||||
uint B = 0, C = 0, D = 0;
|
||||
uint result = 0;
|
||||
const uint bitlen_0_results[5] = {0, 16, 32, 48, 64};
|
||||
switch (encoding) {
|
||||
case JUST_BITS:
|
||||
result = FastReplicateTo6(bitval, bitlen);
|
||||
break;
|
||||
case TRIT: {
|
||||
D = QuintTritValue(val);
|
||||
if (encoding == TRIT) {
|
||||
switch (bitlen) {
|
||||
case 0: return D * 32; //0,32,64
|
||||
case 1: C = 50; break;
|
||||
case 2: C = 23; B = (b << 6) | (b << 2) | b; break;
|
||||
case 3: C = 11; B = (b << 5) | b; break;
|
||||
}
|
||||
} else if (encoding == QUINT) {
|
||||
switch (bitlen) {
|
||||
case 0: return D * 16; //0, 16, 32, 48, 64
|
||||
case 1: C = 28; break;
|
||||
case 2: C = 13; B = (b << 6) | (b << 1); break;
|
||||
}
|
||||
switch (bitlen) {
|
||||
case 0:
|
||||
return bitlen_0_results[D * 2];
|
||||
case 1: {
|
||||
C = 50;
|
||||
break;
|
||||
}
|
||||
uint A = ReplicateBitTo7(bitval & 1);
|
||||
uint res = (A & 0x20) | (((D * C + B) ^ A) >> 2);
|
||||
return res + (res > 32 ? 1 : 0);
|
||||
case 2: {
|
||||
C = 23;
|
||||
const uint b = (bitval >> 1) & 1;
|
||||
B = (b << 6) | (b << 2) | b;
|
||||
break;
|
||||
}
|
||||
case 3: {
|
||||
C = 11;
|
||||
const uint cb = (bitval >> 1) & 3;
|
||||
B = (cb << 5) | cb;
|
||||
break;
|
||||
}
|
||||
default:
|
||||
break;
|
||||
}
|
||||
break;
|
||||
}
|
||||
return 0;
|
||||
case QUINT: {
|
||||
D = QuintTritValue(val);
|
||||
switch (bitlen) {
|
||||
case 0:
|
||||
return bitlen_0_results[D];
|
||||
case 1: {
|
||||
C = 28;
|
||||
break;
|
||||
}
|
||||
case 2: {
|
||||
C = 13;
|
||||
const uint b = (bitval >> 1) & 1;
|
||||
B = (b << 6) | (b << 1);
|
||||
break;
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (encoding != JUST_BITS && bitlen > 0) {
|
||||
result = D * C + B;
|
||||
result ^= A;
|
||||
result = (A & 0x20) | (result >> 2);
|
||||
}
|
||||
if (result > 32) {
|
||||
result += 1;
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
void UnquantizeTexelWeights(uvec2 size, bool is_dual_plane) {
|
||||
@@ -1394,11 +1449,10 @@ void DecompressBlock(ivec3 coord) {
|
||||
}
|
||||
|
||||
uint SwizzleOffset(uvec2 pos) {
|
||||
return ((pos.x & 32u) << 3u) |
|
||||
((pos.y & 6u) << 5u) |
|
||||
((pos.x & 16u) << 1u) |
|
||||
((pos.y & 1u) << 4u) |
|
||||
(pos.x & 15u);
|
||||
const uint x = pos.x;
|
||||
const uint y = pos.y;
|
||||
return ((x % 64) / 32) * 256 + ((y % 8) / 2) * 64 +
|
||||
((x % 32) / 16) * 32 + (y % 2) * 16 + (x % 16);
|
||||
}
|
||||
|
||||
void main() {
|
||||
|
||||
@@ -51,7 +51,7 @@ layout(binding = BINDING_INPUT_BUFFER, std430) buffer InputBufferU128 { uvec4 u1
|
||||
|
||||
layout(binding = BINDING_OUTPUT_IMAGE) uniform writeonly uimage2DArray output_image;
|
||||
|
||||
layout(local_size_x = 32, local_size_y = 32, local_size_z = 1) in;
|
||||
layout(local_size_x = 32, local_size_y = 8, local_size_z = 1) in;
|
||||
|
||||
const uint GOB_SIZE_X = 64;
|
||||
const uint GOB_SIZE_Y = 8;
|
||||
@@ -65,19 +65,11 @@ const uint GOB_SIZE_SHIFT = GOB_SIZE_X_SHIFT + GOB_SIZE_Y_SHIFT + GOB_SIZE_Z_SHI
|
||||
|
||||
const uvec2 SWIZZLE_MASK = uvec2(GOB_SIZE_X - 1, GOB_SIZE_Y - 1);
|
||||
|
||||
uint SwizzleTable(uint pos) {
|
||||
const uint t[8] = uint[](
|
||||
0x12100200, 0x13110301, 0x16140604, 0x17150705,
|
||||
0x1a180a08, 0x1b190b09, 0x1e1c0e0c, 0x1f1d0f0d
|
||||
);
|
||||
const uint i = pos >> 4;
|
||||
const uint h = (t[i / 4] >> ((i % 4) * 8)) & 0xff;
|
||||
return (h << 4) | (pos & 0xf);
|
||||
}
|
||||
|
||||
uint SwizzleOffset(uvec2 pos) {
|
||||
pos = pos & SWIZZLE_MASK;
|
||||
return SwizzleTable(pos.y * 64 + pos.x);
|
||||
return ((pos.x & 32u) << 3u) | ((pos.y & 6u) << 5u) |
|
||||
((pos.x & 16u) << 1u) | ((pos.y & 1u) << 4u) |
|
||||
(pos.x & 15u);
|
||||
}
|
||||
|
||||
uvec4 ReadTexel(uint offset) {
|
||||
|
||||
@@ -53,7 +53,7 @@ layout(binding = BINDING_INPUT_BUFFER, std430) buffer InputBufferU128 { uvec4 u1
|
||||
|
||||
layout(binding = BINDING_OUTPUT_IMAGE) uniform writeonly uimage3D output_image;
|
||||
|
||||
layout(local_size_x = 16, local_size_y = 8, local_size_z = 8) in;
|
||||
layout(local_size_x = 16, local_size_y = 8, local_size_z = 2) in;
|
||||
|
||||
const uint GOB_SIZE_X = 64;
|
||||
const uint GOB_SIZE_Y = 8;
|
||||
@@ -67,19 +67,11 @@ const uint GOB_SIZE_SHIFT = GOB_SIZE_X_SHIFT + GOB_SIZE_Y_SHIFT + GOB_SIZE_Z_SHI
|
||||
|
||||
const uvec2 SWIZZLE_MASK = uvec2(GOB_SIZE_X - 1, GOB_SIZE_Y - 1);
|
||||
|
||||
uint SwizzleTable(uint pos) {
|
||||
const uint t[8] = uint[](
|
||||
0x12100200, 0x13110301, 0x16140604, 0x17150705,
|
||||
0x1a180a08, 0x1b190b09, 0x1e1c0e0c, 0x1f1d0f0d
|
||||
);
|
||||
const uint i = pos >> 4;
|
||||
const uint h = (t[i / 4] >> ((i % 4) * 8)) & 0xff;
|
||||
return (h << 4) | (pos & 0xf);
|
||||
}
|
||||
|
||||
uint SwizzleOffset(uvec2 pos) {
|
||||
pos = pos & SWIZZLE_MASK;
|
||||
return SwizzleTable(pos.y * 64 + pos.x);
|
||||
return ((pos.x & 32u) << 3u) | ((pos.y & 6u) << 5u) |
|
||||
((pos.x & 16u) << 1u) | ((pos.y & 1u) << 4u) |
|
||||
(pos.x & 15u);
|
||||
}
|
||||
|
||||
uvec4 ReadTexel(uint offset) {
|
||||
|
||||
@@ -1,164 +0,0 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#version 430
|
||||
|
||||
#ifdef VULKAN
|
||||
#extension GL_EXT_shader_16bit_storage : require
|
||||
#extension GL_EXT_shader_8bit_storage : require
|
||||
#define HAS_EXTENDED_TYPES 1
|
||||
#define BEGIN_PUSH_CONSTANTS layout(push_constant) uniform PushConstants {
|
||||
#define END_PUSH_CONSTANTS };
|
||||
#define UNIFORM(n)
|
||||
#define BINDING_INPUT_BUFFER 0
|
||||
#define BINDING_OUTPUT_BUFFER 1
|
||||
#else
|
||||
#extension GL_NV_gpu_shader5 : enable
|
||||
#ifdef GL_NV_gpu_shader5
|
||||
#define HAS_EXTENDED_TYPES 1
|
||||
#else
|
||||
#define HAS_EXTENDED_TYPES 0
|
||||
#endif
|
||||
#define BEGIN_PUSH_CONSTANTS
|
||||
#define END_PUSH_CONSTANTS
|
||||
#define UNIFORM(n) layout(location = n) uniform
|
||||
#define BINDING_INPUT_BUFFER 1
|
||||
#define BINDING_OUTPUT_BUFFER 0
|
||||
#endif
|
||||
|
||||
// --- Push Constants / Uniforms ---
|
||||
#ifdef VULKAN
|
||||
layout(push_constant) uniform PushConstants {
|
||||
uvec3 blocks_dim; // Offset 0
|
||||
uint bytes_per_block_log2; // Offset 12
|
||||
|
||||
uvec3 origin; // Offset 16
|
||||
uint slice_size; // Offset 28
|
||||
|
||||
uint block_size; // Offset 32
|
||||
uint x_shift; // Offset 36
|
||||
uint block_height; // Offset 40
|
||||
uint block_height_mask; // Offset 44
|
||||
|
||||
uint block_depth; // Offset 48
|
||||
uint block_depth_mask; // Offset 52
|
||||
int _pad; // Offset 56
|
||||
|
||||
ivec3 destination; // Offset 60
|
||||
} pc;
|
||||
#else
|
||||
BEGIN_PUSH_CONSTANTS
|
||||
UNIFORM(0) uvec3 origin;
|
||||
UNIFORM(1) ivec3 destination;
|
||||
UNIFORM(2) uint bytes_per_block_log2;
|
||||
UNIFORM(3) uint slice_size;
|
||||
UNIFORM(4) uint block_size;
|
||||
UNIFORM(5) uint x_shift;
|
||||
UNIFORM(6) uint block_height;
|
||||
UNIFORM(7) uint block_height_mask;
|
||||
UNIFORM(8) uint block_depth;
|
||||
UNIFORM(9) uint block_depth_mask;
|
||||
UNIFORM(10) uvec3 blocks_dim;
|
||||
END_PUSH_CONSTANTS
|
||||
#define pc // Map pc prefix to nothing for OpenGL compatibility
|
||||
#endif
|
||||
|
||||
// --- Buffers ---
|
||||
#if HAS_EXTENDED_TYPES
|
||||
layout(binding = BINDING_INPUT_BUFFER, std430) buffer InputBufferU8 { uint8_t u8data[]; };
|
||||
layout(binding = BINDING_INPUT_BUFFER, std430) buffer InputBufferU16 { uint16_t u16data[]; };
|
||||
#endif
|
||||
layout(binding = BINDING_INPUT_BUFFER, std430) buffer InputBufferU32 { uint u32data[]; };
|
||||
layout(binding = BINDING_INPUT_BUFFER, std430) buffer InputBufferU64 { uvec2 u64data[]; };
|
||||
layout(binding = BINDING_INPUT_BUFFER, std430) buffer InputBufferU128 { uvec4 u128data[]; };
|
||||
|
||||
layout(binding = BINDING_OUTPUT_BUFFER, std430) writeonly buffer OutputBuffer {
|
||||
uint out_u32[];
|
||||
};
|
||||
|
||||
// --- Constants ---
|
||||
layout(local_size_x = 8, local_size_y = 8, local_size_z = 4) in;
|
||||
|
||||
const uint GOB_SIZE_X = 64;
|
||||
const uint GOB_SIZE_Y = 8;
|
||||
const uint GOB_SIZE_Z = 1;
|
||||
const uint GOB_SIZE = GOB_SIZE_X * GOB_SIZE_Y * GOB_SIZE_Z;
|
||||
|
||||
const uint GOB_SIZE_X_SHIFT = 6;
|
||||
const uint GOB_SIZE_Y_SHIFT = 3;
|
||||
const uint GOB_SIZE_Z_SHIFT = 0;
|
||||
const uint GOB_SIZE_SHIFT = GOB_SIZE_X_SHIFT + GOB_SIZE_Y_SHIFT + GOB_SIZE_Z_SHIFT;
|
||||
const uvec2 SWIZZLE_MASK = uvec2(GOB_SIZE_X - 1u, GOB_SIZE_Y - 1u);
|
||||
|
||||
uint SwizzleTable(uint pos) {
|
||||
const uint t[8] = uint[](
|
||||
0x12100200, 0x13110301, 0x16140604, 0x17150705,
|
||||
0x1a180a08, 0x1b190b09, 0x1e1c0e0c, 0x1f1d0f0d
|
||||
);
|
||||
const uint i = pos >> 4;
|
||||
const uint h = (t[i / 4] >> ((i % 4) * 8)) & 0xff;
|
||||
return (h << 4) | (pos & 0xf);
|
||||
}
|
||||
|
||||
// --- Helpers ---
|
||||
uint SwizzleOffset(uvec2 pos) {
|
||||
pos &= SWIZZLE_MASK;
|
||||
return SwizzleTable(pos.y * 64u + pos.x);
|
||||
}
|
||||
|
||||
uvec4 ReadTexel(uint offset) {
|
||||
uint bpl2 = pc.bytes_per_block_log2;
|
||||
switch (bpl2) {
|
||||
#if HAS_EXTENDED_TYPES
|
||||
case 0u: return uvec4(u8data[offset], 0u, 0u, 0u);
|
||||
case 1u: return uvec4(u16data[offset / 2u], 0u, 0u, 0u);
|
||||
#else
|
||||
case 0u: return uvec4(bitfieldExtract(u32data[offset / 4u], int((offset * 8u) & 24u), 8), 0u, 0u, 0u);
|
||||
case 1u: return uvec4(bitfieldExtract(u32data[offset / 4u], int((offset * 8u) & 16u), 16), 0u, 0u, 0u);
|
||||
#endif
|
||||
case 2u: return uvec4(u32data[offset / 4u], 0u, 0u, 0u);
|
||||
case 3u: return uvec4(u64data[offset / 8u], 0u, 0u);
|
||||
case 4u: return u128data[offset / 16u];
|
||||
}
|
||||
return uvec4(0u);
|
||||
}
|
||||
|
||||
void main() {
|
||||
uvec3 block_coord = gl_GlobalInvocationID;
|
||||
if (any(greaterThanEqual(block_coord, pc.blocks_dim))) {
|
||||
return;
|
||||
}
|
||||
|
||||
uint bytes_per_block = 1u << pc.bytes_per_block_log2;
|
||||
// Origin is in pixels, divide by 4 for block-space (e.g. BCn formats)
|
||||
uvec3 pos;
|
||||
pos.x = (block_coord.x + (pc.origin.x >> 2u)) * bytes_per_block;
|
||||
pos.y = block_coord.y + (pc.origin.y >> 2u);
|
||||
pos.z = block_coord.z + pc.origin.z;
|
||||
|
||||
uint swizzle = SwizzleOffset(pos.xy);
|
||||
uint block_y = pos.y >> GOB_SIZE_Y_SHIFT;
|
||||
uint offset = 0u;
|
||||
// Apply block-linear offsets
|
||||
offset += (pos.z >> pc.block_depth) * pc.slice_size;
|
||||
offset += (pos.z & pc.block_depth_mask) << (GOB_SIZE_SHIFT + pc.block_height);
|
||||
offset += (block_y >> pc.block_height) * pc.block_size;
|
||||
offset += (block_y & pc.block_height_mask) << GOB_SIZE_SHIFT;
|
||||
offset += (pos.x >> GOB_SIZE_X_SHIFT) << pc.x_shift;
|
||||
offset += swizzle;
|
||||
|
||||
uvec4 texel = ReadTexel(offset);
|
||||
|
||||
// Calculate linear output index
|
||||
uint block_index = block_coord.x +
|
||||
(block_coord.y * pc.blocks_dim.x) +
|
||||
(block_coord.z * pc.blocks_dim.x * pc.blocks_dim.y);
|
||||
uint out_idx = block_index * (bytes_per_block >> 2u);
|
||||
|
||||
out_u32[out_idx] = texel.x;
|
||||
out_u32[out_idx + 1u] = texel.y;
|
||||
if (pc.bytes_per_block_log2 == 4u) {
|
||||
out_u32[out_idx + 2u] = texel.z;
|
||||
out_u32[out_idx + 3u] = texel.w;
|
||||
}
|
||||
}
|
||||
@@ -33,7 +33,7 @@
|
||||
BEGIN_PUSH_CONSTANTS
|
||||
UNIFORM(0) uvec2 origin;
|
||||
UNIFORM(1) ivec2 destination;
|
||||
UNIFORM(2) uint bytes_per_block;
|
||||
UNIFORM(2) uint bytes_per_block_log2;
|
||||
UNIFORM(3) uint pitch;
|
||||
END_PUSH_CONSTANTS
|
||||
|
||||
@@ -47,26 +47,26 @@ layout(binding = BINDING_INPUT_BUFFER, std430) readonly buffer InputBufferU128 {
|
||||
|
||||
layout(binding = BINDING_OUTPUT_IMAGE) writeonly uniform uimage2D output_image;
|
||||
|
||||
layout(local_size_x = 32, local_size_y = 32, local_size_z = 1) in;
|
||||
layout(local_size_x = 32, local_size_y = 8, local_size_z = 1) in;
|
||||
|
||||
uvec4 ReadTexel(uint offset) {
|
||||
switch (bytes_per_block) {
|
||||
switch (bytes_per_block_log2) {
|
||||
#if HAS_EXTENDED_TYPES
|
||||
case 1:
|
||||
case 0:
|
||||
return uvec4(u8data[offset], 0, 0, 0);
|
||||
case 2:
|
||||
case 1:
|
||||
return uvec4(u16data[offset / 2], 0, 0, 0);
|
||||
#else
|
||||
case 1:
|
||||
case 0:
|
||||
return uvec4(bitfieldExtract(u32data[offset / 4], int((offset * 8) & 24), 8), 0, 0, 0);
|
||||
case 2:
|
||||
case 1:
|
||||
return uvec4(bitfieldExtract(u32data[offset / 4], int((offset * 8) & 16), 16), 0, 0, 0);
|
||||
#endif
|
||||
case 4:
|
||||
case 2:
|
||||
return uvec4(u32data[offset / 4], 0, 0, 0);
|
||||
case 8:
|
||||
case 3:
|
||||
return uvec4(u64data[offset / 8], 0, 0);
|
||||
case 16:
|
||||
case 4:
|
||||
return u128data[offset / 16];
|
||||
}
|
||||
return uvec4(0);
|
||||
@@ -76,7 +76,7 @@ void main() {
|
||||
uvec2 pos = gl_GlobalInvocationID.xy + origin;
|
||||
|
||||
uint offset = 0;
|
||||
offset += pos.x * bytes_per_block;
|
||||
offset += pos.x << bytes_per_block_log2;
|
||||
offset += pos.y * pitch;
|
||||
|
||||
const uvec4 texel = ReadTexel(offset);
|
||||
|
||||
@@ -556,7 +556,7 @@ void TextureCacheRuntime::Finish() {
|
||||
glFinish();
|
||||
}
|
||||
|
||||
StagingBufferMap TextureCacheRuntime::UploadStagingBuffer(size_t size, bool deferred) {
|
||||
StagingBufferMap TextureCacheRuntime::UploadStagingBuffer(size_t size) {
|
||||
return staging_buffer_pool.RequestUploadBuffer(size);
|
||||
}
|
||||
|
||||
@@ -651,8 +651,7 @@ void TextureCacheRuntime::BlitFramebuffer(Framebuffer* dst, Framebuffer* src,
|
||||
}
|
||||
|
||||
void TextureCacheRuntime::AccelerateImageUpload(Image& image, const StagingBufferMap& map,
|
||||
std::span<const SwizzleParameters> swizzles,
|
||||
u32 z_start, u32 z_count) {
|
||||
std::span<const SwizzleParameters> swizzles) {
|
||||
switch (image.info.type) {
|
||||
case ImageType::e2D:
|
||||
if (IsPixelFormatASTC(image.info.format)) {
|
||||
|
||||
@@ -77,7 +77,7 @@ public:
|
||||
|
||||
void FlushDeferredClear() {}
|
||||
|
||||
StagingBufferMap UploadStagingBuffer(size_t size, bool deferred = false);
|
||||
StagingBufferMap UploadStagingBuffer(size_t size);
|
||||
|
||||
StagingBufferMap DownloadStagingBuffer(size_t size, bool deferred = false);
|
||||
|
||||
@@ -121,8 +121,7 @@ public:
|
||||
Tegra::Engines::Fermi2D::Operation operation);
|
||||
|
||||
void AccelerateImageUpload(Image& image, const StagingBufferMap& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles,
|
||||
u32 z_start, u32 z_count);
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles);
|
||||
|
||||
void InsertUploadMemoryBarrier();
|
||||
|
||||
@@ -229,8 +228,6 @@ public:
|
||||
|
||||
bool ScaleDown(bool ignore = false);
|
||||
|
||||
u64 allocation_tick;
|
||||
|
||||
private:
|
||||
void CopyBufferToImage(const VideoCommon::BufferImageCopy& copy, size_t buffer_offset);
|
||||
|
||||
|
||||
@@ -4,6 +4,7 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <bit>
|
||||
#include <span>
|
||||
#include <string_view>
|
||||
|
||||
@@ -116,7 +117,7 @@ void UtilShaders::ASTCDecode(Image& image, const StagingBufferMap& map,
|
||||
|
||||
void UtilShaders::BlockLinearUpload2D(Image& image, const StagingBufferMap& map,
|
||||
std::span<const SwizzleParameters> swizzles) {
|
||||
static constexpr Extent3D WORKGROUP_SIZE{32, 32, 1};
|
||||
static constexpr Extent3D WORKGROUP_SIZE{32, 8, 1};
|
||||
static constexpr GLuint BINDING_INPUT_BUFFER = 0;
|
||||
static constexpr GLuint BINDING_OUTPUT_IMAGE = 0;
|
||||
|
||||
@@ -151,7 +152,7 @@ void UtilShaders::BlockLinearUpload2D(Image& image, const StagingBufferMap& map,
|
||||
|
||||
void UtilShaders::BlockLinearUpload3D(Image& image, const StagingBufferMap& map,
|
||||
std::span<const SwizzleParameters> swizzles) {
|
||||
static constexpr Extent3D WORKGROUP_SIZE{16, 8, 8};
|
||||
static constexpr Extent3D WORKGROUP_SIZE{16, 8, 2};
|
||||
static constexpr GLuint BINDING_INPUT_BUFFER = 0;
|
||||
static constexpr GLuint BINDING_OUTPUT_IMAGE = 0;
|
||||
|
||||
@@ -189,12 +190,12 @@ void UtilShaders::BlockLinearUpload3D(Image& image, const StagingBufferMap& map,
|
||||
|
||||
void UtilShaders::PitchUpload(Image& image, const StagingBufferMap& map,
|
||||
std::span<const SwizzleParameters> swizzles) {
|
||||
static constexpr Extent3D WORKGROUP_SIZE{32, 32, 1};
|
||||
static constexpr Extent3D WORKGROUP_SIZE{32, 8, 1};
|
||||
static constexpr GLuint BINDING_INPUT_BUFFER = 0;
|
||||
static constexpr GLuint BINDING_OUTPUT_IMAGE = 0;
|
||||
static constexpr GLuint LOC_ORIGIN = 0;
|
||||
static constexpr GLuint LOC_DESTINATION = 1;
|
||||
static constexpr GLuint LOC_BYTES_PER_BLOCK = 2;
|
||||
static constexpr GLuint LOC_BYTES_PER_BLOCK_LOG2 = 2;
|
||||
static constexpr GLuint LOC_PITCH = 3;
|
||||
|
||||
const u32 bytes_per_block = BytesPerBlock(image.info.format);
|
||||
@@ -208,7 +209,7 @@ void UtilShaders::PitchUpload(Image& image, const StagingBufferMap& map,
|
||||
glFlushMappedNamedBufferRange(map.buffer, map.offset, image.guest_size_bytes);
|
||||
glUniform2ui(LOC_ORIGIN, 0, 0);
|
||||
glUniform2i(LOC_DESTINATION, 0, 0);
|
||||
glUniform1ui(LOC_BYTES_PER_BLOCK, bytes_per_block);
|
||||
glUniform1ui(LOC_BYTES_PER_BLOCK_LOG2, static_cast<GLuint>(std::countr_zero(bytes_per_block)));
|
||||
glUniform1ui(LOC_PITCH, pitch);
|
||||
glBindImageTexture(BINDING_OUTPUT_IMAGE, image.StorageHandle(), 0, GL_FALSE, 0, GL_WRITE_ONLY,
|
||||
format);
|
||||
|
||||
@@ -16,13 +16,16 @@
|
||||
#include "common/common_types.h"
|
||||
#include "common/div_ceil.h"
|
||||
#include "common/vector_math.h"
|
||||
#include <bit>
|
||||
#include "video_core/host_shaders/astc_decoder_comp_spv.h"
|
||||
#include "video_core/host_shaders/queries_prefix_scan_sum_comp_spv.h"
|
||||
#include "video_core/host_shaders/queries_prefix_scan_sum_nosubgroups_comp_spv.h"
|
||||
#include "video_core/host_shaders/resolve_conditional_render_comp_spv.h"
|
||||
#include "video_core/host_shaders/vulkan_quad_indexed_comp_spv.h"
|
||||
#include "video_core/host_shaders/vulkan_uint8_comp_spv.h"
|
||||
#include "video_core/host_shaders/block_linear_unswizzle_3d_bcn_comp_spv.h"
|
||||
#include "video_core/host_shaders/block_linear_unswizzle_2d_comp_spv.h"
|
||||
#include "video_core/host_shaders/block_linear_unswizzle_3d_comp_spv.h"
|
||||
#include "video_core/host_shaders/pitch_unswizzle_comp_spv.h"
|
||||
#include "video_core/renderer_vulkan/vk_compute_pass.h"
|
||||
#include "video_core/surface.h"
|
||||
#include "video_core/renderer_vulkan/vk_descriptor_pool.h"
|
||||
@@ -185,6 +188,26 @@ struct AstcPushConstants {
|
||||
u32 block_height_mask;
|
||||
};
|
||||
|
||||
struct BlockLinear3DImagePushConstants {
|
||||
alignas(16) std::array<u32, 3> origin;
|
||||
alignas(16) std::array<s32, 3> destination;
|
||||
u32 bytes_per_block_log2;
|
||||
u32 slice_size;
|
||||
u32 block_size;
|
||||
u32 x_shift;
|
||||
u32 block_height;
|
||||
u32 block_height_mask;
|
||||
u32 block_depth;
|
||||
u32 block_depth_mask;
|
||||
};
|
||||
|
||||
struct PitchUnswizzlePushConstants {
|
||||
std::array<u32, 2> origin;
|
||||
std::array<s32, 2> destination;
|
||||
u32 bytes_per_block_log2;
|
||||
u32 pitch;
|
||||
};
|
||||
|
||||
struct QueriesPrefixScanPushConstants {
|
||||
u32 min_accumulation_base;
|
||||
u32 max_accumulation_base;
|
||||
@@ -616,260 +639,230 @@ void ASTCDecoderPass::Assemble(Image& image, const StagingBufferRef& map,
|
||||
scheduler.Finish();
|
||||
}
|
||||
|
||||
constexpr u32 BL3D_BINDING_INPUT_BUFFER = 0;
|
||||
constexpr u32 BL3D_BINDING_OUTPUT_BUFFER = 1;
|
||||
namespace {
|
||||
|
||||
struct alignas(16) BlockLinearUnswizzle3DPushConstants {
|
||||
u32 blocks_dim[3]; // Offset 0
|
||||
u32 bytes_per_block_log2; // Offset 12
|
||||
|
||||
u32 origin[3]; // Offset 16
|
||||
u32 slice_size; // Offset 28
|
||||
|
||||
u32 block_size; // Offset 32
|
||||
u32 x_shift; // Offset 36
|
||||
u32 block_height; // Offset 40
|
||||
u32 block_height_mask; // Offset 44
|
||||
|
||||
u32 block_depth; // Offset 48
|
||||
u32 block_depth_mask; // Offset 52
|
||||
s32 _pad; // Offset 56
|
||||
|
||||
s32 destination[3]; // Offset 60
|
||||
s32 _pad_end; // Offset 72
|
||||
};
|
||||
static_assert(sizeof(BlockLinearUnswizzle3DPushConstants) <= 128);
|
||||
|
||||
BlockLinearUnswizzle3DPass::BlockLinearUnswizzle3DPass(
|
||||
const Device& device_, Scheduler& scheduler_,
|
||||
DescriptorPool& descriptor_pool_,
|
||||
StagingBufferPool& staging_buffer_pool_,
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue_)
|
||||
: ComputePass(device_, scheduler_, descriptor_pool_,
|
||||
std::array<VkDescriptorSetLayoutBinding, 2>{{
|
||||
{
|
||||
.binding = BL3D_BINDING_INPUT_BUFFER,
|
||||
.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER, // block-linear input
|
||||
.descriptorCount = 1,
|
||||
.stageFlags = VK_SHADER_STAGE_COMPUTE_BIT,
|
||||
.pImmutableSamplers = nullptr,
|
||||
},
|
||||
{
|
||||
.binding = BL3D_BINDING_OUTPUT_BUFFER,
|
||||
.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER,
|
||||
.descriptorCount = 1,
|
||||
.stageFlags = VK_SHADER_STAGE_COMPUTE_BIT,
|
||||
.pImmutableSamplers = nullptr,
|
||||
},
|
||||
}},
|
||||
std::array<VkDescriptorUpdateTemplateEntry, 2>{{
|
||||
{
|
||||
.dstBinding = BL3D_BINDING_INPUT_BUFFER,
|
||||
.dstArrayElement = 0,
|
||||
.descriptorCount = 1,
|
||||
.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER,
|
||||
.offset = BL3D_BINDING_INPUT_BUFFER * sizeof(DescriptorUpdateEntry),
|
||||
.stride = sizeof(DescriptorUpdateEntry),
|
||||
},
|
||||
{
|
||||
.dstBinding = BL3D_BINDING_OUTPUT_BUFFER,
|
||||
.dstArrayElement = 0,
|
||||
.descriptorCount = 1,
|
||||
.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER,
|
||||
.offset = BL3D_BINDING_OUTPUT_BUFFER * sizeof(DescriptorUpdateEntry),
|
||||
.stride = sizeof(DescriptorUpdateEntry),
|
||||
}
|
||||
}},
|
||||
DescriptorBankInfo{
|
||||
.uniform_buffers = 0,
|
||||
.storage_buffers = 2,
|
||||
.texture_buffers = 0,
|
||||
.image_buffers = 0,
|
||||
.textures = 0,
|
||||
.images = 0,
|
||||
.score = 2,
|
||||
},
|
||||
COMPUTE_PUSH_CONSTANT_RANGE<sizeof(BlockLinearUnswizzle3DPushConstants)>,
|
||||
BLOCK_LINEAR_UNSWIZZLE_3D_BCN_COMP_SPV),
|
||||
scheduler{scheduler_},
|
||||
staging_buffer_pool{staging_buffer_pool_},
|
||||
compute_pass_descriptor_queue{compute_pass_descriptor_queue_} {}
|
||||
|
||||
BlockLinearUnswizzle3DPass::~BlockLinearUnswizzle3DPass() = default;
|
||||
|
||||
// God have mercy on my soul
|
||||
void BlockLinearUnswizzle3DPass::Unswizzle(
|
||||
Image& image,
|
||||
const StagingBufferRef& swizzled,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles,
|
||||
u32 z_start, u32 z_count)
|
||||
{
|
||||
using namespace VideoCommon::Accelerated;
|
||||
|
||||
const u32 MAX_BATCH_SLICES = (std::min)(z_count, image.info.size.depth);
|
||||
|
||||
// Allocate or grow to cover this batch's slice count
|
||||
image.AllocateComputeUnswizzleBuffer(MAX_BATCH_SLICES);
|
||||
|
||||
ASSERT(swizzles.size() == 1);
|
||||
const auto& sw = swizzles[0];
|
||||
const auto params = MakeBlockLinearSwizzle3DParams(sw, image.info);
|
||||
|
||||
const u32 blocks_x = (image.info.size.width + 3) / 4;
|
||||
const u32 blocks_y = (image.info.size.height + 3) / 4;
|
||||
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
for (u32 z_offset = 0; z_offset < z_count; z_offset += MAX_BATCH_SLICES) {
|
||||
const u32 current_chunk_slices = (std::min)(MAX_BATCH_SLICES, z_count - z_offset);
|
||||
const u32 current_z_start = z_start + z_offset;
|
||||
|
||||
UnswizzleChunk(image, swizzled, sw, params, blocks_x, blocks_y,
|
||||
current_z_start, current_chunk_slices);
|
||||
void RecordUnswizzleBeginBarrier(Scheduler& scheduler, VkPipeline vk_pipeline, VkImage vk_image,
|
||||
VkImageAspectFlags aspect_mask, bool is_initialized) {
|
||||
VkAccessFlags src_access = VK_ACCESS_NONE;
|
||||
VkImageLayout old_layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
VkPipelineStageFlags src_stage = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT;
|
||||
if (is_initialized) {
|
||||
src_access = VK_ACCESS_SHADER_WRITE_BIT;
|
||||
old_layout = VK_IMAGE_LAYOUT_GENERAL;
|
||||
src_stage = vk::PIPELINE_STAGE_GRAPHICS_COMPUTE_TRANSFER;
|
||||
}
|
||||
}
|
||||
|
||||
void BlockLinearUnswizzle3DPass::UnswizzleChunk(
|
||||
Image& image,
|
||||
const StagingBufferRef& swizzled,
|
||||
const VideoCommon::SwizzleParameters& sw,
|
||||
const BlockLinearSwizzle3DParams& params,
|
||||
u32 blocks_x, u32 blocks_y,
|
||||
u32 z_start, u32 z_count)
|
||||
{
|
||||
BlockLinearUnswizzle3DPushConstants pc{};
|
||||
pc.origin[0] = params.origin[0];
|
||||
pc.origin[1] = params.origin[1];
|
||||
pc.origin[2] = z_start; // Current chunk's Z start
|
||||
|
||||
pc.destination[0] = params.destination[0];
|
||||
pc.destination[1] = params.destination[1];
|
||||
pc.destination[2] = 0; // Shader writes to start of output buffer
|
||||
|
||||
pc.bytes_per_block_log2 = params.bytes_per_block_log2;
|
||||
pc.slice_size = params.slice_size;
|
||||
pc.block_size = params.block_size;
|
||||
pc.x_shift = params.x_shift;
|
||||
pc.block_height = params.block_height;
|
||||
pc.block_height_mask = params.block_height_mask;
|
||||
pc.block_depth = params.block_depth;
|
||||
pc.block_depth_mask = params.block_depth_mask;
|
||||
|
||||
pc.blocks_dim[0] = blocks_x;
|
||||
pc.blocks_dim[1] = blocks_y;
|
||||
pc.blocks_dim[2] = z_count; // Only process the count
|
||||
|
||||
compute_pass_descriptor_queue.Acquire(scheduler, 3);
|
||||
compute_pass_descriptor_queue.AddBuffer(swizzled.buffer,
|
||||
sw.buffer_offset + swizzled.offset,
|
||||
image.guest_size_bytes - sw.buffer_offset);
|
||||
compute_pass_descriptor_queue.AddBuffer(*image.compute_unswizzle_buffer, 0,
|
||||
image.compute_unswizzle_buffer_size);
|
||||
|
||||
const void* descriptor_data = compute_pass_descriptor_queue.UpdateData();
|
||||
const VkDescriptorSet set = descriptor_allocator.Commit();
|
||||
|
||||
const u32 gx = Common::DivCeil(blocks_x, 8u);
|
||||
const u32 gy = Common::DivCeil(blocks_y, 8u);
|
||||
const u32 gz = Common::DivCeil(z_count, 4u);
|
||||
|
||||
const u32 bytes_per_block = 1u << pc.bytes_per_block_log2;
|
||||
const VkDeviceSize output_slice_size =
|
||||
static_cast<VkDeviceSize>(blocks_x) * blocks_y * bytes_per_block;
|
||||
const VkDeviceSize barrier_size = output_slice_size * z_count;
|
||||
|
||||
const bool is_first_chunk = (z_start == 0);
|
||||
|
||||
const VkBuffer out_buffer = *image.compute_unswizzle_buffer;
|
||||
const VkImage dst_image = image.Handle();
|
||||
const VkImageAspectFlags aspect = image.AspectMask();
|
||||
const u32 image_width = image.info.size.width;
|
||||
const u32 image_height = image.info.size.height;
|
||||
|
||||
scheduler.Record([this, set, descriptor_data, pc, gx, gy, gz, z_start, z_count,
|
||||
barrier_size, is_first_chunk, out_buffer, dst_image, aspect,
|
||||
image_width, image_height
|
||||
](vk::CommandBuffer cmdbuf) {
|
||||
|
||||
if (dst_image == VK_NULL_HANDLE || out_buffer == VK_NULL_HANDLE) {
|
||||
return;
|
||||
}
|
||||
|
||||
device.GetLogical().UpdateDescriptorSet(set, *descriptor_template, descriptor_data);
|
||||
cmdbuf.BindPipeline(VK_PIPELINE_BIND_POINT_COMPUTE, *pipeline);
|
||||
cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_COMPUTE, *layout, 0, set, {});
|
||||
cmdbuf.PushConstants(*layout, VK_SHADER_STAGE_COMPUTE_BIT, 0, sizeof(pc), &pc);
|
||||
cmdbuf.Dispatch(gx, gy, gz);
|
||||
|
||||
// Single barrier for compute -> transfer (buffer ready, image transition)
|
||||
const VkBufferMemoryBarrier buffer_barrier{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER,
|
||||
.pNext = nullptr,
|
||||
.srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT,
|
||||
.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT,
|
||||
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
||||
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
||||
.buffer = out_buffer,
|
||||
.offset = 0,
|
||||
.size = barrier_size,
|
||||
};
|
||||
|
||||
// Image layout transition
|
||||
const VkImageMemoryBarrier pre_barrier{
|
||||
scheduler.Record([vk_pipeline, vk_image, aspect_mask, src_access, old_layout,
|
||||
src_stage](vk::CommandBuffer cmdbuf) {
|
||||
const VkImageMemoryBarrier image_barrier{
|
||||
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
|
||||
.pNext = nullptr,
|
||||
.srcAccessMask = is_first_chunk ? VkAccessFlags{} :
|
||||
static_cast<VkAccessFlags>(VK_ACCESS_TRANSFER_WRITE_BIT),
|
||||
.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT,
|
||||
.oldLayout = is_first_chunk ? VK_IMAGE_LAYOUT_UNDEFINED :
|
||||
VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
|
||||
.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
|
||||
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
||||
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
||||
.image = dst_image,
|
||||
.subresourceRange = {aspect, 0, 1, 0, 1},
|
||||
};
|
||||
|
||||
// Single barrier handles both buffer and image
|
||||
cmdbuf.PipelineBarrier(
|
||||
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
|
||||
VK_PIPELINE_STAGE_TRANSFER_BIT,
|
||||
0,
|
||||
nullptr, buffer_barrier, pre_barrier
|
||||
);
|
||||
|
||||
// Copy chunk to correct Z position in image
|
||||
const VkBufferImageCopy copy{
|
||||
.bufferOffset = 0, // Read from start of staging buffer
|
||||
.bufferRowLength = 0,
|
||||
.bufferImageHeight = 0,
|
||||
.imageSubresource = {aspect, 0, 0, 1},
|
||||
.imageOffset = {0, 0, static_cast<s32>(z_start)}, // Write to correct Z
|
||||
.imageExtent = {image_width, image_height, z_count},
|
||||
};
|
||||
cmdbuf.CopyBufferToImage(out_buffer, dst_image,
|
||||
VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, copy);
|
||||
|
||||
// Post-copy transition
|
||||
const VkImageMemoryBarrier post_barrier{
|
||||
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
|
||||
.pNext = nullptr,
|
||||
.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT,
|
||||
.srcAccessMask = src_access,
|
||||
.dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT,
|
||||
.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
|
||||
.oldLayout = old_layout,
|
||||
.newLayout = VK_IMAGE_LAYOUT_GENERAL,
|
||||
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
||||
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
||||
.image = dst_image,
|
||||
.subresourceRange = {aspect, 0, 1, 0, 1},
|
||||
.image = vk_image,
|
||||
.subresourceRange{
|
||||
.aspectMask = aspect_mask,
|
||||
.baseMipLevel = 0,
|
||||
.levelCount = VK_REMAINING_MIP_LEVELS,
|
||||
.baseArrayLayer = 0,
|
||||
.layerCount = VK_REMAINING_ARRAY_LAYERS,
|
||||
},
|
||||
};
|
||||
|
||||
cmdbuf.PipelineBarrier(
|
||||
VK_PIPELINE_STAGE_TRANSFER_BIT,
|
||||
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT | VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
|
||||
0,
|
||||
nullptr, nullptr, post_barrier
|
||||
);
|
||||
cmdbuf.PipelineBarrier(src_stage, VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, 0, image_barrier);
|
||||
cmdbuf.BindPipeline(VK_PIPELINE_BIND_POINT_COMPUTE, vk_pipeline);
|
||||
});
|
||||
}
|
||||
|
||||
void RecordUnswizzleEndBarrier(Scheduler& scheduler, VkImage vk_image,
|
||||
VkImageAspectFlags aspect_mask) {
|
||||
scheduler.Record([vk_image, aspect_mask](vk::CommandBuffer cmdbuf) {
|
||||
const VkImageMemoryBarrier image_barrier{
|
||||
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
|
||||
.pNext = nullptr,
|
||||
.srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT,
|
||||
.dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT |
|
||||
VK_ACCESS_TRANSFER_READ_BIT | VK_ACCESS_TRANSFER_WRITE_BIT,
|
||||
.oldLayout = VK_IMAGE_LAYOUT_GENERAL,
|
||||
.newLayout = VK_IMAGE_LAYOUT_GENERAL,
|
||||
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
||||
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
||||
.image = vk_image,
|
||||
.subresourceRange{
|
||||
.aspectMask = aspect_mask,
|
||||
.baseMipLevel = 0,
|
||||
.levelCount = VK_REMAINING_MIP_LEVELS,
|
||||
.baseArrayLayer = 0,
|
||||
.layerCount = VK_REMAINING_ARRAY_LAYERS,
|
||||
},
|
||||
};
|
||||
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
|
||||
vk::PIPELINE_STAGE_GRAPHICS_COMPUTE_TRANSFER, 0, image_barrier);
|
||||
});
|
||||
}
|
||||
|
||||
} // Anonymous namespace
|
||||
|
||||
BlockLinearUnswizzleImage2DPass::BlockLinearUnswizzleImage2DPass(
|
||||
const Device& device_, Scheduler& scheduler_, DescriptorPool& descriptor_pool_,
|
||||
StagingBufferPool& staging_buffer_pool_,
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue_)
|
||||
: ComputePass(device_, scheduler_, descriptor_pool_, ASTC_DESCRIPTOR_SET_BINDINGS,
|
||||
ASTC_PASS_DESCRIPTOR_UPDATE_TEMPLATE_ENTRY, ASTC_BANK_INFO,
|
||||
COMPUTE_PUSH_CONSTANT_RANGE<sizeof(BlockLinearSwizzle2DParams)>,
|
||||
BLOCK_LINEAR_UNSWIZZLE_2D_COMP_SPV),
|
||||
scheduler{scheduler_}, staging_buffer_pool{staging_buffer_pool_},
|
||||
compute_pass_descriptor_queue{compute_pass_descriptor_queue_} {}
|
||||
|
||||
BlockLinearUnswizzleImage2DPass::~BlockLinearUnswizzleImage2DPass() = default;
|
||||
|
||||
void BlockLinearUnswizzleImage2DPass::Unswizzle(
|
||||
Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles) {
|
||||
using namespace VideoCommon::Accelerated;
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
const VkImageAspectFlags aspect_mask = image.AspectMask();
|
||||
const VkImage vk_image = image.Handle();
|
||||
RecordUnswizzleBeginBarrier(scheduler, *pipeline, vk_image, aspect_mask,
|
||||
image.ExchangeInitialization());
|
||||
for (const VideoCommon::SwizzleParameters& swizzle : swizzles) {
|
||||
const size_t input_offset = swizzle.buffer_offset + map.offset;
|
||||
const u32 num_dispatches_x = Common::DivCeil(swizzle.num_tiles.width, 32U);
|
||||
const u32 num_dispatches_y = Common::DivCeil(swizzle.num_tiles.height, 8U);
|
||||
const u32 num_dispatches_z = image.info.resources.layers;
|
||||
|
||||
compute_pass_descriptor_queue.Acquire(scheduler, 2);
|
||||
compute_pass_descriptor_queue.AddBuffer(map.buffer, input_offset,
|
||||
image.guest_size_bytes - swizzle.buffer_offset);
|
||||
compute_pass_descriptor_queue.AddImage(image.StorageImageView(swizzle.level));
|
||||
const void* const descriptor_data{compute_pass_descriptor_queue.UpdateData()};
|
||||
|
||||
const auto params = MakeBlockLinearSwizzle2DParams(swizzle, image.info);
|
||||
scheduler.Record([this, num_dispatches_x, num_dispatches_y, num_dispatches_z, params,
|
||||
descriptor_data](vk::CommandBuffer cmdbuf) {
|
||||
const VkDescriptorSet set = descriptor_allocator.Commit();
|
||||
device.GetLogical().UpdateDescriptorSet(set, *descriptor_template, descriptor_data);
|
||||
cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_COMPUTE, *layout, 0, set, {});
|
||||
cmdbuf.PushConstants(*layout, VK_SHADER_STAGE_COMPUTE_BIT, params);
|
||||
cmdbuf.Dispatch(num_dispatches_x, num_dispatches_y, num_dispatches_z);
|
||||
});
|
||||
}
|
||||
RecordUnswizzleEndBarrier(scheduler, vk_image, aspect_mask);
|
||||
scheduler.Finish();
|
||||
}
|
||||
|
||||
BlockLinearUnswizzleImage3DPass::BlockLinearUnswizzleImage3DPass(
|
||||
const Device& device_, Scheduler& scheduler_, DescriptorPool& descriptor_pool_,
|
||||
StagingBufferPool& staging_buffer_pool_,
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue_)
|
||||
: ComputePass(device_, scheduler_, descriptor_pool_, ASTC_DESCRIPTOR_SET_BINDINGS,
|
||||
ASTC_PASS_DESCRIPTOR_UPDATE_TEMPLATE_ENTRY, ASTC_BANK_INFO,
|
||||
COMPUTE_PUSH_CONSTANT_RANGE<sizeof(BlockLinear3DImagePushConstants)>,
|
||||
BLOCK_LINEAR_UNSWIZZLE_3D_COMP_SPV),
|
||||
scheduler{scheduler_}, staging_buffer_pool{staging_buffer_pool_},
|
||||
compute_pass_descriptor_queue{compute_pass_descriptor_queue_} {}
|
||||
|
||||
BlockLinearUnswizzleImage3DPass::~BlockLinearUnswizzleImage3DPass() = default;
|
||||
|
||||
void BlockLinearUnswizzleImage3DPass::Unswizzle(
|
||||
Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles) {
|
||||
using namespace VideoCommon::Accelerated;
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
const VkImageAspectFlags aspect_mask = image.AspectMask();
|
||||
const VkImage vk_image = image.Handle();
|
||||
RecordUnswizzleBeginBarrier(scheduler, *pipeline, vk_image, aspect_mask,
|
||||
image.ExchangeInitialization());
|
||||
for (const VideoCommon::SwizzleParameters& swizzle : swizzles) {
|
||||
const size_t input_offset = swizzle.buffer_offset + map.offset;
|
||||
const u32 num_dispatches_x = Common::DivCeil(swizzle.num_tiles.width, 16U);
|
||||
const u32 num_dispatches_y = Common::DivCeil(swizzle.num_tiles.height, 8U);
|
||||
const u32 num_dispatches_z = Common::DivCeil(swizzle.num_tiles.depth, 2U);
|
||||
|
||||
compute_pass_descriptor_queue.Acquire(scheduler, 2);
|
||||
compute_pass_descriptor_queue.AddBuffer(map.buffer, input_offset,
|
||||
image.guest_size_bytes - swizzle.buffer_offset);
|
||||
compute_pass_descriptor_queue.AddImage(image.StorageImageView(swizzle.level));
|
||||
const void* const descriptor_data{compute_pass_descriptor_queue.UpdateData()};
|
||||
|
||||
const auto p = MakeBlockLinearSwizzle3DParams(swizzle, image.info);
|
||||
const BlockLinear3DImagePushConstants params{
|
||||
.origin = p.origin,
|
||||
.destination = p.destination,
|
||||
.bytes_per_block_log2 = p.bytes_per_block_log2,
|
||||
.slice_size = p.slice_size,
|
||||
.block_size = p.block_size,
|
||||
.x_shift = p.x_shift,
|
||||
.block_height = p.block_height,
|
||||
.block_height_mask = p.block_height_mask,
|
||||
.block_depth = p.block_depth,
|
||||
.block_depth_mask = p.block_depth_mask,
|
||||
};
|
||||
scheduler.Record([this, num_dispatches_x, num_dispatches_y, num_dispatches_z, params,
|
||||
descriptor_data](vk::CommandBuffer cmdbuf) {
|
||||
const VkDescriptorSet set = descriptor_allocator.Commit();
|
||||
device.GetLogical().UpdateDescriptorSet(set, *descriptor_template, descriptor_data);
|
||||
cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_COMPUTE, *layout, 0, set, {});
|
||||
cmdbuf.PushConstants(*layout, VK_SHADER_STAGE_COMPUTE_BIT, params);
|
||||
cmdbuf.Dispatch(num_dispatches_x, num_dispatches_y, num_dispatches_z);
|
||||
});
|
||||
}
|
||||
RecordUnswizzleEndBarrier(scheduler, vk_image, aspect_mask);
|
||||
scheduler.Finish();
|
||||
}
|
||||
|
||||
PitchUnswizzlePass::PitchUnswizzlePass(const Device& device_, Scheduler& scheduler_,
|
||||
DescriptorPool& descriptor_pool_,
|
||||
StagingBufferPool& staging_buffer_pool_,
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue_)
|
||||
: ComputePass(device_, scheduler_, descriptor_pool_, ASTC_DESCRIPTOR_SET_BINDINGS,
|
||||
ASTC_PASS_DESCRIPTOR_UPDATE_TEMPLATE_ENTRY, ASTC_BANK_INFO,
|
||||
COMPUTE_PUSH_CONSTANT_RANGE<sizeof(PitchUnswizzlePushConstants)>,
|
||||
PITCH_UNSWIZZLE_COMP_SPV),
|
||||
scheduler{scheduler_}, staging_buffer_pool{staging_buffer_pool_},
|
||||
compute_pass_descriptor_queue{compute_pass_descriptor_queue_} {}
|
||||
|
||||
PitchUnswizzlePass::~PitchUnswizzlePass() = default;
|
||||
|
||||
void PitchUnswizzlePass::Unswizzle(Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles) {
|
||||
const u32 bytes_per_block = VideoCore::Surface::BytesPerBlock(image.info.format);
|
||||
ASSERT(std::has_single_bit(bytes_per_block));
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
const VkImageAspectFlags aspect_mask = image.AspectMask();
|
||||
const VkImage vk_image = image.Handle();
|
||||
RecordUnswizzleBeginBarrier(scheduler, *pipeline, vk_image, aspect_mask,
|
||||
image.ExchangeInitialization());
|
||||
for (const VideoCommon::SwizzleParameters& swizzle : swizzles) {
|
||||
const size_t input_offset = swizzle.buffer_offset + map.offset;
|
||||
const u32 num_dispatches_x = Common::DivCeil(swizzle.num_tiles.width, 32U);
|
||||
const u32 num_dispatches_y = Common::DivCeil(swizzle.num_tiles.height, 8U);
|
||||
|
||||
compute_pass_descriptor_queue.Acquire(scheduler, 2);
|
||||
compute_pass_descriptor_queue.AddBuffer(map.buffer, input_offset,
|
||||
image.guest_size_bytes - swizzle.buffer_offset);
|
||||
compute_pass_descriptor_queue.AddImage(image.StorageImageView(swizzle.level));
|
||||
const void* const descriptor_data{compute_pass_descriptor_queue.UpdateData()};
|
||||
|
||||
const PitchUnswizzlePushConstants params{
|
||||
.origin = {0, 0},
|
||||
.destination = {0, 0},
|
||||
.bytes_per_block_log2 = static_cast<u32>(std::countr_zero(bytes_per_block)),
|
||||
.pitch = image.info.pitch,
|
||||
};
|
||||
scheduler.Record([this, num_dispatches_x, num_dispatches_y, params,
|
||||
descriptor_data](vk::CommandBuffer cmdbuf) {
|
||||
const VkDescriptorSet set = descriptor_allocator.Commit();
|
||||
device.GetLogical().UpdateDescriptorSet(set, *descriptor_template, descriptor_data);
|
||||
cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_COMPUTE, *layout, 0, set, {});
|
||||
cmdbuf.PushConstants(*layout, VK_SHADER_STAGE_COMPUTE_BIT, params);
|
||||
cmdbuf.Dispatch(num_dispatches_x, num_dispatches_y, 1);
|
||||
});
|
||||
}
|
||||
RecordUnswizzleEndBarrier(scheduler, vk_image, aspect_mask);
|
||||
scheduler.Finish();
|
||||
}
|
||||
|
||||
} // namespace Vulkan
|
||||
|
||||
@@ -17,7 +17,6 @@
|
||||
#include "video_core/texture_cache/types.h"
|
||||
#include "video_core/vulkan_common/vulkan_memory_allocator.h"
|
||||
#include "video_core/vulkan_common/vulkan_wrapper.h"
|
||||
#include "video_core/texture_cache/accelerated_swizzle.h"
|
||||
|
||||
namespace VideoCommon {
|
||||
struct SwizzleParameters;
|
||||
@@ -25,8 +24,6 @@ struct SwizzleParameters;
|
||||
|
||||
namespace Vulkan {
|
||||
|
||||
using VideoCommon::Accelerated::BlockLinearSwizzle3DParams;
|
||||
|
||||
class Device;
|
||||
class StagingBufferPool;
|
||||
class Scheduler;
|
||||
@@ -137,26 +134,50 @@ private:
|
||||
MemoryAllocator& memory_allocator;
|
||||
};
|
||||
|
||||
class BlockLinearUnswizzle3DPass final : public ComputePass {
|
||||
class BlockLinearUnswizzleImage2DPass final : public ComputePass {
|
||||
public:
|
||||
explicit BlockLinearUnswizzle3DPass(const Device& device_, Scheduler& scheduler_,
|
||||
DescriptorPool& descriptor_pool_,
|
||||
StagingBufferPool& staging_buffer_pool_,
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue_);
|
||||
~BlockLinearUnswizzle3DPass();
|
||||
explicit BlockLinearUnswizzleImage2DPass(
|
||||
const Device& device_, Scheduler& scheduler_, DescriptorPool& descriptor_pool_,
|
||||
StagingBufferPool& staging_buffer_pool_,
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue_);
|
||||
~BlockLinearUnswizzleImage2DPass();
|
||||
|
||||
void Unswizzle(Image& image,
|
||||
const StagingBufferRef& swizzled,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles,
|
||||
u32 z_start, u32 z_count);
|
||||
void Unswizzle(Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles);
|
||||
|
||||
void UnswizzleChunk(
|
||||
Image& image,
|
||||
const StagingBufferRef& swizzled,
|
||||
const VideoCommon::SwizzleParameters& sw,
|
||||
const BlockLinearSwizzle3DParams& params,
|
||||
u32 blocks_x, u32 blocks_y,
|
||||
u32 z_start, u32 z_count);
|
||||
private:
|
||||
Scheduler& scheduler;
|
||||
StagingBufferPool& staging_buffer_pool;
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue;
|
||||
};
|
||||
|
||||
class BlockLinearUnswizzleImage3DPass final : public ComputePass {
|
||||
public:
|
||||
explicit BlockLinearUnswizzleImage3DPass(
|
||||
const Device& device_, Scheduler& scheduler_, DescriptorPool& descriptor_pool_,
|
||||
StagingBufferPool& staging_buffer_pool_,
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue_);
|
||||
~BlockLinearUnswizzleImage3DPass();
|
||||
|
||||
void Unswizzle(Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles);
|
||||
|
||||
private:
|
||||
Scheduler& scheduler;
|
||||
StagingBufferPool& staging_buffer_pool;
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue;
|
||||
};
|
||||
|
||||
class PitchUnswizzlePass final : public ComputePass {
|
||||
public:
|
||||
explicit PitchUnswizzlePass(const Device& device_, Scheduler& scheduler_,
|
||||
DescriptorPool& descriptor_pool_,
|
||||
StagingBufferPool& staging_buffer_pool_,
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue_);
|
||||
~PitchUnswizzlePass();
|
||||
|
||||
void Unswizzle(Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles);
|
||||
|
||||
private:
|
||||
Scheduler& scheduler;
|
||||
|
||||
@@ -160,6 +160,73 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
|
||||
info.size.depth == 1;
|
||||
}
|
||||
|
||||
[[nodiscard]] VkFormat UnswizzleStorageFormat(u32 bytes_per_block) {
|
||||
switch (bytes_per_block) {
|
||||
case 1:
|
||||
return VK_FORMAT_R8_UINT;
|
||||
case 2:
|
||||
return VK_FORMAT_R16_UINT;
|
||||
case 4:
|
||||
return VK_FORMAT_R32_UINT;
|
||||
case 8:
|
||||
return VK_FORMAT_R32G32_UINT;
|
||||
case 16:
|
||||
return VK_FORMAT_R32G32B32A32_UINT;
|
||||
default:
|
||||
return VK_FORMAT_UNDEFINED;
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]] bool IsUnswizzleStorageFormatSupported(const Device& device, u32 bytes_per_block) {
|
||||
switch (bytes_per_block) {
|
||||
case 1:
|
||||
return device.IsStorageBuffer8BitAccessSupported();
|
||||
case 2:
|
||||
return device.IsStorageBuffer16BitAccessSupported();
|
||||
case 4:
|
||||
case 8:
|
||||
case 16:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]] bool IsUnswizzleAcceleratedFormat(const Device& device, PixelFormat format) {
|
||||
if (IsPixelFormatASTC(format) || VideoCore::Surface::IsPixelFormatBCn(format)) {
|
||||
return false;
|
||||
}
|
||||
if (VideoCore::Surface::GetFormatType(format) != SurfaceType::ColorTexture) {
|
||||
return false;
|
||||
}
|
||||
return IsUnswizzleStorageFormatSupported(device, BytesPerBlock(format));
|
||||
}
|
||||
|
||||
[[nodiscard]] bool WillUseAcceleratedUnswizzle(const Device& device, const ImageInfo& info) {
|
||||
switch (info.type) {
|
||||
case ImageType::e2D:
|
||||
case ImageType::e3D:
|
||||
case ImageType::Linear:
|
||||
break;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
if (info.num_samples > 1 || !device.IsKhrImageFormatListSupported()) {
|
||||
return false;
|
||||
}
|
||||
return IsUnswizzleAcceleratedFormat(device, info.format);
|
||||
}
|
||||
|
||||
[[nodiscard]] VkImageViewType UnswizzleStorageViewType(ImageType type) {
|
||||
if (type == ImageType::e3D) {
|
||||
return VK_IMAGE_VIEW_TYPE_3D;
|
||||
}
|
||||
if (type == ImageType::Linear) {
|
||||
return VK_IMAGE_VIEW_TYPE_2D;
|
||||
}
|
||||
return VK_IMAGE_VIEW_TYPE_2D_ARRAY;
|
||||
}
|
||||
|
||||
[[nodiscard]] VkImageCreateInfo MakeImageCreateInfo(const Device& device, const ImageInfo& info,
|
||||
std::optional<VkFormat> format_override = {}) {
|
||||
auto format_info =
|
||||
@@ -249,7 +316,8 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
|
||||
}
|
||||
|
||||
[[nodiscard]] vk::ImageView MakeStorageView(const vk::Device& device, u32 level, VkImage image,
|
||||
VkFormat format) {
|
||||
VkFormat format,
|
||||
VkImageViewType view_type = VK_IMAGE_VIEW_TYPE_2D_ARRAY) {
|
||||
static constexpr VkImageViewUsageCreateInfo storage_image_view_usage_create_info{
|
||||
.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_USAGE_CREATE_INFO,
|
||||
.pNext = nullptr,
|
||||
@@ -260,7 +328,7 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
|
||||
.pNext = &storage_image_view_usage_create_info,
|
||||
.flags = 0,
|
||||
.image = image,
|
||||
.viewType = VK_IMAGE_VIEW_TYPE_2D_ARRAY,
|
||||
.viewType = view_type,
|
||||
.format = format,
|
||||
.components{
|
||||
.r = VK_COMPONENT_SWIZZLE_IDENTITY,
|
||||
@@ -946,6 +1014,14 @@ TextureCacheRuntime::TextureCacheRuntime(const Device& device_, Scheduler& sched
|
||||
astc_decoder_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool,
|
||||
compute_pass_descriptor_queue, memory_allocator);
|
||||
}
|
||||
if (device.IsKhrImageFormatListSupported()) {
|
||||
bl_unswizzle_2d_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool,
|
||||
compute_pass_descriptor_queue);
|
||||
bl_unswizzle_3d_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool,
|
||||
compute_pass_descriptor_queue);
|
||||
pitch_unswizzle_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool,
|
||||
compute_pass_descriptor_queue);
|
||||
}
|
||||
if (!device.IsKhrImageFormatListSupported()) {
|
||||
return;
|
||||
}
|
||||
@@ -953,6 +1029,8 @@ TextureCacheRuntime::TextureCacheRuntime(const Device& device_, Scheduler& sched
|
||||
const auto image_format = static_cast<PixelFormat>(index_a);
|
||||
if (IsPixelFormatASTC(image_format) && !device.IsOptimalAstcSupported()) {
|
||||
view_formats[index_a].push_back(VK_FORMAT_A8B8G8R8_UNORM_PACK32);
|
||||
} else if (IsUnswizzleAcceleratedFormat(device, image_format)) {
|
||||
view_formats[index_a].push_back(UnswizzleStorageFormat(BytesPerBlock(image_format)));
|
||||
}
|
||||
for (size_t index_b = 0; index_b < VideoCore::Surface::MaxPixelFormat; index_b++) {
|
||||
const auto view_format = static_cast<PixelFormat>(index_b);
|
||||
@@ -964,18 +1042,14 @@ TextureCacheRuntime::TextureCacheRuntime(const Device& device_, Scheduler& sched
|
||||
}
|
||||
}
|
||||
|
||||
if (Settings::values.gpu_unswizzle_enabled.GetValue()) {
|
||||
bl3d_unswizzle_pass.emplace(device, scheduler, descriptor_pool,
|
||||
staging_buffer_pool, compute_pass_descriptor_queue);
|
||||
}
|
||||
}
|
||||
|
||||
void TextureCacheRuntime::Finish() {
|
||||
scheduler.Finish();
|
||||
}
|
||||
|
||||
StagingBufferRef TextureCacheRuntime::UploadStagingBuffer(size_t size, bool deferred) {
|
||||
return staging_buffer_pool.Request(size, MemoryUsage::Upload, deferred);
|
||||
StagingBufferRef TextureCacheRuntime::UploadStagingBuffer(size_t size) {
|
||||
return staging_buffer_pool.Request(size, MemoryUsage::Upload);
|
||||
}
|
||||
|
||||
StagingBufferRef TextureCacheRuntime::DownloadStagingBuffer(size_t size, bool deferred) {
|
||||
@@ -1899,6 +1973,10 @@ Image::Image(TextureCacheRuntime& runtime_, const ImageInfo& info_, GPUVAddr gpu
|
||||
flags |= VideoCommon::ImageFlagBits::Converted;
|
||||
flags |= VideoCommon::ImageFlagBits::CostlyLoad;
|
||||
}
|
||||
if (False(flags & VideoCommon::ImageFlagBits::Converted) &&
|
||||
WillUseAcceleratedUnswizzle(runtime->device, info)) {
|
||||
flags |= VideoCommon::ImageFlagBits::AcceleratedUpload;
|
||||
}
|
||||
if (runtime->device.HasDebuggingToolAttached()) {
|
||||
original_image.SetObjectNameEXT(VideoCommon::Name(*this).c_str());
|
||||
}
|
||||
@@ -1911,6 +1989,14 @@ Image::Image(TextureCacheRuntime& runtime_, const ImageInfo& info_, GPUVAddr gpu
|
||||
storage_image_views[level] =
|
||||
MakeStorageView(device, level, *original_image, storage_format);
|
||||
}
|
||||
} else if (True(flags & VideoCommon::ImageFlagBits::AcceleratedUpload)) {
|
||||
const auto& device = runtime->device.GetLogical();
|
||||
const VkFormat storage_format = UnswizzleStorageFormat(BytesPerBlock(info.format));
|
||||
const VkImageViewType view_type = UnswizzleStorageViewType(info.type);
|
||||
for (s32 level = 0; level < info.resources.levels; ++level) {
|
||||
storage_image_views[level] =
|
||||
MakeStorageView(device, level, *original_image, storage_format, view_type);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1927,48 +2013,6 @@ Image::~Image() {
|
||||
}
|
||||
}
|
||||
|
||||
void Image::AllocateComputeUnswizzleBuffer(u32 max_slices) {
|
||||
using VideoCore::Surface::BytesPerBlock;
|
||||
|
||||
const u32 block_bytes = BytesPerBlock(info.format); // 8 for BC1, 16 for BC6H
|
||||
const u32 block_width = 4;
|
||||
const u32 block_height = 4;
|
||||
|
||||
// BCn is 4x4x1 blocks
|
||||
const u32 blocks_x = (info.size.width + block_width - 1) / block_width;
|
||||
const u32 blocks_y = (info.size.height + block_height - 1) / block_height;
|
||||
const u32 blocks_z = (std::min)(max_slices, info.size.depth);
|
||||
|
||||
const u64 block_count =
|
||||
static_cast<u64>(blocks_x) *
|
||||
static_cast<u64>(blocks_y) *
|
||||
static_cast<u64>(blocks_z);
|
||||
|
||||
const VkDeviceSize required_size = block_count * block_bytes;
|
||||
if (has_compute_unswizzle_buffer && required_size <= compute_unswizzle_buffer_size) {
|
||||
return;
|
||||
}
|
||||
|
||||
compute_unswizzle_buffer_size = required_size;
|
||||
|
||||
VkBufferCreateInfo ci{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
|
||||
.pNext = nullptr,
|
||||
.flags = 0,
|
||||
.size = compute_unswizzle_buffer_size,
|
||||
.usage = VK_BUFFER_USAGE_STORAGE_BUFFER_BIT |
|
||||
VK_BUFFER_USAGE_TRANSFER_SRC_BIT,
|
||||
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
};
|
||||
|
||||
compute_unswizzle_buffer =
|
||||
runtime->memory_allocator.CreateBuffer(ci, MemoryUsage::DeviceLocal);
|
||||
|
||||
has_compute_unswizzle_buffer = true;
|
||||
}
|
||||
|
||||
void Image::UploadMemory(VkBuffer buffer, VkDeviceSize offset,
|
||||
std::span<const VideoCommon::BufferImageCopy> copies) {
|
||||
// TODO: Move this to another API
|
||||
@@ -2326,11 +2370,15 @@ VkImageView Image::StorageImageView(s32 level) noexcept {
|
||||
if (!view) {
|
||||
auto format_info =
|
||||
MaxwellToVK::SurfaceFormat(runtime->device, FormatType::Optimal, true, info.format);
|
||||
VkImageViewType view_type = VK_IMAGE_VIEW_TYPE_2D_ARRAY;
|
||||
if (WillUseAcceleratedAstcDecode(runtime->device, info)) {
|
||||
format_info.format = VK_FORMAT_A8B8G8R8_UNORM_PACK32;
|
||||
} else if (True(flags & VideoCommon::ImageFlagBits::AcceleratedUpload)) {
|
||||
format_info.format = UnswizzleStorageFormat(BytesPerBlock(info.format));
|
||||
view_type = UnswizzleStorageViewType(info.type);
|
||||
}
|
||||
view = MakeStorageView(runtime->device.GetLogical(), level, *(this->*current_image),
|
||||
format_info.format);
|
||||
format_info.format, view_type);
|
||||
}
|
||||
return *view;
|
||||
}
|
||||
@@ -3129,25 +3177,20 @@ VkRenderPass Framebuffer::RenderPassVariant(u32 color_clear_mask, bool depth_ste
|
||||
|
||||
void TextureCacheRuntime::AccelerateImageUpload(
|
||||
Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles,
|
||||
u32 z_start, u32 z_count) {
|
||||
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles) {
|
||||
if (IsPixelFormatASTC(image.info.format)) {
|
||||
return astc_decoder_pass->Assemble(image, map, swizzles);
|
||||
}
|
||||
|
||||
if (!Settings::values.gpu_unswizzle_enabled.GetValue() || !bl3d_unswizzle_pass) {
|
||||
if (IsPixelFormatBCn(image.info.format) && image.info.type == ImageType::e3D) {
|
||||
ASSERT(false && "GPU unswizzle is disabled for BCn 3D texture");
|
||||
}
|
||||
ASSERT(false);
|
||||
return;
|
||||
switch (image.info.type) {
|
||||
case ImageType::e2D:
|
||||
return bl_unswizzle_2d_pass->Unswizzle(image, map, swizzles);
|
||||
case ImageType::e3D:
|
||||
return bl_unswizzle_3d_pass->Unswizzle(image, map, swizzles);
|
||||
case ImageType::Linear:
|
||||
return pitch_unswizzle_pass->Unswizzle(image, map, swizzles);
|
||||
default:
|
||||
break;
|
||||
}
|
||||
|
||||
if (bl3d_unswizzle_pass && IsPixelFormatBCn(image.info.format) && image.info.type == ImageType::e3D && image.info.resources.levels == 1 && image.info.resources.layers == 1) {
|
||||
return bl3d_unswizzle_pass->Unswizzle(image, map, swizzles, z_start, z_count);
|
||||
}
|
||||
|
||||
ASSERT(false);
|
||||
}
|
||||
|
||||
|
||||
@@ -52,7 +52,7 @@ public:
|
||||
|
||||
void Finish();
|
||||
|
||||
StagingBufferRef UploadStagingBuffer(size_t size, bool deferred = false);
|
||||
StagingBufferRef UploadStagingBuffer(size_t size);
|
||||
|
||||
StagingBufferRef DownloadStagingBuffer(size_t size, bool deferred = false);
|
||||
|
||||
@@ -98,8 +98,7 @@ public:
|
||||
}
|
||||
|
||||
void AccelerateImageUpload(Image&, const StagingBufferRef&,
|
||||
std::span<const VideoCommon::SwizzleParameters>,
|
||||
u32 z_start, u32 z_count);
|
||||
std::span<const VideoCommon::SwizzleParameters>);
|
||||
|
||||
void InsertUploadMemoryBarrier() {}
|
||||
|
||||
@@ -157,8 +156,9 @@ public:
|
||||
BlitImageHelper& blit_image_helper;
|
||||
RenderPassCache& render_pass_cache;
|
||||
std::optional<ASTCDecoderPass> astc_decoder_pass;
|
||||
|
||||
std::optional<BlockLinearUnswizzle3DPass> bl3d_unswizzle_pass;
|
||||
std::optional<BlockLinearUnswizzleImage2DPass> bl_unswizzle_2d_pass;
|
||||
std::optional<BlockLinearUnswizzleImage3DPass> bl_unswizzle_3d_pass;
|
||||
std::optional<PitchUnswizzlePass> pitch_unswizzle_pass;
|
||||
const Settings::ResolutionScalingInfo& resolution;
|
||||
std::array<std::vector<VkFormat>, VideoCore::Surface::MaxPixelFormat> view_formats;
|
||||
|
||||
@@ -334,8 +334,6 @@ public:
|
||||
void DownloadMemory(const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::BufferImageCopy> copies);
|
||||
|
||||
void AllocateComputeUnswizzleImage();
|
||||
|
||||
[[nodiscard]] VkImage Handle() const noexcept {
|
||||
return *(this->*current_image);
|
||||
}
|
||||
@@ -361,10 +359,6 @@ public:
|
||||
|
||||
bool ScaleDown(bool ignore = false);
|
||||
|
||||
u64 allocation_tick;
|
||||
|
||||
friend class BlockLinearUnswizzle3DPass;
|
||||
|
||||
private:
|
||||
bool BlitScaleHelper(bool scale_up);
|
||||
|
||||
@@ -376,12 +370,6 @@ private:
|
||||
vk::Image original_image;
|
||||
vk::Image scaled_image;
|
||||
|
||||
vk::Buffer compute_unswizzle_buffer;
|
||||
VkDeviceSize compute_unswizzle_buffer_size = 0;
|
||||
bool has_compute_unswizzle_buffer = false;
|
||||
|
||||
void AllocateComputeUnswizzleBuffer(u32 max_slices);
|
||||
|
||||
// Use a pointer to field because it is relative, so that the object can be
|
||||
// moved without breaking the reference.
|
||||
vk::Image Image::*current_image{};
|
||||
|
||||
@@ -8,7 +8,6 @@
|
||||
|
||||
#include <limits>
|
||||
#include <optional>
|
||||
#include <bit>
|
||||
#include "common/container/unordered_map.h"
|
||||
#include <boost/container/small_vector.hpp>
|
||||
|
||||
@@ -76,41 +75,6 @@ TextureCache<P>::TextureCache(Runtime& runtime_, Tegra::MaxwellDeviceMemoryManag
|
||||
critical_memory = DEFAULT_CRITICAL_MEMORY + 1_GiB;
|
||||
minimum_memory = 0;
|
||||
}
|
||||
|
||||
const bool gpu_unswizzle_enabled = Settings::values.gpu_unswizzle_enabled.GetValue();
|
||||
|
||||
if (gpu_unswizzle_enabled) {
|
||||
switch (Settings::values.gpu_unswizzle_texture_size.GetValue()) {
|
||||
case Settings::GpuUnswizzleSize::VerySmall: gpu_unswizzle_maxsize = 16_MiB; break;
|
||||
case Settings::GpuUnswizzleSize::Small: gpu_unswizzle_maxsize = 32_MiB; break;
|
||||
case Settings::GpuUnswizzleSize::Normal: gpu_unswizzle_maxsize = 128_MiB; break;
|
||||
case Settings::GpuUnswizzleSize::Large: gpu_unswizzle_maxsize = 256_MiB; break;
|
||||
case Settings::GpuUnswizzleSize::VeryLarge: gpu_unswizzle_maxsize = 512_MiB; break;
|
||||
default: gpu_unswizzle_maxsize = 128_MiB; break;
|
||||
}
|
||||
|
||||
switch (Settings::values.gpu_unswizzle_stream_size.GetValue()) {
|
||||
case Settings::GpuUnswizzle::VeryLow: swizzle_chunk_size = 4_MiB; break;
|
||||
case Settings::GpuUnswizzle::Low: swizzle_chunk_size = 8_MiB; break;
|
||||
case Settings::GpuUnswizzle::Normal: swizzle_chunk_size = 16_MiB; break;
|
||||
case Settings::GpuUnswizzle::Medium: swizzle_chunk_size = 32_MiB; break;
|
||||
case Settings::GpuUnswizzle::High: swizzle_chunk_size = 64_MiB; break;
|
||||
default: swizzle_chunk_size = 16_MiB;
|
||||
}
|
||||
|
||||
switch (Settings::values.gpu_unswizzle_chunk_size.GetValue()) {
|
||||
case Settings::GpuUnswizzleChunk::VeryLow: swizzle_slices_per_batch = 32; break;
|
||||
case Settings::GpuUnswizzleChunk::Low: swizzle_slices_per_batch = 64; break;
|
||||
case Settings::GpuUnswizzleChunk::Normal: swizzle_slices_per_batch = 128; break;
|
||||
case Settings::GpuUnswizzleChunk::Medium: swizzle_slices_per_batch = 256; break;
|
||||
case Settings::GpuUnswizzleChunk::High: swizzle_slices_per_batch = 512; break;
|
||||
default: swizzle_slices_per_batch = 128;
|
||||
}
|
||||
} else {
|
||||
gpu_unswizzle_maxsize = 0;
|
||||
swizzle_chunk_size = 0;
|
||||
swizzle_slices_per_batch = 0;
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
@@ -180,7 +144,6 @@ void TextureCache<P>::TickFrame() {
|
||||
sentenced_framebuffers.Tick();
|
||||
sentenced_image_view.Tick();
|
||||
TickAsyncDecode();
|
||||
TickAsyncUnswizzle();
|
||||
|
||||
runtime.TickFrame();
|
||||
++frame_tick;
|
||||
@@ -1132,19 +1095,6 @@ void TextureCache<P>::RefreshContents(Image& image, ImageId image_id) {
|
||||
return;
|
||||
}
|
||||
|
||||
const bool gpu_unswizzle_enabled = Settings::values.gpu_unswizzle_enabled.GetValue();
|
||||
|
||||
if (gpu_unswizzle_enabled &&
|
||||
IsPixelFormatBCn(image.info.format) &&
|
||||
image.info.type == ImageType::e3D &&
|
||||
image.info.resources.levels == 1 &&
|
||||
image.info.resources.layers == 1 &&
|
||||
MapSizeBytes(image) >= gpu_unswizzle_maxsize &&
|
||||
False(image.flags & ImageFlagBits::GpuModified)) {
|
||||
|
||||
QueueAsyncUnswizzle(image, image_id);
|
||||
return;
|
||||
}
|
||||
auto staging = runtime.UploadStagingBuffer(MapSizeBytes(image));
|
||||
UploadImageContents(image, staging);
|
||||
runtime.InsertUploadMemoryBarrier();
|
||||
@@ -1160,7 +1110,7 @@ void TextureCache<P>::UploadImageContents(Image& image, StagingBuffer& staging)
|
||||
gpu_memory->ReadBlock(gpu_addr, mapped_span.data(), mapped_span.size_bytes(),
|
||||
VideoCommon::CacheType::NoTextureCache);
|
||||
const auto uploads = FullUploadSwizzles(image.info);
|
||||
runtime.AccelerateImageUpload(image, staging, FixSmallVectorADL(uploads), 0, 0);
|
||||
runtime.AccelerateImageUpload(image, staging, FixSmallVectorADL(uploads));
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -1372,20 +1322,6 @@ void TextureCache<P>::QueueAsyncDecode(Image& image, ImageId image_id) {
|
||||
texture_decode_worker.QueueWork(std::move(func));
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void TextureCache<P>::QueueAsyncUnswizzle(Image& image, ImageId image_id) {
|
||||
if (True(image.flags & ImageFlagBits::IsDecoding)) {
|
||||
return;
|
||||
}
|
||||
|
||||
image.flags |= ImageFlagBits::IsDecoding;
|
||||
|
||||
unswizzle_queue.push_back({
|
||||
.image_id = image_id,
|
||||
.info = image.info
|
||||
});
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void TextureCache<P>::TickAsyncDecode() {
|
||||
bool has_uploads{};
|
||||
@@ -1411,83 +1347,6 @@ void TextureCache<P>::TickAsyncDecode() {
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void TextureCache<P>::TickAsyncUnswizzle() {
|
||||
if (unswizzle_queue.empty()) {
|
||||
return;
|
||||
}
|
||||
|
||||
if(current_unswizzle_frame > 0) {
|
||||
current_unswizzle_frame--;
|
||||
return;
|
||||
}
|
||||
|
||||
PendingUnswizzle& task = unswizzle_queue.front();
|
||||
Image& image = slot_images[task.image_id];
|
||||
|
||||
if (!task.initialized) {
|
||||
task.total_size = MapSizeBytes(image);
|
||||
task.staging_buffer = runtime.UploadStagingBuffer(task.total_size, true);
|
||||
|
||||
const auto& info = image.info;
|
||||
const u32 bytes_per_block = BytesPerBlock(info.format);
|
||||
const u32 width_blocks = Common::DivCeil(info.size.width, 4u);
|
||||
const u32 height_blocks = Common::DivCeil(info.size.height, 4u);
|
||||
|
||||
const u32 stride = width_blocks * bytes_per_block;
|
||||
const u32 aligned_height = height_blocks;
|
||||
task.bytes_per_slice = static_cast<size_t>(stride) * aligned_height;
|
||||
task.last_submitted_offset = 0;
|
||||
task.initialized = true;
|
||||
}
|
||||
|
||||
// Read data
|
||||
if (task.current_offset < task.total_size) {
|
||||
const size_t remaining = task.total_size - task.current_offset;
|
||||
|
||||
size_t copy_amount = (std::min)(swizzle_chunk_size, remaining);
|
||||
|
||||
if (remaining > swizzle_chunk_size) {
|
||||
copy_amount = (copy_amount / task.bytes_per_slice) * task.bytes_per_slice;
|
||||
if (copy_amount == 0) copy_amount = task.bytes_per_slice;
|
||||
}
|
||||
|
||||
gpu_memory->ReadBlock(image.gpu_addr + task.current_offset,
|
||||
task.staging_buffer.mapped_span.data() + task.current_offset,
|
||||
copy_amount);
|
||||
task.current_offset += copy_amount;
|
||||
}
|
||||
|
||||
const bool is_final_batch = task.current_offset >= task.total_size;
|
||||
const size_t bytes_ready = task.current_offset - task.last_submitted_offset;
|
||||
const u32 complete_slices = static_cast<u32>(bytes_ready / task.bytes_per_slice);
|
||||
|
||||
if (complete_slices >= swizzle_slices_per_batch || (is_final_batch && complete_slices > 0)) {
|
||||
const u32 z_start = static_cast<u32>(task.last_submitted_offset / task.bytes_per_slice);
|
||||
const u32 slices_to_process = (std::min)(complete_slices, swizzle_slices_per_batch);
|
||||
const u32 z_count = (std::min)(slices_to_process, image.info.size.depth - z_start);
|
||||
|
||||
if (z_count > 0) {
|
||||
const auto uploads = FullUploadSwizzles(task.info);
|
||||
runtime.AccelerateImageUpload(image, task.staging_buffer, FixSmallVectorADL(uploads), z_start, z_count);
|
||||
task.last_submitted_offset += (static_cast<size_t>(z_count) * task.bytes_per_slice);
|
||||
}
|
||||
}
|
||||
|
||||
// Check if complete
|
||||
const u32 slices_submitted = static_cast<u32>(task.last_submitted_offset / task.bytes_per_slice);
|
||||
const bool all_slices_submitted = slices_submitted >= image.info.size.depth;
|
||||
|
||||
if (is_final_batch && all_slices_submitted) {
|
||||
runtime.FreeDeferredStagingBuffer(task.staging_buffer);
|
||||
image.flags &= ~ImageFlagBits::IsDecoding;
|
||||
unswizzle_queue.pop_front();
|
||||
|
||||
// Wait 4 frames to process the next entry
|
||||
current_unswizzle_frame = 4u;
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
bool TextureCache<P>::ScaleUp(Image& image) {
|
||||
const bool has_copy = image.HasScaled();
|
||||
@@ -1637,8 +1496,6 @@ ImageId TextureCache<P>::JoinImages(const ImageInfo& info, GPUVAddr gpu_addr, DA
|
||||
const ImageId new_image_id = slot_images.insert(runtime, new_info, gpu_addr, cpu_addr);
|
||||
Image& new_image = slot_images[new_image_id];
|
||||
|
||||
new_image.allocation_tick = frame_tick;
|
||||
|
||||
if (!gpu_memory->IsContinuousRange(new_image.gpu_addr, new_image.guest_size_bytes) &&
|
||||
new_info.is_sparse) {
|
||||
new_image.flags |= ImageFlagBits::Sparse;
|
||||
|
||||
@@ -131,17 +131,6 @@ class TextureCache : public VideoCommon::ChannelSetupCaches<TextureCacheChannelI
|
||||
using AsyncBuffer = typename P::AsyncBuffer;
|
||||
using BufferType = typename P::BufferType;
|
||||
|
||||
struct PendingUnswizzle {
|
||||
ImageId image_id;
|
||||
VideoCommon::ImageInfo info;
|
||||
size_t current_offset = 0;
|
||||
size_t total_size = 0;
|
||||
AsyncBuffer staging_buffer;
|
||||
size_t last_submitted_offset = 0;
|
||||
size_t bytes_per_slice;
|
||||
bool initialized = false;
|
||||
};
|
||||
|
||||
struct BlitImages {
|
||||
ImageId dst_id;
|
||||
ImageId src_id;
|
||||
@@ -422,9 +411,6 @@ private:
|
||||
void QueueAsyncDecode(Image& image, ImageId image_id);
|
||||
void TickAsyncDecode();
|
||||
|
||||
void QueueAsyncUnswizzle(Image& image, ImageId image_id);
|
||||
void TickAsyncUnswizzle();
|
||||
|
||||
Runtime& runtime;
|
||||
|
||||
Tegra::MaxwellDeviceMemoryManager& device_memory;
|
||||
@@ -454,9 +440,6 @@ private:
|
||||
u64 minimum_memory;
|
||||
u64 expected_memory;
|
||||
u64 critical_memory;
|
||||
size_t gpu_unswizzle_maxsize = 0;
|
||||
size_t swizzle_chunk_size = 0;
|
||||
u32 swizzle_slices_per_batch = 0;
|
||||
|
||||
struct BufferDownload {
|
||||
GPUVAddr address;
|
||||
@@ -512,9 +495,6 @@ private:
|
||||
Common::ThreadPlacement::Efficiency};
|
||||
std::vector<std::unique_ptr<AsyncDecodeContext>> async_decodes;
|
||||
|
||||
std::deque<PendingUnswizzle> unswizzle_queue;
|
||||
u8 current_unswizzle_frame;
|
||||
|
||||
// Join caching
|
||||
boost::container::small_vector<ImageId, 4> join_overlap_ids;
|
||||
::Common::unordered_set<ImageId> join_overlaps_found;
|
||||
|
||||
Reference in New Issue
Block a user