mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-09-15 07:47:44 +00:00
another try to add compute shaders to the uma path
This commit is contained in:
@@ -78,6 +78,11 @@ void BufferCache<P>::TickFrame() {
|
||||
return;
|
||||
}
|
||||
runtime.TickFrame(slot_buffers);
|
||||
if constexpr (USE_UNIFIED_MEMORY) {
|
||||
if (!unified_written_ranges.Empty() && runtime.KnownGpuTick() >= unified_write_tick) {
|
||||
unified_written_ranges.Clear();
|
||||
}
|
||||
}
|
||||
|
||||
// Calculate hits and shots and move hit bits to the right
|
||||
const u32 hits = std::reduce(channel_state->uniform_cache_hits.begin(),
|
||||
@@ -565,7 +570,8 @@ void BufferCache<P>::FlushCachedWrites() {
|
||||
|
||||
template <class P>
|
||||
bool BufferCache<P>::HasUncommittedFlushes() const noexcept {
|
||||
return !uncommitted_gpu_modified_ranges.Empty() || !committed_gpu_modified_ranges.empty();
|
||||
return !uncommitted_gpu_modified_ranges.Empty() || !committed_gpu_modified_ranges.empty() ||
|
||||
uncommitted_unified_writes;
|
||||
}
|
||||
|
||||
template <class P>
|
||||
@@ -582,15 +588,18 @@ bool BufferCache<P>::ShouldWaitAsyncFlushes() const noexcept {
|
||||
return false;
|
||||
}
|
||||
return async_buffers.front().has_value() ||
|
||||
!pending_downloads.front().unified_copies.empty();
|
||||
!pending_downloads.front().unified_copies.empty() ||
|
||||
pending_downloads.front().unified_writes;
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::CommitAsyncFlushesHigh() {
|
||||
AccumulateFlushes();
|
||||
const bool unified_writes = uncommitted_unified_writes;
|
||||
uncommitted_unified_writes = false;
|
||||
|
||||
if (committed_gpu_modified_ranges.empty()) {
|
||||
pending_downloads.emplace_back();
|
||||
pending_downloads.emplace_back(AsyncDownloadBatch{.unified_writes = unified_writes});
|
||||
async_buffers.emplace_back(std::optional<Async_Buffer>{});
|
||||
return;
|
||||
}
|
||||
@@ -650,7 +659,7 @@ void BufferCache<P>::CommitAsyncFlushesHigh() {
|
||||
}
|
||||
committed_gpu_modified_ranges.clear();
|
||||
if (downloads.empty()) {
|
||||
pending_downloads.emplace_back();
|
||||
pending_downloads.emplace_back(AsyncDownloadBatch{.unified_writes = unified_writes});
|
||||
async_buffers.emplace_back(std::optional<Async_Buffer>{});
|
||||
return;
|
||||
}
|
||||
@@ -726,6 +735,7 @@ void BufferCache<P>::CommitAsyncFlushesHigh() {
|
||||
}
|
||||
}
|
||||
runtime.PostCopyBarrier();
|
||||
batch.unified_writes = unified_writes;
|
||||
pending_downloads.emplace_back(std::move(batch));
|
||||
async_buffers.emplace_back(std::move(download_staging));
|
||||
}
|
||||
@@ -1030,6 +1040,7 @@ void BufferCache<P>::BindHostGraphicsUniformBuffer(size_t stage, u32 index, u32
|
||||
|| (has_host_buffer && size <= channel_state->uniform_buffer_skip_cache_size
|
||||
&& !memory_tracker.IsRegionGpuModified(device_addr, size));
|
||||
if (use_fast_buffer) {
|
||||
WaitForUnifiedWrites(device_addr, size);
|
||||
if constexpr (IS_OPENGL) {
|
||||
if (runtime.HasFastBufferSubData()) {
|
||||
// Fast path for Nvidia
|
||||
@@ -1154,6 +1165,47 @@ bool BufferCache<P>::BindMultiRangeStorage(const Binding& binding, bool is_writt
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::WaitForUnifiedWrites([[maybe_unused]] DAddr device_addr,
|
||||
[[maybe_unused]] u64 size) {
|
||||
if constexpr (USE_UNIFIED_MEMORY) {
|
||||
if (unified_written_ranges.Empty()) {
|
||||
return;
|
||||
}
|
||||
bool overlaps = false;
|
||||
unified_written_ranges.ForEachInRange(device_addr, size,
|
||||
[&overlaps](DAddr, DAddr) { overlaps = true; });
|
||||
if (!overlaps) {
|
||||
return;
|
||||
}
|
||||
runtime.Wait(unified_write_tick);
|
||||
unified_written_ranges.Clear();
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
bool BufferCache<P>::BindUnifiedStorage([[maybe_unused]] const Binding& binding,
|
||||
[[maybe_unused]] bool is_written) {
|
||||
if constexpr (USE_UNIFIED_MEMORY) {
|
||||
const auto window = TryResolveUnifiedRange(binding.device_addr, binding.size);
|
||||
if (!window || !runtime.IsUnifiedStorageRange(binding.size, window->offset)) {
|
||||
return false;
|
||||
}
|
||||
if (is_written) {
|
||||
memory_tracker.MarkRegionAsCpuModified(binding.device_addr, binding.size);
|
||||
unified_written_ranges.Add(binding.device_addr, binding.size);
|
||||
unified_write_tick = runtime.CurrentTick();
|
||||
uncommitted_unified_writes = true;
|
||||
}
|
||||
runtime.BindStorageBuffer(runtime.UnifiedWindowBuffer(window->window),
|
||||
runtime.UnifiedWindowAddress(window->window),
|
||||
static_cast<u32>(window->offset), binding.size, is_written);
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) {
|
||||
u32 binding_index = 0;
|
||||
@@ -1165,20 +1217,10 @@ void BufferCache<P>::BindHostGraphicsStorageBuffers(size_t stage) {
|
||||
}
|
||||
Buffer& buffer = slot_buffers[binding.buffer_id];
|
||||
TouchBuffer(buffer, binding.buffer_id);
|
||||
const u32 size = binding.size;
|
||||
|
||||
if constexpr (USE_UNIFIED_MEMORY) {
|
||||
const auto window = TryResolveUnifiedRange(binding.device_addr, size);
|
||||
if (window && runtime.IsUnifiedStorageRange(size, window->offset)) {
|
||||
if (is_written) {
|
||||
memory_tracker.MarkRegionAsCpuModified(binding.device_addr, size);
|
||||
}
|
||||
runtime.BindStorageBuffer(runtime.UnifiedWindowBuffer(window->window),
|
||||
runtime.UnifiedWindowAddress(window->window),
|
||||
static_cast<u32>(window->offset), size, is_written);
|
||||
return;
|
||||
}
|
||||
if (BindUnifiedStorage(binding, is_written)) {
|
||||
return;
|
||||
}
|
||||
const u32 size = binding.size;
|
||||
|
||||
SynchronizeBuffer(buffer, binding.device_addr, size);
|
||||
|
||||
@@ -1288,6 +1330,7 @@ void BufferCache<P>::BindHostComputeUniformBuffers() {
|
||||
}();
|
||||
if constexpr (!IS_OPENGL) {
|
||||
if (needs_alignment_stream) {
|
||||
WaitForUnifiedWrites(binding.device_addr, size);
|
||||
const std::span<u8> span =
|
||||
runtime.BindMappedUniformBuffer(0, binding_index, size);
|
||||
device_memory.ReadBlockUnsafe(binding.device_addr, span.data(), size);
|
||||
@@ -1319,6 +1362,9 @@ void BufferCache<P>::BindHostComputeStorageBuffers() {
|
||||
}
|
||||
Buffer& buffer = slot_buffers[binding.buffer_id];
|
||||
TouchBuffer(buffer, binding.buffer_id);
|
||||
if (BindUnifiedStorage(binding, is_written)) {
|
||||
return;
|
||||
}
|
||||
const u32 size = binding.size;
|
||||
SynchronizeBuffer(buffer, binding.device_addr, size);
|
||||
|
||||
@@ -1847,6 +1893,7 @@ bool BufferCache<P>::SynchronizeBuffer(Buffer& buffer, DAddr device_addr, u32 si
|
||||
if (total_size_bytes == 0) {
|
||||
return true;
|
||||
}
|
||||
WaitForUnifiedWrites(device_addr, size);
|
||||
const std::span<BufferCopy> copies_span(upload_copies.data(), upload_copies.size());
|
||||
UploadMemory(buffer, total_size_bytes, largest_copy, copies_span);
|
||||
any_buffer_uploaded = true;
|
||||
|
||||
@@ -230,6 +230,8 @@ public:
|
||||
bool BindMultiRangeStorage(const Binding& binding, bool is_written,
|
||||
std::span<const MultiRangeSegment> pool);
|
||||
|
||||
bool BindUnifiedStorage(const Binding& binding, bool is_written);
|
||||
|
||||
void ResolveMultiRangeStorage(Binding& binding, bool is_written,
|
||||
std::vector<MultiRangeSegment>& pool);
|
||||
|
||||
@@ -473,6 +475,8 @@ private:
|
||||
|
||||
std::optional<UnifiedWindowRange> TryResolveUnifiedRange(DAddr device_addr, u64 size);
|
||||
|
||||
void WaitForUnifiedWrites(DAddr device_addr, u64 size);
|
||||
|
||||
using UnifiedWindowGroups =
|
||||
boost::container::small_vector<boost::container::small_vector<BufferCopy, 16>, 4>;
|
||||
|
||||
@@ -533,11 +537,15 @@ private:
|
||||
Common::RangeSet<DAddr> uncommitted_gpu_modified_ranges;
|
||||
Common::RangeSet<DAddr> gpu_modified_ranges;
|
||||
std::deque<Common::RangeSet<DAddr>> committed_gpu_modified_ranges;
|
||||
Common::RangeSet<DAddr> unified_written_ranges;
|
||||
u64 unified_write_tick = 0;
|
||||
bool uncommitted_unified_writes = false;
|
||||
|
||||
// Async Buffers
|
||||
struct AsyncDownloadBatch {
|
||||
boost::container::small_vector<BufferCopy, 4> staging_copies;
|
||||
boost::container::small_vector<BufferCopy, 4> unified_copies;
|
||||
bool unified_writes = false;
|
||||
};
|
||||
|
||||
Common::OverlapRangeSet<DAddr> async_downloads;
|
||||
|
||||
@@ -16,6 +16,7 @@ set(GLSL_INCLUDES
|
||||
set(SHADER_FILES
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/astc_decoder.comp
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/blit_color_float.frag
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/block_linear_swizzle_2d_buffer.comp
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/block_linear_unswizzle_2d.comp
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/block_linear_unswizzle_2d_buffer.comp
|
||||
${CMAKE_CURRENT_SOURCE_DIR}/blit_color_msaa.frag
|
||||
|
||||
@@ -0,0 +1,83 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#version 430
|
||||
|
||||
#define BINDING_INPUT_BUFFER 0
|
||||
#define BINDING_OUTPUT_BUFFER 1
|
||||
|
||||
layout(push_constant) uniform PushConstants {
|
||||
uvec3 dim;
|
||||
uint bytes_per_block_log2;
|
||||
|
||||
uvec3 origin;
|
||||
uint layer_stride;
|
||||
|
||||
uint block_size;
|
||||
uint x_shift;
|
||||
uint block_height;
|
||||
uint block_height_mask;
|
||||
} pc;
|
||||
|
||||
layout(binding = BINDING_INPUT_BUFFER, std430) readonly buffer InputBuffer {
|
||||
uint in_u32[];
|
||||
};
|
||||
|
||||
layout(binding = BINDING_OUTPUT_BUFFER, std430) buffer OutputBuffer {
|
||||
uint out_u32[];
|
||||
};
|
||||
|
||||
layout(local_size_x = 16, local_size_y = 8, local_size_z = 1) in;
|
||||
|
||||
const uint GOB_SIZE_X = 64;
|
||||
const uint GOB_SIZE_Y = 8;
|
||||
|
||||
const uint GOB_SIZE_X_SHIFT = 6;
|
||||
const uint GOB_SIZE_Y_SHIFT = 3;
|
||||
const uint GOB_SIZE_SHIFT = GOB_SIZE_X_SHIFT + GOB_SIZE_Y_SHIFT;
|
||||
|
||||
const uvec2 SWIZZLE_MASK = uvec2(GOB_SIZE_X - 1u, GOB_SIZE_Y - 1u);
|
||||
|
||||
uint SwizzleTable(uint pos) {
|
||||
const uint t[8] = uint[](
|
||||
0x12100200, 0x13110301, 0x16140604, 0x17150705,
|
||||
0x1a180a08, 0x1b190b09, 0x1e1c0e0c, 0x1f1d0f0d
|
||||
);
|
||||
const uint i = pos >> 4;
|
||||
const uint h = (t[i / 4] >> ((i % 4) * 8)) & 0xff;
|
||||
return (h << 4) | (pos & 0xf);
|
||||
}
|
||||
|
||||
uint SwizzleOffset(uvec2 pos) {
|
||||
pos = pos & SWIZZLE_MASK;
|
||||
return SwizzleTable(pos.y * 64u + pos.x);
|
||||
}
|
||||
|
||||
void main() {
|
||||
uvec3 coord = gl_GlobalInvocationID;
|
||||
if (coord.x >= pc.dim.x || coord.y >= pc.dim.y || coord.z >= pc.dim.z) {
|
||||
return;
|
||||
}
|
||||
|
||||
uvec3 pos = coord + pc.origin;
|
||||
pos.x <<= pc.bytes_per_block_log2;
|
||||
|
||||
uint swizzle = SwizzleOffset(pos.xy);
|
||||
uint block_y = pos.y >> GOB_SIZE_Y_SHIFT;
|
||||
|
||||
uint offset = 0u;
|
||||
offset += pos.z * pc.layer_stride;
|
||||
offset += (block_y >> pc.block_height) * pc.block_size;
|
||||
offset += (block_y & pc.block_height_mask) << GOB_SIZE_SHIFT;
|
||||
offset += (pos.x >> GOB_SIZE_X_SHIFT) << pc.x_shift;
|
||||
offset += swizzle;
|
||||
|
||||
uint words = 1u << (pc.bytes_per_block_log2 - 2u);
|
||||
uint linear_index = coord.x + coord.y * pc.dim.x + coord.z * pc.dim.x * pc.dim.y;
|
||||
uint in_idx = linear_index * words;
|
||||
uint out_idx = offset >> 2u;
|
||||
|
||||
for (uint word = 0u; word < words; ++word) {
|
||||
out_u32[out_idx + word] = in_u32[in_idx + word];
|
||||
}
|
||||
}
|
||||
@@ -24,6 +24,7 @@
|
||||
#include "video_core/host_shaders/resolve_conditional_render_comp_spv.h"
|
||||
#include "video_core/host_shaders/vulkan_quad_indexed_comp_spv.h"
|
||||
#include "video_core/host_shaders/vulkan_uint8_comp_spv.h"
|
||||
#include "video_core/host_shaders/block_linear_swizzle_2d_buffer_comp_spv.h"
|
||||
#include "video_core/host_shaders/block_linear_unswizzle_2d_buffer_comp_spv.h"
|
||||
#include "video_core/host_shaders/block_linear_unswizzle_3d_bcn_comp_spv.h"
|
||||
#include "video_core/host_shaders/block_linear_unswizzle_3d_buffer_comp_spv.h"
|
||||
@@ -35,6 +36,7 @@
|
||||
#include "video_core/renderer_vulkan/vk_update_descriptor.h"
|
||||
#include "video_core/texture_cache/accelerated_swizzle.h"
|
||||
#include "video_core/texture_cache/types.h"
|
||||
#include "video_core/texture_cache/util.h"
|
||||
#include "video_core/textures/decoders.h"
|
||||
#include "video_core/vulkan_common/vulkan_device.h"
|
||||
#include "video_core/vulkan_common/vulkan_wrapper.h"
|
||||
@@ -1156,6 +1158,141 @@ void BlockLinearUnswizzle2DPass::UnswizzleFrom(
|
||||
});
|
||||
}
|
||||
|
||||
BlockLinearSwizzle2DPass::BlockLinearSwizzle2DPass(
|
||||
const Device& device_, Scheduler& scheduler_, DescriptorPool& descriptor_pool_,
|
||||
StagingBufferPool& staging_buffer_pool_,
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue_)
|
||||
: ComputePass(device_, scheduler_, descriptor_pool_, BL2D_BINDINGS, BL2D_TEMPLATE,
|
||||
BL2D_BANK_INFO,
|
||||
COMPUTE_PUSH_CONSTANT_RANGE<sizeof(BlockLinearUnswizzle2DPushConstants)>,
|
||||
BLOCK_LINEAR_SWIZZLE_2D_BUFFER_COMP_SPV),
|
||||
scheduler{scheduler_}, staging_buffer_pool{staging_buffer_pool_},
|
||||
compute_pass_descriptor_queue{compute_pass_descriptor_queue_} {}
|
||||
|
||||
BlockLinearSwizzle2DPass::~BlockLinearSwizzle2DPass() = default;
|
||||
|
||||
void BlockLinearSwizzle2DPass::SwizzleInto(Image& image, VkBuffer dst_buffer,
|
||||
VkDeviceSize dst_offset, bool foreign_ownership) {
|
||||
const u32 layers = image.info.resources.layers;
|
||||
const VkDeviceSize guest_size = image.guest_size_bytes;
|
||||
const VkDeviceSize input_alignment =
|
||||
(std::max)(device.GetStorageBufferAlignment(), VkDeviceSize{16});
|
||||
auto copies = VideoCommon::FullDownloadCopies(image.info);
|
||||
const auto swizzles = VideoCommon::FullUploadSwizzles(image.info);
|
||||
VkDeviceSize total_size = 0;
|
||||
for (VideoCommon::BufferImageCopy& copy : copies) {
|
||||
total_size = Common::AlignUp(total_size, input_alignment);
|
||||
copy.buffer_offset = static_cast<size_t>(total_size);
|
||||
total_size += copy.buffer_size;
|
||||
}
|
||||
const StagingBufferRef scratch =
|
||||
staging_buffer_pool.Request(static_cast<size_t>(total_size), MemoryUsage::DeviceLocal);
|
||||
const VkBuffer scratch_buffer = scratch.buffer;
|
||||
const VkDeviceSize scratch_offset = scratch.offset;
|
||||
image.DownloadMemory(scratch_buffer, static_cast<size_t>(scratch_offset),
|
||||
std::span<const VideoCommon::BufferImageCopy>(copies.data(),
|
||||
copies.size()));
|
||||
|
||||
const u32 queue_family = device.GetGraphicsFamily();
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
scheduler.Record([scratch_buffer, scratch_offset, total_size, dst_buffer, dst_offset,
|
||||
guest_size, queue_family, foreign_ownership](vk::CommandBuffer cmdbuf) {
|
||||
const VkBufferMemoryBarrier scratch_barrier{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER,
|
||||
.pNext = nullptr,
|
||||
.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT,
|
||||
.dstAccessMask = VK_ACCESS_SHADER_READ_BIT,
|
||||
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
||||
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
|
||||
.buffer = scratch_buffer,
|
||||
.offset = scratch_offset,
|
||||
.size = total_size,
|
||||
};
|
||||
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_TRANSFER_BIT,
|
||||
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, 0, scratch_barrier);
|
||||
if (foreign_ownership) {
|
||||
const VkBufferMemoryBarrier acquire{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER,
|
||||
.pNext = nullptr,
|
||||
.srcAccessMask = 0,
|
||||
.dstAccessMask = VK_ACCESS_SHADER_WRITE_BIT,
|
||||
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_FOREIGN_EXT,
|
||||
.dstQueueFamilyIndex = queue_family,
|
||||
.buffer = dst_buffer,
|
||||
.offset = dst_offset,
|
||||
.size = guest_size,
|
||||
};
|
||||
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT,
|
||||
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, 0, acquire);
|
||||
}
|
||||
});
|
||||
|
||||
for (size_t level = 0; level < copies.size(); ++level) {
|
||||
const VideoCommon::SwizzleParameters& sw = swizzles[level];
|
||||
const VideoCommon::BufferImageCopy& copy = copies[level];
|
||||
const auto params =
|
||||
VideoCommon::Accelerated::MakeBlockLinearSwizzle2DParams(sw, image.info);
|
||||
|
||||
BlockLinearUnswizzle2DPushConstants pc{};
|
||||
pc.dim = {sw.num_tiles.width, sw.num_tiles.height, layers};
|
||||
pc.bytes_per_block_log2 = params.bytes_per_block_log2;
|
||||
pc.origin = params.origin;
|
||||
pc.layer_stride = params.layer_stride;
|
||||
pc.block_size = params.block_size;
|
||||
pc.x_shift = params.x_shift;
|
||||
pc.block_height = params.block_height;
|
||||
pc.block_height_mask = params.block_height_mask;
|
||||
|
||||
compute_pass_descriptor_queue.Acquire(scheduler, 2);
|
||||
compute_pass_descriptor_queue.AddBuffer(scratch_buffer, scratch_offset + copy.buffer_offset,
|
||||
copy.buffer_size);
|
||||
compute_pass_descriptor_queue.AddBuffer(dst_buffer, dst_offset + sw.buffer_offset,
|
||||
guest_size - sw.buffer_offset);
|
||||
const void* descriptor_data = compute_pass_descriptor_queue.UpdateData();
|
||||
const VkDescriptorSet set = descriptor_allocator.Commit();
|
||||
|
||||
const u32 gx = Common::DivCeil(sw.num_tiles.width, 16u);
|
||||
const u32 gy = Common::DivCeil(sw.num_tiles.height, 8u);
|
||||
|
||||
scheduler.Record(
|
||||
[this, set, descriptor_data, pc, gx, gy, layers](vk::CommandBuffer cmdbuf) {
|
||||
device.GetLogical().UpdateDescriptorSet(set, *descriptor_template,
|
||||
descriptor_data);
|
||||
cmdbuf.BindPipeline(VK_PIPELINE_BIND_POINT_COMPUTE, *pipeline);
|
||||
cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_COMPUTE, *layout, 0, set, {});
|
||||
cmdbuf.PushConstants(*layout, VK_SHADER_STAGE_COMPUTE_BIT, 0, sizeof(pc), &pc);
|
||||
cmdbuf.Dispatch(gx, gy, layers);
|
||||
});
|
||||
}
|
||||
|
||||
scheduler.Record([dst_buffer, dst_offset, guest_size, queue_family,
|
||||
foreign_ownership](vk::CommandBuffer cmdbuf) {
|
||||
if (foreign_ownership) {
|
||||
const VkBufferMemoryBarrier release{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER,
|
||||
.pNext = nullptr,
|
||||
.srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT,
|
||||
.dstAccessMask = 0,
|
||||
.srcQueueFamilyIndex = queue_family,
|
||||
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_FOREIGN_EXT,
|
||||
.buffer = dst_buffer,
|
||||
.offset = dst_offset,
|
||||
.size = guest_size,
|
||||
};
|
||||
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
|
||||
VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, release);
|
||||
}
|
||||
static constexpr VkMemoryBarrier HOST_BARRIER{
|
||||
.sType = VK_STRUCTURE_TYPE_MEMORY_BARRIER,
|
||||
.pNext = nullptr,
|
||||
.srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT,
|
||||
.dstAccessMask = VK_ACCESS_HOST_READ_BIT,
|
||||
};
|
||||
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, VK_PIPELINE_STAGE_HOST_BIT,
|
||||
0, HOST_BARRIER);
|
||||
});
|
||||
}
|
||||
|
||||
namespace {
|
||||
constexpr u32 BL3DB_BINDING_INPUT_BUFFER = 0;
|
||||
constexpr u32 BL3DB_BINDING_OUTPUT_BUFFER = 1;
|
||||
|
||||
@@ -187,6 +187,23 @@ private:
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue;
|
||||
};
|
||||
|
||||
class BlockLinearSwizzle2DPass final : public ComputePass {
|
||||
public:
|
||||
explicit BlockLinearSwizzle2DPass(const Device& device_, Scheduler& scheduler_,
|
||||
DescriptorPool& descriptor_pool_,
|
||||
StagingBufferPool& staging_buffer_pool_,
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue_);
|
||||
~BlockLinearSwizzle2DPass();
|
||||
|
||||
void SwizzleInto(Image& image, VkBuffer dst_buffer, VkDeviceSize dst_offset,
|
||||
bool foreign_ownership);
|
||||
|
||||
private:
|
||||
Scheduler& scheduler;
|
||||
StagingBufferPool& staging_buffer_pool;
|
||||
ComputePassDescriptorQueue& compute_pass_descriptor_queue;
|
||||
};
|
||||
|
||||
class BlockLinearUnswizzle3DBufferPass final : public ComputePass {
|
||||
public:
|
||||
explicit BlockLinearUnswizzle3DBufferPass(
|
||||
|
||||
@@ -972,6 +972,8 @@ TextureCacheRuntime::TextureCacheRuntime(const Device& device_, Scheduler& sched
|
||||
compute_pass_descriptor_queue);
|
||||
bl3db_unswizzle_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool,
|
||||
compute_pass_descriptor_queue);
|
||||
bl2d_swizzle_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool,
|
||||
compute_pass_descriptor_queue);
|
||||
}
|
||||
|
||||
void TextureCacheRuntime::Finish() {
|
||||
@@ -3209,20 +3211,26 @@ bool TextureCacheRuntime::CanUploadImageDirectly(const VideoCommon::ImageInfo& i
|
||||
BlockLinearUnswizzle2DPass::IsSupported(device, info);
|
||||
}
|
||||
|
||||
bool TextureCacheRuntime::UploadImageDirectly(
|
||||
Image& image, size_t window_index, u64 window_offset,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles) {
|
||||
VkBuffer TextureCacheRuntime::ResolveDirectWindow(size_t window_index, u64 window_offset,
|
||||
u64 size) const {
|
||||
if ((window_offset % device.GetStorageBufferAlignment()) != 0) {
|
||||
return false;
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
if (image.guest_size_bytes > device.GetMaxStorageBufferRange()) {
|
||||
return false;
|
||||
if (size > device.GetMaxStorageBufferRange()) {
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
const HostMemoryImport* const import = memory_allocator.GetHostMemoryImport();
|
||||
if (import == nullptr || window_index >= import->GetWindowCount()) {
|
||||
return false;
|
||||
return VK_NULL_HANDLE;
|
||||
}
|
||||
const VkBuffer window_buffer = import->GetWindowBuffer(window_index);
|
||||
return import->GetWindowBuffer(window_index);
|
||||
}
|
||||
|
||||
bool TextureCacheRuntime::UploadImageDirectly(
|
||||
Image& image, size_t window_index, u64 window_offset,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles) {
|
||||
const VkBuffer window_buffer =
|
||||
ResolveDirectWindow(window_index, window_offset, image.guest_size_bytes);
|
||||
if (window_buffer == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
@@ -3231,6 +3239,23 @@ bool TextureCacheRuntime::UploadImageDirectly(
|
||||
return true;
|
||||
}
|
||||
|
||||
bool TextureCacheRuntime::CanDownloadImageDirectly(const VideoCommon::ImageInfo& info) const {
|
||||
return bl2d_swizzle_pass.has_value() && BlockLinearUnswizzle2DPass::IsSupported(device, info);
|
||||
}
|
||||
|
||||
bool TextureCacheRuntime::DownloadImageDirectly(Image& image, size_t window_index,
|
||||
u64 window_offset) {
|
||||
const u64 size = (std::max)(image.guest_size_bytes, image.unswizzled_size_bytes);
|
||||
const VkBuffer window_buffer = ResolveDirectWindow(window_index, window_offset, size);
|
||||
if (window_buffer == VK_NULL_HANDLE) {
|
||||
return false;
|
||||
}
|
||||
const HostMemoryImport* const import = memory_allocator.GetHostMemoryImport();
|
||||
bl2d_swizzle_pass->SwizzleInto(image, window_buffer, static_cast<VkDeviceSize>(window_offset),
|
||||
import->NeedsForeignOwnershipTransfer());
|
||||
return true;
|
||||
}
|
||||
|
||||
u64 TextureCacheRuntime::CurrentTick() const noexcept {
|
||||
return scheduler.CurrentTick();
|
||||
}
|
||||
|
||||
@@ -114,6 +114,13 @@ public:
|
||||
bool UploadImageDirectly(Image& image, size_t window_index, u64 window_offset,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles);
|
||||
|
||||
[[nodiscard]] bool CanDownloadImageDirectly(const VideoCommon::ImageInfo& info) const;
|
||||
|
||||
bool DownloadImageDirectly(Image& image, size_t window_index, u64 window_offset);
|
||||
|
||||
[[nodiscard]] VkBuffer ResolveDirectWindow(size_t window_index, u64 window_offset,
|
||||
u64 size) const;
|
||||
|
||||
[[nodiscard]] u64 CurrentTick() const noexcept;
|
||||
|
||||
[[nodiscard]] bool IsDirectUploadRetired(u64 tick);
|
||||
@@ -178,6 +185,7 @@ public:
|
||||
std::optional<BlockLinearUnswizzle3DPass> bl3d_unswizzle_pass;
|
||||
std::optional<BlockLinearUnswizzle2DPass> bl2d_unswizzle_pass;
|
||||
std::optional<BlockLinearUnswizzle3DBufferPass> bl3db_unswizzle_pass;
|
||||
std::optional<BlockLinearSwizzle2DPass> bl2d_swizzle_pass;
|
||||
const Settings::ResolutionScalingInfo& resolution;
|
||||
std::array<std::vector<VkFormat>, VideoCore::Surface::MaxPixelFormat> view_formats;
|
||||
|
||||
|
||||
@@ -146,11 +146,15 @@ void TextureCache<P>::RunGarbageCollector() {
|
||||
return false;
|
||||
}
|
||||
--num_downloads;
|
||||
auto map = runtime.DownloadStagingBuffer(image.unswizzled_size_bytes);
|
||||
const auto copies = FixSmallVectorADL(FullDownloadCopies(image.info));
|
||||
image.DownloadMemory(map, copies);
|
||||
runtime.Finish();
|
||||
SwizzleImage(*gpu_memory, image.gpu_addr, image.info, copies, map.mapped_span, swizzle_data_buffer);
|
||||
if (TryDownloadToUnifiedMemory(image)) {
|
||||
runtime.Finish();
|
||||
} else {
|
||||
auto map = runtime.DownloadStagingBuffer(image.unswizzled_size_bytes);
|
||||
const auto copies = FixSmallVectorADL(FullDownloadCopies(image.info));
|
||||
image.DownloadMemory(map, copies);
|
||||
runtime.Finish();
|
||||
SwizzleImage(*gpu_memory, image.gpu_addr, image.info, copies, map.mapped_span, swizzle_data_buffer);
|
||||
}
|
||||
}
|
||||
if (True(image.flags & ImageFlagBits::Tracked)) {
|
||||
UntrackImage(image, image_id);
|
||||
@@ -642,15 +646,24 @@ void TextureCache<P>::DownloadMemory(DAddr cpu_addr, size_t size) {
|
||||
std::ranges::sort(images, [this](ImageId lhs, ImageId rhs) {
|
||||
return slot_images[lhs].modification_tick < slot_images[rhs].modification_tick;
|
||||
});
|
||||
bool pending_unified = false;
|
||||
for (const ImageId image_id : images) {
|
||||
Image& image = slot_images[image_id];
|
||||
if (TryDownloadToUnifiedMemory(image)) {
|
||||
pending_unified = true;
|
||||
continue;
|
||||
}
|
||||
auto map = runtime.DownloadStagingBuffer(image.unswizzled_size_bytes);
|
||||
const auto copies = FixSmallVectorADL(FullDownloadCopies(image.info));
|
||||
image.DownloadMemory(map, copies);
|
||||
runtime.Finish();
|
||||
pending_unified = false;
|
||||
SwizzleImage(*gpu_memory, image.gpu_addr, image.info, copies, map.mapped_span,
|
||||
swizzle_data_buffer);
|
||||
}
|
||||
if (pending_unified) {
|
||||
runtime.Finish();
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
@@ -897,6 +910,10 @@ void TextureCache<P>::CommitAsyncFlushes() {
|
||||
bool any_none_dma = false;
|
||||
for (PendingDownload& download_info : download_ids) {
|
||||
if (download_info.is_swizzle) {
|
||||
if (TryDownloadToUnifiedMemory(slot_images[download_info.object_id])) {
|
||||
download_info.is_unified = true;
|
||||
continue;
|
||||
}
|
||||
total_size_bytes +=
|
||||
Common::AlignUp(slot_images[download_info.object_id].unswizzled_size_bytes, 64);
|
||||
any_none_dma = true;
|
||||
@@ -907,7 +924,7 @@ void TextureCache<P>::CommitAsyncFlushes() {
|
||||
if (any_none_dma) {
|
||||
auto download_map = runtime.DownloadStagingBuffer(total_size_bytes, true);
|
||||
for (const PendingDownload& download_info : download_ids) {
|
||||
if (download_info.is_swizzle) {
|
||||
if (download_info.is_swizzle && !download_info.is_unified) {
|
||||
Image& image = slot_images[download_info.object_id];
|
||||
const auto copies = FixSmallVectorADL(FullDownloadCopies(image.info));
|
||||
image.DownloadMemory(download_map, copies);
|
||||
@@ -939,6 +956,9 @@ void TextureCache<P>::PopAsyncFlushes() {
|
||||
auto download_map = std::move(async_buffers.front());
|
||||
for (size_t i = download_ids.size(); i > 0; i--) {
|
||||
auto& download_info = download_ids[i - 1];
|
||||
if (download_info.is_unified) {
|
||||
continue;
|
||||
}
|
||||
auto& download_buffer = download_map[download_info.async_buffer_id];
|
||||
if (download_info.is_swizzle) {
|
||||
const ImageBase& image = slot_images[download_info.object_id];
|
||||
@@ -1172,39 +1192,53 @@ void TextureCache<P>::RefreshContents(Image& image, ImageId image_id) {
|
||||
}
|
||||
|
||||
template <class P>
|
||||
bool TextureCache<P>::TryUploadFromUnifiedMemory([[maybe_unused]] Image& image) {
|
||||
std::optional<std::pair<size_t, u64>> TextureCache<P>::ResolveUnifiedImageWindow(
|
||||
[[maybe_unused]] const ImageBase& image) {
|
||||
if constexpr (USE_UNIFIED_MEMORY) {
|
||||
if (image.direct_upload_blocked || image.guest_size_bytes == 0) {
|
||||
return false;
|
||||
}
|
||||
if (!runtime.IsUnifiedMemoryBindable() || !runtime.CanUploadImageDirectly(image.info)) {
|
||||
return false;
|
||||
if (image.guest_size_bytes == 0 || !runtime.IsUnifiedMemoryBindable()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
const u64 window_size = runtime.UnifiedMemoryWindowSize();
|
||||
if (window_size == 0) {
|
||||
return false;
|
||||
return std::nullopt;
|
||||
}
|
||||
const u8* const first = gpu_memory->GetSpan(image.gpu_addr, image.guest_size_bytes);
|
||||
if (first == nullptr) {
|
||||
return false;
|
||||
return std::nullopt;
|
||||
}
|
||||
const u64 phys_offset = static_cast<u64>(first - device_memory.GetPhysicalBase());
|
||||
const u64 unified_base = runtime.UnifiedMemoryBase();
|
||||
if (phys_offset < unified_base) {
|
||||
return false;
|
||||
return std::nullopt;
|
||||
}
|
||||
const u64 relative = phys_offset - unified_base;
|
||||
const u64 unified_size = runtime.UnifiedMemorySize();
|
||||
if (relative >= unified_size || unified_size - relative < image.guest_size_bytes) {
|
||||
return false;
|
||||
return std::nullopt;
|
||||
}
|
||||
const u64 local_offset = relative % window_size;
|
||||
if (window_size - local_offset < image.guest_size_bytes) {
|
||||
return std::nullopt;
|
||||
}
|
||||
return std::pair{static_cast<size_t>(relative / window_size), local_offset};
|
||||
} else {
|
||||
return std::nullopt;
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
bool TextureCache<P>::TryUploadFromUnifiedMemory([[maybe_unused]] Image& image) {
|
||||
if constexpr (USE_UNIFIED_MEMORY) {
|
||||
if (image.direct_upload_blocked || !runtime.CanUploadImageDirectly(image.info)) {
|
||||
return false;
|
||||
}
|
||||
const auto window = ResolveUnifiedImageWindow(image);
|
||||
if (!window) {
|
||||
return false;
|
||||
}
|
||||
const auto swizzles = FullUploadSwizzles(image.info);
|
||||
if (!runtime.UploadImageDirectly(image, static_cast<size_t>(relative / window_size),
|
||||
local_offset, FixSmallVectorADL(swizzles))) {
|
||||
if (!runtime.UploadImageDirectly(image, window->first, window->second,
|
||||
FixSmallVectorADL(swizzles))) {
|
||||
return false;
|
||||
}
|
||||
image.direct_upload_tick = runtime.CurrentTick();
|
||||
@@ -1214,6 +1248,26 @@ bool TextureCache<P>::TryUploadFromUnifiedMemory([[maybe_unused]] Image& image)
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
bool TextureCache<P>::TryDownloadToUnifiedMemory([[maybe_unused]] Image& image) {
|
||||
if constexpr (USE_UNIFIED_MEMORY) {
|
||||
if (!runtime.CanDownloadImageDirectly(image.info)) {
|
||||
return false;
|
||||
}
|
||||
if (image.info.resources.layers > 1 &&
|
||||
image.info.layer_stride != CalculateLayerStride(image.info)) {
|
||||
return false;
|
||||
}
|
||||
const auto window = ResolveUnifiedImageWindow(image);
|
||||
if (!window) {
|
||||
return false;
|
||||
}
|
||||
return runtime.DownloadImageDirectly(image, window->first, window->second);
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
template <typename StagingBuffer>
|
||||
void TextureCache<P>::UploadImageContents(Image& image, StagingBuffer& staging) {
|
||||
|
||||
@@ -310,8 +310,13 @@ private:
|
||||
|
||||
void RefreshContents(Image& image, ImageId image_id);
|
||||
|
||||
[[nodiscard]] std::optional<std::pair<size_t, u64>> ResolveUnifiedImageWindow(
|
||||
const ImageBase& image);
|
||||
|
||||
bool TryUploadFromUnifiedMemory(Image& image);
|
||||
|
||||
bool TryDownloadToUnifiedMemory(Image& image);
|
||||
|
||||
/// Upload data from guest to an image
|
||||
template <typename StagingBuffer>
|
||||
void UploadImageContents(Image& image, StagingBuffer& staging_buffer);
|
||||
@@ -471,6 +476,7 @@ private:
|
||||
bool is_swizzle;
|
||||
size_t async_buffer_id;
|
||||
Common::SlotId object_id;
|
||||
bool is_unified = false;
|
||||
};
|
||||
|
||||
Common::SlotVector<Image> slot_images;
|
||||
|
||||
Reference in New Issue
Block a user