mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-10-07 06:18:23 +00:00
Revert some previous changes partially + maintenance5 changes
This commit is contained in:
@@ -5,7 +5,9 @@
|
||||
|
||||
#include <atomic>
|
||||
#include <limits>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <vector>
|
||||
|
||||
#include <boost/container/small_vector.hpp>
|
||||
|
||||
@@ -26,22 +28,23 @@ using VirtualSegments = boost::container::small_vector<VirtualSegment, 8>;
|
||||
class VirtualRangeCache {
|
||||
public:
|
||||
static constexpr size_t MAX_ENTRIES = 8192;
|
||||
static constexpr size_t MAX_DEFERRED = 4096;
|
||||
|
||||
const VirtualSegments* Query(Tegra::MemoryManager& memory, GPUVAddr gpu_addr, u32 size) {
|
||||
const u64 current = generation.load(std::memory_order_acquire);
|
||||
if (has_deferred.load(std::memory_order_acquire)) {
|
||||
ApplyDeferred();
|
||||
}
|
||||
if (entries.size() > MAX_ENTRIES) {
|
||||
entries.clear();
|
||||
}
|
||||
const size_t as_id = memory.GetID();
|
||||
const u64 key = MakeKey(as_id, gpu_addr);
|
||||
const auto it = entries.find(key);
|
||||
if (it != entries.end() && it->second.generation == current &&
|
||||
it->second.as_id == as_id && it->second.gpu_addr == gpu_addr &&
|
||||
it->second.size == size) {
|
||||
if (it != entries.end() && it->second.as_id == as_id &&
|
||||
it->second.gpu_addr == gpu_addr && it->second.size == size) {
|
||||
return &it->second.segments;
|
||||
}
|
||||
Entry entry{};
|
||||
entry.generation = current;
|
||||
entry.as_id = as_id;
|
||||
entry.gpu_addr = gpu_addr;
|
||||
entry.size = size;
|
||||
@@ -76,27 +79,96 @@ public:
|
||||
return &result.first->second.segments;
|
||||
}
|
||||
|
||||
void Unmap(size_t, GPUVAddr, u64 size) {
|
||||
if (size != 0) {
|
||||
generation.fetch_add(1, std::memory_order_release);
|
||||
void Unmap(size_t as_id, GPUVAddr gpu_addr, u64 size) {
|
||||
if (size == 0) {
|
||||
return;
|
||||
}
|
||||
{
|
||||
std::scoped_lock lock{deferred_mutex};
|
||||
if (!deferred.empty()) {
|
||||
DeferredUnmap& last = deferred.back();
|
||||
if (last.as_id == as_id && last.gpu_addr + last.size == gpu_addr) {
|
||||
last.size += size;
|
||||
has_deferred.store(true, std::memory_order_release);
|
||||
return;
|
||||
}
|
||||
}
|
||||
if (deferred.size() >= MAX_DEFERRED) {
|
||||
deferred.clear();
|
||||
deferred_overflow = true;
|
||||
} else {
|
||||
deferred.push_back(DeferredUnmap{
|
||||
.as_id = as_id,
|
||||
.gpu_addr = gpu_addr,
|
||||
.size = size,
|
||||
});
|
||||
}
|
||||
}
|
||||
has_deferred.store(true, std::memory_order_release);
|
||||
}
|
||||
|
||||
private:
|
||||
struct Entry {
|
||||
VirtualSegments segments;
|
||||
u64 generation{};
|
||||
size_t as_id{};
|
||||
GPUVAddr gpu_addr{};
|
||||
u32 size{};
|
||||
};
|
||||
|
||||
struct DeferredUnmap {
|
||||
size_t as_id;
|
||||
GPUVAddr gpu_addr;
|
||||
u64 size;
|
||||
};
|
||||
|
||||
static u64 MakeKey(size_t as_id, GPUVAddr gpu_addr) {
|
||||
return (static_cast<u64>(as_id) << 48) ^ gpu_addr;
|
||||
}
|
||||
|
||||
void ApplyDeferred() {
|
||||
bool overflow = false;
|
||||
{
|
||||
std::scoped_lock lock{deferred_mutex};
|
||||
has_deferred.store(false, std::memory_order_release);
|
||||
pending.clear();
|
||||
pending.swap(deferred);
|
||||
overflow = deferred_overflow;
|
||||
deferred_overflow = false;
|
||||
}
|
||||
if (overflow) {
|
||||
entries.clear();
|
||||
return;
|
||||
}
|
||||
if (pending.empty() || entries.empty()) {
|
||||
return;
|
||||
}
|
||||
for (auto it = entries.begin(); it != entries.end();) {
|
||||
const Entry& entry = it->second;
|
||||
const GPUVAddr entry_end = entry.gpu_addr + entry.size;
|
||||
bool overlaps = false;
|
||||
for (const DeferredUnmap& unmap : pending) {
|
||||
if (unmap.as_id != entry.as_id) {
|
||||
continue;
|
||||
}
|
||||
if (entry.gpu_addr < unmap.gpu_addr + unmap.size && unmap.gpu_addr < entry_end) {
|
||||
overlaps = true;
|
||||
break;
|
||||
}
|
||||
}
|
||||
if (overlaps) {
|
||||
it = entries.erase(it);
|
||||
} else {
|
||||
++it;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
::Common::unordered_map<u64, Entry> entries;
|
||||
std::atomic<u64> generation{};
|
||||
std::vector<DeferredUnmap> deferred;
|
||||
std::vector<DeferredUnmap> pending;
|
||||
std::mutex deferred_mutex;
|
||||
std::atomic<bool> has_deferred{false};
|
||||
bool deferred_overflow{};
|
||||
};
|
||||
|
||||
} // namespace VideoCommon
|
||||
|
||||
@@ -62,10 +62,6 @@ bool IsTopologySafe(Maxwell3D::Regs::PrimitiveTopology topology) {
|
||||
}
|
||||
}
|
||||
|
||||
bool IsTopologySafeIndexedIndirect(Maxwell3D::Regs::PrimitiveTopology topology) {
|
||||
return IsTopologySafe(topology) || topology == Maxwell3D::Regs::PrimitiveTopology::Quads;
|
||||
}
|
||||
|
||||
} // Anonymous namespace
|
||||
|
||||
void HLE_DrawArraysIndirect::Execute(Core::System& system, Engines::Maxwell3D& maxwell3d, std::span<const u32> parameters, [[maybe_unused]] u32 method) {
|
||||
@@ -129,7 +125,7 @@ void HLE_DrawArraysIndirect::Fallback(Core::System& system, Engines::Maxwell3D&
|
||||
|
||||
void HLE_DrawIndexedIndirect::Execute(Core::System& system, Engines::Maxwell3D& maxwell3d, std::span<const u32> parameters, [[maybe_unused]] u32 method) {
|
||||
auto topology = static_cast<Maxwell3D::Regs::PrimitiveTopology>(parameters[0]);
|
||||
if (!maxwell3d.AnyParametersDirty() || !IsTopologySafeIndexedIndirect(topology)) {
|
||||
if (!maxwell3d.AnyParametersDirty() || !IsTopologySafe(topology)) {
|
||||
Fallback(system, maxwell3d, parameters);
|
||||
return;
|
||||
}
|
||||
@@ -202,7 +198,7 @@ void HLE_MultiLayerClear::Execute(Core::System& system, Engines::Maxwell3D& maxw
|
||||
}
|
||||
void HLE_MultiDrawIndexedIndirectCount::Execute(Core::System& system, Engines::Maxwell3D& maxwell3d, std::span<const u32> parameters, [[maybe_unused]] u32 method) {
|
||||
const auto topology = Maxwell3D::Regs::PrimitiveTopology(parameters[2]);
|
||||
if (IsTopologySafeIndexedIndirect(topology)) {
|
||||
if (IsTopologySafe(topology)) {
|
||||
const u32 start_indirect = parameters[0];
|
||||
const u32 end_indirect = parameters[1];
|
||||
if (start_indirect >= end_indirect) {
|
||||
|
||||
@@ -578,29 +578,40 @@ bool BufferCacheRuntime::BindMultiRangeStorageBuffer(u64 key, bool is_written) {
|
||||
|
||||
void BufferCacheRuntime::BindIndexBuffer(PrimitiveTopology topology, IndexFormat index_format,
|
||||
u32 base_vertex, u32 num_indices, VkBuffer buffer,
|
||||
u32 offset, [[maybe_unused]] u32 size) {
|
||||
u32 offset, u32 size) {
|
||||
VkIndexType vk_index_type = MaxwellToVK::IndexFormat(index_format);
|
||||
VkDeviceSize vk_offset = offset;
|
||||
VkDeviceSize vk_size = size;
|
||||
VkBuffer vk_buffer = buffer;
|
||||
if (topology == PrimitiveTopology::Quads || topology == PrimitiveTopology::QuadStrip) {
|
||||
vk_index_type = VK_INDEX_TYPE_UINT32;
|
||||
vk_size = VK_WHOLE_SIZE;
|
||||
std::tie(vk_buffer, vk_offset) =
|
||||
quad_index_pass.Assemble(index_format, num_indices, base_vertex, buffer, offset,
|
||||
topology == PrimitiveTopology::QuadStrip);
|
||||
} else if (vk_index_type == VK_INDEX_TYPE_UINT8_EXT && !device.IsExtIndexTypeUint8Supported()) {
|
||||
vk_index_type = VK_INDEX_TYPE_UINT16;
|
||||
if (uint8_pass) {
|
||||
vk_size = VK_WHOLE_SIZE;
|
||||
std::tie(vk_buffer, vk_offset) = uint8_pass->Assemble(num_indices, buffer, offset);
|
||||
} else if (device.GetDriverID() == VK_DRIVER_ID_QUALCOMM_PROPRIETARY) {
|
||||
ReserveNullBuffer();
|
||||
vk_buffer = *null_buffer;
|
||||
vk_offset = 0;
|
||||
vk_size = VK_WHOLE_SIZE;
|
||||
}
|
||||
}
|
||||
if (vk_buffer == VK_NULL_HANDLE) {
|
||||
// Vulkan doesn't support null index buffers. Replace it with our own null buffer.
|
||||
ReserveNullBuffer();
|
||||
vk_buffer = *null_buffer;
|
||||
vk_size = VK_WHOLE_SIZE;
|
||||
}
|
||||
if (device.IsKhrMaintenance5Supported()) {
|
||||
scheduler.Record([vk_buffer, vk_offset, vk_size, vk_index_type](vk::CommandBuffer cmdbuf) {
|
||||
cmdbuf.BindIndexBuffer2KHR(vk_buffer, vk_offset, vk_size, vk_index_type);
|
||||
});
|
||||
return;
|
||||
}
|
||||
scheduler.Record([vk_buffer, vk_offset, vk_index_type](vk::CommandBuffer cmdbuf) {
|
||||
cmdbuf.BindIndexBuffer(vk_buffer, vk_offset, vk_index_type);
|
||||
|
||||
@@ -342,7 +342,17 @@ std::pair<VkBuffer, VkDeviceSize> QuadIndexedPass::Assemble(
|
||||
return 2;
|
||||
}();
|
||||
const u32 input_size = num_vertices << index_shift;
|
||||
const u32 num_tri_vertices = (is_strip ? (num_vertices - 2) / 2 : num_vertices / 4) * 6;
|
||||
u32 quads = num_vertices / 4;
|
||||
if (is_strip) {
|
||||
quads = 0;
|
||||
if (num_vertices >= 2) {
|
||||
quads = (num_vertices - 2) / 2;
|
||||
}
|
||||
}
|
||||
if (quads == 0) {
|
||||
quads = 1;
|
||||
}
|
||||
const u32 num_tri_vertices = quads * 6;
|
||||
|
||||
const std::size_t staging_size = num_tri_vertices * sizeof(u32);
|
||||
const auto staging = staging_buffer_pool.Request(staging_size, MemoryUsage::DeviceLocal);
|
||||
|
||||
@@ -212,8 +212,6 @@ RasterizerVulkan::RasterizerVulkan(Core::Frontend::EmuWindow& emu_window_, Tegra
|
||||
compute_pass_descriptor_queue(device, UpdateDescriptorQueue::COMPUTE_FRAME_PAYLOAD_SIZE),
|
||||
descriptor_buffer_ring(device, memory_allocator),
|
||||
blit_image(device, scheduler, state_tracker, descriptor_pool), render_pass_cache(device),
|
||||
indirect_quads_pass(device, scheduler, descriptor_pool, staging_pool,
|
||||
compute_pass_descriptor_queue),
|
||||
texture_cache_runtime{
|
||||
device, scheduler, memory_allocator, staging_pool,
|
||||
blit_image, render_pass_cache, descriptor_pool, compute_pass_descriptor_queue},
|
||||
@@ -319,15 +317,6 @@ void RasterizerVulkan::DrawIndirect() {
|
||||
VkBuffer command_buffer = buffer->Handle();
|
||||
VkDeviceSize command_offset = offset;
|
||||
u32 command_stride = static_cast<u32>(params.stride);
|
||||
if (params.is_indexed &&
|
||||
maxwell3d->draw_manager.draw_state.topology == Maxwell::PrimitiveTopology::Quads) {
|
||||
const auto patched = indirect_quads_pass.Assemble(
|
||||
static_cast<u32>(params.max_draw_counts), command_stride, command_buffer,
|
||||
static_cast<u32>(offset));
|
||||
command_buffer = patched.first;
|
||||
command_offset = patched.second;
|
||||
command_stride = IndirectQuadsPass::COMMAND_WORDS * static_cast<u32>(sizeof(u32));
|
||||
}
|
||||
if (params.is_byte_count) {
|
||||
scheduler.Record([buffer_obj = buffer->Handle(), offset,
|
||||
stride = params.stride](vk::CommandBuffer cmdbuf) {
|
||||
|
||||
@@ -17,7 +17,6 @@
|
||||
#include "video_core/rasterizer_interface.h"
|
||||
#include "video_core/renderer_vulkan/blit_image.h"
|
||||
#include "video_core/renderer_vulkan/vk_buffer_cache.h"
|
||||
#include "video_core/renderer_vulkan/vk_compute_pass.h"
|
||||
#include "video_core/renderer_vulkan/vk_descriptor_buffer.h"
|
||||
#include "video_core/renderer_vulkan/vk_descriptor_pool.h"
|
||||
#include "video_core/renderer_vulkan/vk_fence_manager.h"
|
||||
@@ -212,7 +211,6 @@ private:
|
||||
DescriptorBufferRing descriptor_buffer_ring;
|
||||
BlitImageHelper blit_image;
|
||||
RenderPassCache render_pass_cache;
|
||||
IndirectQuadsPass indirect_quads_pass;
|
||||
|
||||
TextureCacheRuntime texture_cache_runtime;
|
||||
TextureCache texture_cache;
|
||||
|
||||
@@ -96,6 +96,7 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
|
||||
X(vkCmdBeginDebugUtilsLabelEXT);
|
||||
X(vkCmdBindDescriptorSets);
|
||||
X(vkCmdBindIndexBuffer);
|
||||
X(vkCmdBindIndexBuffer2KHR);
|
||||
X(vkCmdBindPipeline);
|
||||
X(vkCmdBindTransformFeedbackBuffersEXT);
|
||||
X(vkCmdBindVertexBuffers);
|
||||
|
||||
@@ -211,6 +211,7 @@ struct DeviceDispatch : InstanceDispatch {
|
||||
PFN_vkCmdBeginTransformFeedbackEXT vkCmdBeginTransformFeedbackEXT{};
|
||||
PFN_vkCmdBindDescriptorSets vkCmdBindDescriptorSets{};
|
||||
PFN_vkCmdBindIndexBuffer vkCmdBindIndexBuffer{};
|
||||
PFN_vkCmdBindIndexBuffer2KHR vkCmdBindIndexBuffer2KHR{};
|
||||
PFN_vkCmdBindPipeline vkCmdBindPipeline{};
|
||||
PFN_vkCmdBindTransformFeedbackBuffersEXT vkCmdBindTransformFeedbackBuffersEXT{};
|
||||
PFN_vkCmdBindVertexBuffers vkCmdBindVertexBuffers{};
|
||||
@@ -1280,6 +1281,11 @@ public:
|
||||
dld->vkCmdBindIndexBuffer(handle, buffer, offset, index_type);
|
||||
}
|
||||
|
||||
void BindIndexBuffer2KHR(VkBuffer buffer, VkDeviceSize offset, VkDeviceSize size,
|
||||
VkIndexType index_type) const noexcept {
|
||||
dld->vkCmdBindIndexBuffer2KHR(handle, buffer, offset, size, index_type);
|
||||
}
|
||||
|
||||
void BindVertexBuffers(u32 first, u32 count, const VkBuffer* buffers,
|
||||
const VkDeviceSize* offsets) const noexcept {
|
||||
dld->vkCmdBindVertexBuffers(handle, first, count, buffers, offsets);
|
||||
|
||||
Reference in New Issue
Block a user