Compare commits

...

32 Commits

Author SHA1 Message Date
CamilleLaVey 71cf0e0daa fix build 2026-02-14 03:41:57 -04:00
CamilleLaVey 59444d5c48 [vulkan] bump to vk 1.4 and some features to 1.4 tree 2026-02-14 03:28:00 -04:00
CamilleLaVey 3aa4fb867c [vulkan, tu] Returning timeline semaphores to turnip 2026-02-14 01:07:52 -04:00
lizzie ab1ad1dc4a [maxwell] fix divide by 0 (#3540)
Signed-off-by: lizzie <lizzie@eden-emu.dev>
Reviewed-on: https://git.eden-emu.dev/eden-emu/eden/pulls/3540
Co-authored-by: lizzie <lizzie@eden-emu.dev>
Co-committed-by: lizzie <lizzie@eden-emu.dev>
2026-02-14 05:12:01 +01:00
CamilleLaVey 0dc64e2a48 fix build. 2026-02-14 03:46:23 +01:00
CamilleLaVey 420b002448 Revert "[texture_cache, buffer_cache] Rebuild modified pages in texture cache" 2026-02-14 03:46:23 +01:00
CamilleLaVey 413cebca20 Revert "smol change" 2026-02-14 03:46:23 +01:00
CamilleLaVey a2d830e51b [vulkan] BindVertexBuffers2EXT reset per pipeline configuration 2026-02-14 03:46:23 +01:00
CamilleLaVey 47d6360038 smol change 2026-02-14 03:46:23 +01:00
CamilleLaVey e0daa0d83d [texture_cache, buffer_cache] Rebuild modified pages in texture cache 2026-02-14 03:46:23 +01:00
lizzie c487a5cbc4 properly do tls + array pod (#3529)
Signed-off-by: lizzie <lizzie@eden-emu.dev>
Reviewed-on: https://git.eden-emu.dev/eden-emu/eden/pulls/3529
Co-authored-by: lizzie <lizzie@eden-emu.dev>
Co-committed-by: lizzie <lizzie@eden-emu.dev>
2026-02-14 03:46:23 +01:00
CamilleLaVey 6b7e54e115 Revert "[engine, dma] Adjustment of the execution mask handling" 2026-02-14 03:46:23 +01:00
CamilleLaVey 8bbbd28f48 Revert "fix license headers" 2026-02-14 03:46:23 +01:00
CamilleLaVey ca7c1c7230 Revert "small change" 2026-02-14 03:46:23 +01:00
CamilleLaVey db10193a16 Revert "fix build" 2026-02-14 03:46:23 +01:00
CamilleLaVey 8c045a47d2 fix build 2026-02-14 03:46:23 +01:00
CamilleLaVey 616e0a93c7 small change 2026-02-14 03:46:23 +01:00
CamilleLaVey 78e7037ccc fix license headers 2026-02-14 03:46:23 +01:00
CamilleLaVey 635335fc85 [engine, dma] Adjustment of the execution mask handling 2026-02-14 03:46:23 +01:00
CamilleLaVey e35efd3db1 fix build 2026-02-14 03:46:23 +01:00
CamilleLaVey 901f556af5 [buffer_cache, memory, pipeline] Added translation entry structure and update buffer cache handling 2026-02-14 03:46:23 +01:00
CamilleLaVey ef730ed490 fix build 2026-02-14 03:46:23 +01:00
CamilleLaVey c188b6a819 [vulkan, texture_cache] Add RT's management and feedback loop tracking in texture cache 2026-02-14 03:46:23 +01:00
wildcard 476f035672 [vulkan] disable robustbufferaccess when gpu debugging is off 2026-02-14 03:46:23 +01:00
CamilleLaVey b4a1207dab Revert "[vulkan] Added dirty tracking in PushDescriptor" 2026-02-14 03:46:23 +01:00
CamilleLaVey 59c41f0746 Revert "fix building" 2026-02-14 03:46:23 +01:00
CamilleLaVey c2f0a1c786 fix building 2026-02-14 03:46:23 +01:00
CamilleLaVey a711a4e6ab [vulkan] Added dirty tracking in PushDescriptor 2026-02-14 03:46:23 +01:00
CamilleLaVey 5aa9ed4dc0 fix license headers 2026-02-14 03:46:23 +01:00
CamilleLaVey 1b8f4c2062 [common, gpu] Added a page bitset range to convert Subtract within linear operations 2026-02-14 03:46:23 +01:00
CamilleLaVey 1eaf8bb28f [buffer_cache] Adjusted PopAsyncBuffers for range management 2026-02-14 03:46:23 +01:00
CamilleLaVey 28bf78db3a [common, gpu_thread] Exchanging command queue + adjusting Subtract in the OverlapRangeSetImpl 2026-02-14 03:46:23 +01:00
23 changed files with 576 additions and 153 deletions
+143
View File
@@ -0,0 +1,143 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
#pragma once
#include <algorithm>
#include <cstddef>
#include <cstdint>
#include <vector>
#include "common/common_types.h"
#include "common/div_ceil.h"
namespace Common {
template <typename AddressType, u32 PageBits, u64 MaxAddress>
class PageBitsetRangeSet {
public:
PageBitsetRangeSet() : m_words(kWordCount) {
static_assert((MaxAddress % kPageSize) == 0, "MaxAddress must be page-aligned.");
}
void Add(AddressType base_address, size_t size) {
Modify(base_address, size, true);
}
void Subtract(AddressType base_address, size_t size) {
Modify(base_address, size, false);
}
void Clear() {
std::fill(m_words.begin(), m_words.end(), 0);
}
bool Empty() const {
return std::none_of(m_words.begin(), m_words.end(), [](u64 word) { return word != 0; });
}
template <typename Func>
void ForEachInRange(AddressType base_address, size_t size, Func&& func) const {
if (size == 0 || m_words.empty()) {
return;
}
const AddressType end_address = base_address + static_cast<AddressType>(size);
const u64 start_page = static_cast<u64>(base_address) >> PageBits;
const u64 end_page = Common::DivCeil(static_cast<u64>(end_address), kPageSize);
const u64 clamped_start = (std::min)(start_page, kPageCount);
const u64 clamped_end = (std::min)(end_page, kPageCount);
if (clamped_start >= clamped_end) {
return;
}
bool in_run = false;
u64 run_start_page = 0;
for (u64 page = clamped_start; page < clamped_end; ++page) {
if (Test(page)) {
if (!in_run) {
in_run = true;
run_start_page = page;
}
continue;
}
if (in_run) {
EmitRun(run_start_page, page, base_address, end_address, func);
in_run = false;
}
}
if (in_run) {
EmitRun(run_start_page, clamped_end, base_address, end_address, func);
}
}
private:
static constexpr u64 kPageSize = u64{1} << PageBits;
static constexpr u64 kPageCount = MaxAddress / kPageSize;
static constexpr u64 kWordCount = (kPageCount + 63) / 64;
void Modify(AddressType base_address, size_t size, bool set_bits) {
if (size == 0 || m_words.empty()) {
return;
}
const AddressType end_address = base_address + static_cast<AddressType>(size);
const u64 start_page = static_cast<u64>(base_address) >> PageBits;
const u64 end_page = Common::DivCeil(static_cast<u64>(end_address), kPageSize);
const u64 clamped_start = (std::min)(start_page, kPageCount);
const u64 clamped_end = (std::min)(end_page, kPageCount);
if (clamped_start >= clamped_end) {
return;
}
const u64 start_word = clamped_start / 64;
const u64 end_word = (clamped_end - 1) / 64;
const u64 start_mask = ~0ULL << (clamped_start % 64);
const u64 end_mask = (clamped_end % 64) == 0 ? ~0ULL : ((1ULL << (clamped_end % 64)) - 1);
if (start_word == end_word) {
const u64 mask = start_mask & end_mask;
if (set_bits) {
m_words[start_word] |= mask;
} else {
m_words[start_word] &= ~mask;
}
return;
}
if (set_bits) {
m_words[start_word] |= start_mask;
m_words[end_word] |= end_mask;
std::fill(m_words.begin() + static_cast<std::ptrdiff_t>(start_word + 1),
m_words.begin() + static_cast<std::ptrdiff_t>(end_word), ~0ULL);
} else {
m_words[start_word] &= ~start_mask;
m_words[end_word] &= ~end_mask;
std::fill(m_words.begin() + static_cast<std::ptrdiff_t>(start_word + 1),
m_words.begin() + static_cast<std::ptrdiff_t>(end_word), 0ULL);
}
}
bool Test(u64 page) const {
const u64 word = page / 64;
const u64 bit = page % 64;
return (m_words[word] & (1ULL << bit)) != 0;
}
template <typename Func>
static void EmitRun(u64 run_start_page, u64 run_end_page, AddressType base_address,
AddressType end_address, Func&& func) {
AddressType run_start = static_cast<AddressType>(run_start_page * kPageSize);
AddressType run_end = static_cast<AddressType>(run_end_page * kPageSize);
if (run_start < base_address) {
run_start = base_address;
}
if (run_end > end_address) {
run_end = end_address;
}
if (run_start < run_end) {
func(run_start, run_end);
}
}
std::vector<u64> m_words;
};
} // namespace Common
+18 -18
View File
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: 2024 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later
@@ -116,28 +119,25 @@ struct OverlapRangeSet<AddressType>::OverlapRangeSetImpl {
}
AddressType end_address = base_address + static_cast<AddressType>(size);
IntervalType interval{base_address, end_address};
bool any_removals = false;
m_split_ranges_set += std::make_pair(interval, -amount);
do {
any_removals = false;
auto it = m_split_ranges_set.lower_bound(interval);
if (it == m_split_ranges_set.end()) {
return;
}
auto end_it = m_split_ranges_set.upper_bound(interval);
for (; it != end_it; it++) {
if (it->second <= 0) {
if constexpr (has_on_delete) {
if (it->second == 0) {
on_delete(it->first.lower(), it->first.upper());
}
auto it = m_split_ranges_set.lower_bound(interval);
if (it == m_split_ranges_set.end()) {
return;
}
auto end_it = m_split_ranges_set.upper_bound(interval);
while (it != end_it) {
if (it->second <= 0) {
if constexpr (has_on_delete) {
if (it->second == 0) {
on_delete(it->first.lower(), it->first.upper());
}
any_removals = true;
m_split_ranges_set.erase(it);
break;
}
auto to_erase = it++;
m_split_ranges_set.erase(to_erase);
continue;
}
} while (any_removals);
++it;
}
}
template <typename Func>
+7
View File
@@ -127,6 +127,11 @@ public:
void UpdatePagesCachedBatch(std::span<const std::pair<DAddr, size_t>> ranges, s32 delta);
private:
struct TranslationEntry {
DAddr guest_page{};
u8* host_ptr{};
};
// Internal helper that performs the update assuming the caller already holds the necessary lock.
void UpdatePagesCachedCountNoLock(DAddr addr, size_t size, s32 delta);
@@ -195,6 +200,8 @@ private:
}
Common::VirtualBuffer<VAddr> cpu_backing_address;
std::array<TranslationEntry, 4> t_slot{};
u32 cache_cursor = 0;
using CounterType = u8;
using CounterAtomicType = std::atomic_uint8_t;
static constexpr size_t subentries = 8 / sizeof(CounterType);
+43 -1
View File
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
@@ -247,6 +247,7 @@ void DeviceMemoryManager<Traits>::Map(DAddr address, VAddr virtual_address, size
}
impl->multi_dev_address.Register(new_dev, start_id);
}
t_slot = {};
if (track) {
TrackContinuityImpl(address, virtual_address, size, asid);
}
@@ -278,6 +279,7 @@ void DeviceMemoryManager<Traits>::Unmap(DAddr address, size_t size) {
compressed_device_addr[phys_addr - 1] = new_start | MULTI_FLAG;
}
}
t_slot = {};
}
template <typename Traits>
void DeviceMemoryManager<Traits>::TrackContinuityImpl(DAddr address, VAddr virtual_address,
@@ -417,6 +419,26 @@ void DeviceMemoryManager<Traits>::WalkBlock(DAddr addr, std::size_t size, auto o
template <typename Traits>
void DeviceMemoryManager<Traits>::ReadBlock(DAddr address, void* dest_pointer, size_t size) {
device_inter->FlushRegion(address, size);
const std::size_t page_offset = address & Memory::YUZU_PAGEMASK;
if (size <= Memory::YUZU_PAGESIZE - page_offset) {
const DAddr guest_page = address & ~static_cast<DAddr>(Memory::YUZU_PAGEMASK);
for (size_t i = 0; i < 4; ++i) {
if (t_slot[i].guest_page == guest_page && t_slot[i].host_ptr != nullptr) {
std::memcpy(dest_pointer, t_slot[i].host_ptr + page_offset, size);
return;
}
}
const std::size_t page_index = address >> Memory::YUZU_PAGEBITS;
const auto phys_addr = compressed_physical_ptr[page_index];
if (phys_addr != 0) {
auto* const mem_ptr = GetPointerFromRaw<u8>((PAddr(phys_addr - 1) << Memory::YUZU_PAGEBITS));
t_slot[cache_cursor % t_slot.size()] = TranslationEntry{.guest_page = guest_page, .host_ptr = mem_ptr};
cache_cursor = (cache_cursor + 1) & 3U;
std::memcpy(dest_pointer, mem_ptr + page_offset, size);
return;
}
}
WalkBlock(
address, size,
[&](size_t copy_amount, DAddr current_vaddr) {
@@ -455,6 +477,26 @@ void DeviceMemoryManager<Traits>::WriteBlock(DAddr address, const void* src_poin
template <typename Traits>
void DeviceMemoryManager<Traits>::ReadBlockUnsafe(DAddr address, void* dest_pointer, size_t size) {
const std::size_t page_offset = address & Memory::YUZU_PAGEMASK;
if (size <= Memory::YUZU_PAGESIZE - page_offset) {
const DAddr guest_page = address & ~static_cast<DAddr>(Memory::YUZU_PAGEMASK);
for (size_t i = 0; i < 4; ++i) {
if (t_slot[i].guest_page == guest_page && t_slot[i].host_ptr != nullptr) {
std::memcpy(dest_pointer, t_slot[i].host_ptr + page_offset, size);
return;
}
}
const std::size_t page_index = address >> Memory::YUZU_PAGEBITS;
const auto phys_addr = compressed_physical_ptr[page_index];
if (phys_addr != 0) {
auto* const mem_ptr = GetPointerFromRaw<u8>((PAddr(phys_addr - 1) << Memory::YUZU_PAGEBITS));
t_slot[cache_cursor % t_slot.size()] = TranslationEntry{.guest_page = guest_page, .host_ptr = mem_ptr};
cache_cursor = (cache_cursor + 1) & 3U;
std::memcpy(dest_pointer, mem_ptr + page_offset, size);
return;
}
}
WalkBlock(
address, size,
[&](size_t copy_amount, DAddr current_vaddr) {
+7 -1
View File
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2022 yuzu Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
@@ -39,7 +42,8 @@ public:
static constexpr u64 BASE_PAGE_SIZE = 1ULL << BASE_PAGE_BITS;
explicit BufferBase(VAddr cpu_addr_, u64 size_bytes_)
: cpu_addr{cpu_addr_}, size_bytes{size_bytes_} {}
: cpu_addr_cached{static_cast<DAddr>(cpu_addr_)}, cpu_addr{cpu_addr_},
size_bytes{size_bytes_} {}
explicit BufferBase(NullBufferParams) {}
@@ -97,6 +101,8 @@ public:
return cpu_addr;
}
DAddr cpu_addr_cached = 0;
/// Returns the offset relative to the given CPU address
/// @pre IsInBounds returns true
[[nodiscard]] u32 Offset(VAddr other_cpu_addr) const noexcept {
+86 -34
View File
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2022 yuzu Emulator Project
@@ -382,6 +382,10 @@ void BufferCache<P>::BindHostComputeBuffers() {
BindHostComputeUniformBuffers();
BindHostComputeStorageBuffers();
BindHostComputeTextureBuffers();
if (any_buffer_uploaded) {
runtime.PostCopyBarrier();
any_buffer_uploaded = false;
}
}
template <class P>
@@ -680,6 +684,7 @@ void BufferCache<P>::PopAsyncBuffers() {
auto& async_buffer = async_buffers.front();
u8* base = async_buffer->mapped_span.data();
const size_t base_offset = async_buffer->offset;
Common::RangeSet<DAddr> ranges_to_remove;
for (const auto& copy : downloads) {
const DAddr device_addr = static_cast<DAddr>(copy.src_offset);
const u64 dst_offset = copy.dst_offset - base_offset;
@@ -689,9 +694,11 @@ void BufferCache<P>::PopAsyncBuffers() {
end - start);
});
async_downloads.Subtract(device_addr, copy.size, [&](DAddr start, DAddr end) {
gpu_modified_ranges.Subtract(start, end - start);
ranges_to_remove.Add(start, end - start);
});
}
ranges_to_remove.ForEach(
[&](DAddr start, DAddr end) { gpu_modified_ranges.Subtract(start, end - start); });
async_buffers_death_ring.emplace_back(*async_buffer);
async_buffers.pop_front();
pending_downloads.pop_front();
@@ -763,45 +770,85 @@ void BufferCache<P>::BindHostIndexBuffer() {
}
}
template <class P>
void BufferCache<P>::BindHostVertexBuffer(u32 index, Buffer& buffer, u32 offset, u32 size,
u32 stride) {
if constexpr (IS_OPENGL) {
runtime.BindVertexBuffer(index, buffer, offset, size, stride);
} else {
runtime.BindVertexBuffer(index, buffer.Handle(), offset, size, stride);
}
}
template <class P>
Binding& BufferCache<P>::VertexBufferSlot(u32 index) {
ASSERT(index < NUM_VERTEX_BUFFERS);
return v_buffer[index];
}
template <class P>
const Binding& BufferCache<P>::VertexBufferSlot(u32 index) const {
ASSERT(index < NUM_VERTEX_BUFFERS);
return v_buffer[index];
}
template <class P>
void BufferCache<P>::UpdateVertexBufferSlot(u32 index, const Binding& binding) {
Binding& slot = VertexBufferSlot(index);
if (slot.device_addr != binding.device_addr || slot.size != binding.size) {
++vertex_buffers_serial;
}
slot = binding;
if (binding.buffer_id != NULL_BUFFER_ID && binding.size != 0) {
enabled_vertex_buffers_mask |= (1u << index);
} else {
enabled_vertex_buffers_mask &= ~(1u << index);
}
}
template <class P>
void BufferCache<P>::BindHostVertexBuffers() {
HostBindings<typename P::Buffer> host_bindings;
bool any_valid{false};
auto& flags = maxwell3d->dirty.flags;
for (u32 index = 0; index < NUM_VERTEX_BUFFERS; ++index) {
const Binding& binding = channel_state->vertex_buffers[index];
u32 enabled_mask = enabled_vertex_buffers_mask;
HostBindings<Buffer> bindings{};
u32 last_index = std::numeric_limits<u32>::max();
const auto flush_bindings = [&]() {
if (bindings.buffers.empty()) {
return;
}
bindings.max_index = bindings.min_index + static_cast<u32>(bindings.buffers.size());
runtime.BindVertexBuffers(bindings);
bindings = HostBindings<Buffer>{};
last_index = std::numeric_limits<u32>::max();
};
while (enabled_mask != 0) {
const u32 index = std::countr_zero(enabled_mask);
enabled_mask &= (enabled_mask - 1);
const Binding& binding = VertexBufferSlot(index);
Buffer& buffer = slot_buffers[binding.buffer_id];
TouchBuffer(buffer, binding.buffer_id);
SynchronizeBuffer(buffer, binding.device_addr, binding.size);
if (!flags[Dirty::VertexBuffer0 + index]) {
flush_bindings();
continue;
}
flags[Dirty::VertexBuffer0 + index] = false;
host_bindings.min_index = (std::min)(host_bindings.min_index, index);
host_bindings.max_index = (std::max)(host_bindings.max_index, index);
any_valid = true;
}
if (any_valid) {
host_bindings.max_index++;
for (u32 index = host_bindings.min_index; index < host_bindings.max_index; index++) {
flags[Dirty::VertexBuffer0 + index] = false;
const Binding& binding = channel_state->vertex_buffers[index];
Buffer& buffer = slot_buffers[binding.buffer_id];
const u32 stride = maxwell3d->regs.vertex_streams[index].stride;
const u32 offset = buffer.Offset(binding.device_addr);
buffer.MarkUsage(offset, binding.size);
host_bindings.buffers.push_back(&buffer);
host_bindings.offsets.push_back(offset);
host_bindings.sizes.push_back(binding.size);
host_bindings.strides.push_back(stride);
const u32 stride = maxwell3d->regs.vertex_streams[index].stride;
const u32 offset = buffer.Offset(binding.device_addr);
buffer.MarkUsage(offset, binding.size);
if (!bindings.buffers.empty() && index != last_index + 1) {
flush_bindings();
}
runtime.BindVertexBuffers(host_bindings);
if (bindings.buffers.empty()) {
bindings.min_index = index;
}
bindings.buffers.push_back(&buffer);
bindings.offsets.push_back(offset);
bindings.sizes.push_back(binding.size);
bindings.strides.push_back(stride);
last_index = index;
}
flush_bindings();
}
template <class P>
@@ -1205,17 +1252,20 @@ void BufferCache<P>::UpdateVertexBuffer(u32 index) {
u32 size = address_size; // TODO: Analyze stride and number of vertices
if (array.enable == 0 || size == 0 || !device_addr) {
channel_state->vertex_buffers[index] = NULL_BINDING;
UpdateVertexBufferSlot(index, NULL_BINDING);
return;
}
if (!gpu_memory->IsWithinGPUAddressRange(gpu_addr_end) || size >= 64_MiB) {
size = static_cast<u32>(gpu_memory->MaxContinuousRange(gpu_addr_begin, size));
}
const BufferId buffer_id = FindBuffer(*device_addr, size);
channel_state->vertex_buffers[index] = Binding{
const Binding binding{
.device_addr = *device_addr,
.size = size,
.buffer_id = buffer_id,
};
channel_state->vertex_buffers[index] = binding;
UpdateVertexBufferSlot(index, binding);
}
template <class P>
@@ -1528,12 +1578,12 @@ void BufferCache<P>::TouchBuffer(Buffer& buffer, BufferId buffer_id) noexcept {
template <class P>
bool BufferCache<P>::SynchronizeBuffer(Buffer& buffer, DAddr device_addr, u32 size) {
boost::container::small_vector<BufferCopy, 4> copies;
upload_copies.clear();
u64 total_size_bytes = 0;
u64 largest_copy = 0;
DAddr buffer_start = buffer.CpuAddr();
const DAddr buffer_start = buffer.cpu_addr_cached;
memory_tracker.ForEachUploadRange(device_addr, size, [&](u64 device_addr_out, u64 range_size) {
copies.push_back(BufferCopy{
upload_copies.push_back(BufferCopy{
.src_offset = total_size_bytes,
.dst_offset = device_addr_out - buffer_start,
.size = range_size,
@@ -1544,8 +1594,9 @@ bool BufferCache<P>::SynchronizeBuffer(Buffer& buffer, DAddr device_addr, u32 si
if (total_size_bytes == 0) {
return true;
}
const std::span<BufferCopy> copies_span(copies.data(), copies.size());
const std::span<BufferCopy> copies_span(upload_copies.data(), upload_copies.size());
UploadMemory(buffer, total_size_bytes, largest_copy, copies_span);
any_buffer_uploaded = true;
return false;
}
@@ -1735,6 +1786,7 @@ void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) {
auto& binding = channel_state->vertex_buffers[index];
if (binding.buffer_id == buffer_id) {
binding.buffer_id = BufferId{};
UpdateVertexBufferSlot(index, binding);
dirty_vertex_buffers.push_back(index);
}
}
@@ -21,6 +21,7 @@
#include "common/div_ceil.h"
#include "common/literals.h"
#include "common/lru_cache.h"
#include "common/page_bitset_range_set.h"
#include "common/range_sets.h"
#include "common/scope_exit.h"
#include "common/settings.h"
@@ -320,6 +321,7 @@ public:
std::recursive_mutex mutex;
Runtime& runtime;
bool any_buffer_uploaded = false;
private:
template <typename Func>
@@ -372,6 +374,8 @@ private:
void BindHostTransformFeedbackBuffers();
void BindHostVertexBuffer(u32 index, Buffer& buffer, u32 offset, u32 size, u32 stride);
void BindHostComputeUniformBuffers();
void BindHostComputeStorageBuffers();
@@ -453,6 +457,12 @@ private:
[[nodiscard]] bool HasFastUniformBufferBound(size_t stage, u32 binding_index) const noexcept;
[[nodiscard]] Binding& VertexBufferSlot(u32 index);
[[nodiscard]] const Binding& VertexBufferSlot(u32 index) const;
void UpdateVertexBufferSlot(u32 index, const Binding& binding);
void ClearDownload(DAddr base_addr, u64 size);
void InlineMemoryImplementation(DAddr dest_address, size_t copy_size,
@@ -472,9 +482,15 @@ private:
u32 last_index_count = 0;
u32 enabled_vertex_buffers_mask = 0;
u64 vertex_buffers_serial = 0;
std::array<Binding, 32> v_buffer{};
boost::container::small_vector<BufferCopy, 4> upload_copies;
MemoryTracker memory_tracker;
Common::RangeSet<DAddr> uncommitted_gpu_modified_ranges;
Common::RangeSet<DAddr> gpu_modified_ranges;
Common::PageBitsetRangeSet<DAddr, CACHING_PAGEBITS, (1ULL << 34)> gpu_modified_ranges;
std::deque<Common::RangeSet<DAddr>> committed_gpu_modified_ranges;
// Async Buffers
+4 -1
View File
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2019 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later
@@ -89,7 +92,7 @@ struct CommandDataContainer {
/// Struct used to synchronize the GPU thread
struct SynchState final {
using CommandQueue = Common::MPSCQueue<CommandDataContainer>;
using CommandQueue = Common::SPSCQueue<CommandDataContainer>;
std::mutex write_lock;
CommandQueue queue;
u64 last_fence{};
+28 -27
View File
@@ -89,7 +89,7 @@ public:
: HLEMacroImpl(maxwell3d_)
{}
void Execute(const std::vector<u32>& parameters, [[maybe_unused]] u32 method) override {
void Execute(std::span<const u32> parameters, [[maybe_unused]] u32 method) override {
auto topology = static_cast<Maxwell3D::Regs::PrimitiveTopology>(parameters[0]);
if (!maxwell3d.AnyParametersDirty() || !IsTopologySafe(topology)) {
Fallback(parameters);
@@ -120,7 +120,7 @@ public:
}
private:
void Fallback(const std::vector<u32>& parameters) {
void Fallback(std::span<const u32> parameters) {
SCOPE_EXIT {
if (extended) {
maxwell3d.engine_state = Maxwell3D::EngineHint::None;
@@ -167,7 +167,7 @@ class HLE_DrawIndexedIndirect final : public HLEMacroImpl {
public:
explicit HLE_DrawIndexedIndirect(Maxwell3D& maxwell3d_) : HLEMacroImpl(maxwell3d_) {}
void Execute(const std::vector<u32>& parameters, [[maybe_unused]] u32 method) override {
void Execute(std::span<const u32> parameters, [[maybe_unused]] u32 method) override {
auto topology = static_cast<Maxwell3D::Regs::PrimitiveTopology>(parameters[0]);
if (!maxwell3d.AnyParametersDirty() || !IsTopologySafe(topology)) {
Fallback(parameters);
@@ -207,7 +207,7 @@ public:
}
private:
void Fallback(const std::vector<u32>& parameters) {
void Fallback(std::span<const u32> parameters) {
maxwell3d.RefreshParameters();
const u32 instance_count = (maxwell3d.GetRegisterValue(0xD1B) & parameters[2]);
const u32 element_base = parameters[4];
@@ -238,7 +238,7 @@ class HLE_MultiLayerClear final : public HLEMacroImpl {
public:
explicit HLE_MultiLayerClear(Maxwell3D& maxwell3d_) : HLEMacroImpl(maxwell3d_) {}
void Execute(const std::vector<u32>& parameters, [[maybe_unused]] u32 method) override {
void Execute(std::span<const u32> parameters, [[maybe_unused]] u32 method) override {
maxwell3d.RefreshParameters();
ASSERT(parameters.size() == 1);
@@ -256,7 +256,7 @@ class HLE_MultiDrawIndexedIndirectCount final : public HLEMacroImpl {
public:
explicit HLE_MultiDrawIndexedIndirectCount(Maxwell3D& maxwell3d_) : HLEMacroImpl(maxwell3d_) {}
void Execute(const std::vector<u32>& parameters, [[maybe_unused]] u32 method) override {
void Execute(std::span<const u32> parameters, [[maybe_unused]] u32 method) override {
const auto topology = Maxwell3D::Regs::PrimitiveTopology(parameters[2]);
if (!IsTopologySafe(topology)) {
Fallback(parameters);
@@ -301,7 +301,7 @@ public:
}
private:
void Fallback(const std::vector<u32>& parameters) {
void Fallback(std::span<const u32> parameters) {
SCOPE_EXIT {
// Clean everything.
maxwell3d.regs.vertex_id_base = 0x0;
@@ -347,7 +347,7 @@ class HLE_DrawIndirectByteCount final : public HLEMacroImpl {
public:
explicit HLE_DrawIndirectByteCount(Maxwell3D& maxwell3d_) : HLEMacroImpl(maxwell3d_) {}
void Execute(const std::vector<u32>& parameters, [[maybe_unused]] u32 method) override {
void Execute(std::span<const u32> parameters, [[maybe_unused]] u32 method) override {
const bool force = maxwell3d.Rasterizer().HasDrawTransformFeedback();
auto topology = static_cast<Maxwell3D::Regs::PrimitiveTopology>(parameters[0] & 0xFFFFU);
@@ -372,7 +372,7 @@ public:
}
private:
void Fallback(const std::vector<u32>& parameters) {
void Fallback(std::span<const u32> parameters) {
maxwell3d.RefreshParameters();
maxwell3d.regs.draw.begin = parameters[0];
@@ -381,7 +381,8 @@ private:
maxwell3d.draw_manager->DrawArray(
maxwell3d.regs.draw.topology, 0,
maxwell3d.regs.draw_auto_byte_count / maxwell3d.regs.draw_auto_stride, 0, 1);
maxwell3d.regs.draw_auto_stride > 0 ? maxwell3d.regs.draw_auto_byte_count / maxwell3d.regs.draw_auto_stride : 0,
0, 1);
}
};
@@ -389,7 +390,7 @@ class HLE_C713C83D8F63CCF3 final : public HLEMacroImpl {
public:
explicit HLE_C713C83D8F63CCF3(Maxwell3D& maxwell3d_) : HLEMacroImpl(maxwell3d_) {}
void Execute(const std::vector<u32>& parameters, [[maybe_unused]] u32 method) override {
void Execute(std::span<const u32> parameters, [[maybe_unused]] u32 method) override {
maxwell3d.RefreshParameters();
const u32 offset = (parameters[0] & 0x3FFFFFFF) << 2;
const u32 address = maxwell3d.regs.shadow_scratch[24];
@@ -405,7 +406,7 @@ class HLE_D7333D26E0A93EDE final : public HLEMacroImpl {
public:
explicit HLE_D7333D26E0A93EDE(Maxwell3D& maxwell3d_) : HLEMacroImpl(maxwell3d_) {}
void Execute(const std::vector<u32>& parameters, [[maybe_unused]] u32 method) override {
void Execute(std::span<const u32> parameters, [[maybe_unused]] u32 method) override {
maxwell3d.RefreshParameters();
const size_t index = parameters[0];
const u32 address = maxwell3d.regs.shadow_scratch[42 + index];
@@ -421,7 +422,7 @@ class HLE_BindShader final : public HLEMacroImpl {
public:
explicit HLE_BindShader(Maxwell3D& maxwell3d_) : HLEMacroImpl(maxwell3d_) {}
void Execute(const std::vector<u32>& parameters, [[maybe_unused]] u32 method) override {
void Execute(std::span<const u32> parameters, [[maybe_unused]] u32 method) override {
maxwell3d.RefreshParameters();
auto& regs = maxwell3d.regs;
const u32 index = parameters[0];
@@ -451,7 +452,7 @@ class HLE_SetRasterBoundingBox final : public HLEMacroImpl {
public:
explicit HLE_SetRasterBoundingBox(Maxwell3D& maxwell3d_) : HLEMacroImpl(maxwell3d_) {}
void Execute(const std::vector<u32>& parameters, [[maybe_unused]] u32 method) override {
void Execute(std::span<const u32> parameters, [[maybe_unused]] u32 method) override {
maxwell3d.RefreshParameters();
const u32 raster_mode = parameters[0];
auto& regs = maxwell3d.regs;
@@ -467,7 +468,7 @@ class HLE_ClearConstBuffer final : public HLEMacroImpl {
public:
explicit HLE_ClearConstBuffer(Maxwell3D& maxwell3d_) : HLEMacroImpl(maxwell3d_) {}
void Execute(const std::vector<u32>& parameters, [[maybe_unused]] u32 method) override {
void Execute(std::span<const u32> parameters, [[maybe_unused]] u32 method) override {
maxwell3d.RefreshParameters();
static constexpr std::array<u32, base_size> zeroes{};
auto& regs = maxwell3d.regs;
@@ -483,7 +484,7 @@ class HLE_ClearMemory final : public HLEMacroImpl {
public:
explicit HLE_ClearMemory(Maxwell3D& maxwell3d_) : HLEMacroImpl(maxwell3d_) {}
void Execute(const std::vector<u32>& parameters, [[maybe_unused]] u32 method) override {
void Execute(std::span<const u32> parameters, [[maybe_unused]] u32 method) override {
maxwell3d.RefreshParameters();
const u32 needed_memory = parameters[2] / sizeof(u32);
@@ -507,7 +508,7 @@ class HLE_TransformFeedbackSetup final : public HLEMacroImpl {
public:
explicit HLE_TransformFeedbackSetup(Maxwell3D& maxwell3d_) : HLEMacroImpl(maxwell3d_) {}
void Execute(const std::vector<u32>& parameters, [[maybe_unused]] u32 method) override {
void Execute(std::span<const u32> parameters, [[maybe_unused]] u32 method) override {
maxwell3d.RefreshParameters();
auto& regs = maxwell3d.regs;
@@ -560,12 +561,12 @@ std::unique_ptr<CachedMacro> HLEMacro::GetHLEProgram(u64 hash) const {
namespace {
class MacroInterpreterImpl final : public CachedMacro {
public:
explicit MacroInterpreterImpl(Engines::Maxwell3D& maxwell3d_, const std::vector<u32>& code_)
explicit MacroInterpreterImpl(Engines::Maxwell3D& maxwell3d_, std::span<const u32> code_)
: CachedMacro(maxwell3d_)
, code{code_}
{}
void Execute(const std::vector<u32>& params, u32 method) override;
void Execute(std::span<const u32> params, u32 method) override;
private:
/// Resets the execution engine state, zeroing registers, etc.
@@ -630,10 +631,10 @@ private:
u32 next_parameter_index = 0;
bool carry_flag = false;
const std::vector<u32>& code;
std::span<const u32> code;
};
void MacroInterpreterImpl::Execute(const std::vector<u32>& params, u32 method) {
void MacroInterpreterImpl::Execute(std::span<const u32> params, u32 method) {
Reset();
registers[1] = params[0];
@@ -932,7 +933,7 @@ static const auto default_cg_mode = nullptr; //Allow RWE
class MacroJITx64Impl final : public Xbyak::CodeGenerator, public CachedMacro {
public:
explicit MacroJITx64Impl(Engines::Maxwell3D& maxwell3d_, const std::vector<u32>& code_)
explicit MacroJITx64Impl(Engines::Maxwell3D& maxwell3d_, std::span<const u32> code_)
: Xbyak::CodeGenerator(MAX_CODE_SIZE, default_cg_mode)
, CachedMacro(maxwell3d_)
, code{code_}
@@ -940,7 +941,7 @@ public:
Compile();
}
void Execute(const std::vector<u32>& parameters, u32 method) override;
void Execute(std::span<const u32> parameters, u32 method) override;
void Compile_ALU(Macro::Opcode opcode);
void Compile_AddImmediate(Macro::Opcode opcode);
@@ -992,10 +993,10 @@ private:
bool is_delay_slot{};
u32 pc{};
const std::vector<u32>& code;
std::span<const u32> code;
};
void MacroJITx64Impl::Execute(const std::vector<u32>& parameters, u32 method) {
void MacroJITx64Impl::Execute(std::span<const u32> parameters, u32 method) {
ASSERT_OR_EXECUTE(program != nullptr, { return; });
JITState state{};
state.maxwell3d = &maxwell3d;
@@ -1591,7 +1592,7 @@ void MacroEngine::ClearCode(u32 method) {
uploaded_macro_code.erase(method);
}
void MacroEngine::Execute(u32 method, const std::vector<u32>& parameters) {
void MacroEngine::Execute(u32 method, std::span<const u32> parameters) {
auto compiled_macro = macro_cache.find(method);
if (compiled_macro != macro_cache.end()) {
const auto& cache_info = compiled_macro->second;
@@ -1648,7 +1649,7 @@ void MacroEngine::Execute(u32 method, const std::vector<u32>& parameters) {
}
}
std::unique_ptr<CachedMacro> MacroEngine::Compile(const std::vector<u32>& code) {
std::unique_ptr<CachedMacro> MacroEngine::Compile(std::span<const u32> code) {
#ifdef ARCHITECTURE_x86_64
if (!is_interpreted)
return std::make_unique<MacroJITx64Impl>(maxwell3d, code);
+4 -3
View File
@@ -8,6 +8,7 @@
#include <memory>
#include <ankerl/unordered_dense.h>
#include <span>
#include <vector>
#include "common/bit_field.h"
#include "common/common_types.h"
@@ -107,7 +108,7 @@ public:
/// Executes the macro code with the specified input parameters.
/// @param parameters The parameters of the macro
/// @param method The method to execute
virtual void Execute(const std::vector<u32>& parameters, u32 method) = 0;
virtual void Execute(std::span<const u32> parameters, u32 method) = 0;
Engines::Maxwell3D& maxwell3d;
};
@@ -134,10 +135,10 @@ public:
void ClearCode(u32 method);
// Compiles the macro if its not in the cache, and executes the compiled macro
void Execute(u32 method, const std::vector<u32>& parameters);
void Execute(u32 method, std::span<const u32> parameters);
protected:
std::unique_ptr<CachedMacro> Compile(const std::vector<u32>& code);
std::unique_ptr<CachedMacro> Compile(std::span<const u32> code);
private:
struct CacheInfo {
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
@@ -189,6 +189,10 @@ void ComputePipeline::Configure() {
buffer_cache.runtime.SetEnableStorageBuffers(use_storage_buffers);
buffer_cache.runtime.SetImagePointers(textures.data(), images.data());
buffer_cache.BindHostComputeBuffers();
if (buffer_cache.any_buffer_uploaded) {
buffer_cache.runtime.PostCopyBarrier();
buffer_cache.any_buffer_uploaded = false;
}
const VideoCommon::ImageViewInOut* views_it{views.data() + num_texture_buffers +
num_image_buffers};
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
@@ -558,6 +558,10 @@ bool GraphicsPipeline::ConfigureImpl(bool is_indexed) {
if (image_binding != 0) {
glBindImageTextures(0, image_binding, images.data());
}
if (buffer_cache.any_buffer_uploaded) {
buffer_cache.runtime.PostCopyBarrier();
buffer_cache.any_buffer_uploaded = false;
}
return true;
}
@@ -112,12 +112,15 @@ void UploadImage(const Device& device, MemoryAllocator& allocator, Scheduler& sc
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([&](vk::CommandBuffer cmdbuf) {
TransitionImageLayout(cmdbuf, *image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
const VkImageLayout transfer_dst_layout = device.IsKhrUnifiedImageLayoutsSupported()
? VK_IMAGE_LAYOUT_GENERAL
: VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
TransitionImageLayout(cmdbuf, *image, transfer_dst_layout,
VK_IMAGE_LAYOUT_UNDEFINED);
cmdbuf.CopyBufferToImage(*upload_buffer, *image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
cmdbuf.CopyBufferToImage(*upload_buffer, *image, transfer_dst_layout,
regions);
TransitionImageLayout(cmdbuf, *image, VK_IMAGE_LAYOUT_GENERAL,
VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL);
transfer_dst_layout);
});
scheduler.Finish();
}
@@ -203,6 +203,10 @@ void ComputePipeline::Configure(Tegra::Engines::KeplerCompute& kepler_compute,
buffer_cache.UpdateComputeBuffers();
buffer_cache.BindHostComputeBuffers();
if (buffer_cache.any_buffer_uploaded) {
buffer_cache.runtime.PostCopyBarrier();
buffer_cache.any_buffer_uploaded = false;
}
RescalingPushConstant rescaling;
const VideoCommon::SamplerId* samplers_it{samplers.data()};
@@ -496,6 +496,10 @@ bool GraphicsPipeline::ConfigureImpl(bool is_indexed) {
if constexpr (Spec::enabled_stages[4]) {
prepare_stage(4);
}
if (buffer_cache.any_buffer_uploaded) {
buffer_cache.runtime.PostCopyBarrier();
buffer_cache.any_buffer_uploaded = false;
}
texture_cache.UpdateRenderTargets(false);
texture_cache.CheckFeedbackLoop(views);
ConfigureDraw(rescaling, render_area);
@@ -168,6 +168,10 @@ bool Scheduler::UpdateGraphicsPipeline(GraphicsPipeline* pipeline) {
return true;
}
if (pipeline->UsesExtendedDynamicState() && !pipeline->HasDynamicVertexInput()) {
state_tracker.InvalidateVertexBufferState();
}
if (!pipeline->UsesExtendedDynamicState()) {
state.needs_state_enable_refresh = true;
} else if (state.needs_state_enable_refresh) {
@@ -101,6 +101,14 @@ public:
(*flags)[Dirty::StateEnable] = true;
}
void InvalidateVertexBufferState() {
(*flags)[VideoCommon::Dirty::VertexBuffers] = true;
for (int index = VideoCommon::Dirty::VertexBuffer0;
index <= VideoCommon::Dirty::VertexBuffer31; ++index) {
(*flags)[index] = true;
}
}
bool TouchViewports() {
const bool dirty_viewports = Exchange(Dirty::Viewports, false);
const bool rescale_viewports = Exchange(VideoCommon::Dirty::RescaleViewports, false);
@@ -536,6 +536,7 @@ struct RangedBarrierRange {
};
void CopyBufferToImage(vk::CommandBuffer cmdbuf, VkBuffer src_buffer, VkImage image,
VkImageAspectFlags aspect_mask, bool is_initialized,
bool use_unified_layouts,
std::span<const VkBufferImageCopy> copies) {
static constexpr VkAccessFlags WRITE_ACCESS_FLAGS =
VK_ACCESS_SHADER_WRITE_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT |
@@ -550,6 +551,8 @@ void CopyBufferToImage(vk::CommandBuffer cmdbuf, VkBuffer src_buffer, VkImage im
range.AddLayers(region.imageSubresource);
}
const VkImageSubresourceRange subresource_range = range.SubresourceRange(aspect_mask);
const VkImageLayout transfer_dst_layout =
use_unified_layouts ? VK_IMAGE_LAYOUT_GENERAL : VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
const VkImageMemoryBarrier read_barrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
@@ -557,7 +560,7 @@ void CopyBufferToImage(vk::CommandBuffer cmdbuf, VkBuffer src_buffer, VkImage im
.srcAccessMask = WRITE_ACCESS_FLAGS,
.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT,
.oldLayout = is_initialized ? VK_IMAGE_LAYOUT_GENERAL : VK_IMAGE_LAYOUT_UNDEFINED,
.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
.newLayout = transfer_dst_layout,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = image,
@@ -569,7 +572,7 @@ void CopyBufferToImage(vk::CommandBuffer cmdbuf, VkBuffer src_buffer, VkImage im
.pNext = nullptr,
.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT,
.dstAccessMask = WRITE_ACCESS_FLAGS | READ_ACCESS_FLAGS,
.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
.oldLayout = transfer_dst_layout,
.newLayout = VK_IMAGE_LAYOUT_GENERAL,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
@@ -581,7 +584,7 @@ void CopyBufferToImage(vk::CommandBuffer cmdbuf, VkBuffer src_buffer, VkImage im
VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT |
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, VK_PIPELINE_STAGE_TRANSFER_BIT, 0,
read_barrier);
cmdbuf.CopyBufferToImage(src_buffer, image, VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, copies);
cmdbuf.CopyBufferToImage(src_buffer, image, transfer_dst_layout, copies);
// TODO: Move this to another API
cmdbuf.PipelineBarrier(
VK_PIPELINE_STAGE_TRANSFER_BIT,
@@ -703,7 +706,7 @@ void TryTransformSwizzleIfNeeded(PixelFormat format, std::array<SwizzleSource, 4
void BlitScale(Scheduler& scheduler, VkImage src_image, VkImage dst_image, const ImageInfo& info,
VkImageAspectFlags aspect_mask, const Settings::ResolutionScalingInfo& resolution,
bool up_scaling = true) {
bool use_unified_layouts, bool up_scaling = true) {
const bool is_2d = info.type == ImageType::e2D;
const auto resources = info.resources;
const VkExtent2D extent{
@@ -717,7 +720,7 @@ void BlitScale(Scheduler& scheduler, VkImage src_image, VkImage dst_image, const
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([dst_image, src_image, extent, resources, aspect_mask, resolution, is_2d,
vk_filter, up_scaling](vk::CommandBuffer cmdbuf) {
vk_filter, up_scaling, use_unified_layouts](vk::CommandBuffer cmdbuf) {
const VkOffset2D src_size{
.x = static_cast<s32>(up_scaling ? extent.width : resolution.ScaleUp(extent.width)),
.y = static_cast<s32>(is_2d && up_scaling ? extent.height
@@ -777,6 +780,10 @@ void BlitScale(Scheduler& scheduler, VkImage src_image, VkImage dst_image, const
.baseArrayLayer = 0,
.layerCount = VK_REMAINING_ARRAY_LAYERS,
};
const VkImageLayout transfer_src_layout =
use_unified_layouts ? VK_IMAGE_LAYOUT_GENERAL : VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL;
const VkImageLayout transfer_dst_layout =
use_unified_layouts ? VK_IMAGE_LAYOUT_GENERAL : VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL;
const std::array read_barriers{
VkImageMemoryBarrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
@@ -784,7 +791,7 @@ void BlitScale(Scheduler& scheduler, VkImage src_image, VkImage dst_image, const
.srcAccessMask = VK_ACCESS_MEMORY_WRITE_BIT,
.dstAccessMask = VK_ACCESS_TRANSFER_READ_BIT,
.oldLayout = VK_IMAGE_LAYOUT_GENERAL,
.newLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
.newLayout = transfer_src_layout,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = src_image,
@@ -798,7 +805,7 @@ void BlitScale(Scheduler& scheduler, VkImage src_image, VkImage dst_image, const
VK_ACCESS_TRANSFER_WRITE_BIT,
.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT,
.oldLayout = VK_IMAGE_LAYOUT_UNDEFINED, // Discard contents
.newLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
.newLayout = transfer_dst_layout,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = dst_image,
@@ -812,7 +819,7 @@ void BlitScale(Scheduler& scheduler, VkImage src_image, VkImage dst_image, const
.srcAccessMask = 0,
.dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT |
VK_ACCESS_COLOR_ATTACHMENT_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT,
.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL,
.oldLayout = transfer_src_layout,
.newLayout = VK_IMAGE_LAYOUT_GENERAL,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
@@ -825,7 +832,7 @@ void BlitScale(Scheduler& scheduler, VkImage src_image, VkImage dst_image, const
.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT,
.dstAccessMask = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT |
VK_ACCESS_COLOR_ATTACHMENT_READ_BIT | VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT,
.oldLayout = VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL,
.oldLayout = transfer_dst_layout,
.newLayout = VK_IMAGE_LAYOUT_GENERAL,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
@@ -843,8 +850,8 @@ void BlitScale(Scheduler& scheduler, VkImage src_image, VkImage dst_image, const
VK_PIPELINE_STAGE_TRANSFER_BIT;
cmdbuf.PipelineBarrier(src_stages, VK_PIPELINE_STAGE_TRANSFER_BIT,
0, nullptr, nullptr, read_barriers);
cmdbuf.BlitImage(src_image, VK_IMAGE_LAYOUT_TRANSFER_SRC_OPTIMAL, dst_image,
VK_IMAGE_LAYOUT_TRANSFER_DST_OPTIMAL, regions, vk_filter);
cmdbuf.BlitImage(src_image, transfer_src_layout, dst_image,
transfer_dst_layout, regions, vk_filter);
// After transfer, images may be used in graphics, compute, or as attachments
const VkPipelineStageFlags dst_stages =
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT |
@@ -1743,9 +1750,11 @@ void Image::UploadMemory(VkBuffer buffer, VkDeviceSize offset,
const VkImage temp_vk_image = *temp_wrapper->original_image;
const VkImageAspectFlags vk_aspect_mask = temp_wrapper->aspect_mask;
scheduler->Record([src_buffer, temp_vk_image, vk_aspect_mask, vk_copies,
const bool use_unified_layouts = runtime->device.IsKhrUnifiedImageLayoutsSupported();
scheduler->Record([src_buffer, temp_vk_image, vk_aspect_mask, vk_copies, use_unified_layouts,
keep = temp_wrapper](vk::CommandBuffer cmdbuf) {
CopyBufferToImage(cmdbuf, src_buffer, temp_vk_image, vk_aspect_mask, false, VideoCommon::FixSmallVectorADL(vk_copies));
CopyBufferToImage(cmdbuf, src_buffer, temp_vk_image, vk_aspect_mask, false,
use_unified_layouts, VideoCommon::FixSmallVectorADL(vk_copies));
});
// Use MSAACopyPass to convert from non-MSAA to MSAA
@@ -1782,9 +1791,11 @@ void Image::UploadMemory(VkBuffer buffer, VkDeviceSize offset,
const VkImageAspectFlags vk_aspect_mask = aspect_mask;
const bool was_initialized = std::exchange(initialized, true);
const bool use_unified_layouts = runtime->device.IsKhrUnifiedImageLayoutsSupported();
scheduler->Record([src_buffer, vk_image, vk_aspect_mask, was_initialized,
vk_copies](vk::CommandBuffer cmdbuf) {
CopyBufferToImage(cmdbuf, src_buffer, vk_image, vk_aspect_mask, was_initialized, VideoCommon::FixSmallVectorADL(vk_copies));
vk_copies, use_unified_layouts](vk::CommandBuffer cmdbuf) {
CopyBufferToImage(cmdbuf, src_buffer, vk_image, vk_aspect_mask, was_initialized,
use_unified_layouts, VideoCommon::FixSmallVectorADL(vk_copies));
});
if (is_rescaled) {
@@ -2071,7 +2082,8 @@ bool Image::ScaleUp(bool ignore) {
if (NeedsScaleHelper()) {
return BlitScaleHelper(true);
} else {
BlitScale(*scheduler, *original_image, *scaled_image, info, aspect_mask, resolution);
BlitScale(*scheduler, *original_image, *scaled_image, info, aspect_mask, resolution,
runtime->device.IsKhrUnifiedImageLayoutsSupported());
}
return true;
}
@@ -2096,7 +2108,8 @@ bool Image::ScaleDown(bool ignore) {
if (NeedsScaleHelper()) {
return BlitScaleHelper(false);
} else {
BlitScale(*scheduler, *scaled_image, *original_image, info, aspect_mask, resolution, false);
BlitScale(*scheduler, *scaled_image, *original_image, info, aspect_mask, resolution,
runtime->device.IsKhrUnifiedImageLayoutsSupported(), false);
}
return true;
}
+80 -34
View File
@@ -291,55 +291,59 @@ void TextureCache<P>::CheckFeedbackLoop(std::span<const ImageViewInOut> views) {
return;
}
if (render_targets_serial == last_feedback_loop_serial &&
texture_bindings_serial == last_feedback_texture_serial) {
if (last_feedback_loop_result) {
runtime.BarrierFeedbackLoop();
}
return;
}
if (rt_active_mask == 0) {
last_feedback_loop_serial = render_targets_serial;
last_feedback_texture_serial = texture_bindings_serial;
last_feedback_loop_result = false;
return;
}
const u32 depth_bit = 1u << NUM_RT;
const bool depth_active = (rt_active_mask & depth_bit) != 0;
const bool requires_barrier = [&] {
for (const auto& view : views) {
if (!view.id) {
continue;
}
bool is_render_target = false;
for (const auto& ct_view_id : render_targets.color_buffer_ids) {
if (ct_view_id && ct_view_id == view.id) {
is_render_target = true;
break;
}
}
if (!is_render_target && render_targets.depth_buffer_id == view.id) {
is_render_target = true;
}
if (is_render_target) {
continue;
}
auto& image_view = slot_image_views[view.id];
for (const auto& ct_view_id : render_targets.color_buffer_ids) {
if (!ct_view_id) {
{
bool is_continue = false;
for (size_t i = 0; i < 8; ++i)
is_continue |= (rt_active_mask & (1u << i)) && view.id == render_targets.color_buffer_ids[i];
if (is_continue)
continue;
}
auto& ct_view = slot_image_views[ct_view_id];
if (image_view.image_id == ct_view.image_id) {
return true;
}
}
if (render_targets.depth_buffer_id) {
auto& zt_view = slot_image_views[render_targets.depth_buffer_id];
if (depth_active && view.id == render_targets.depth_buffer_id)
continue;
if (image_view.image_id == zt_view.image_id) {
return true;
}
const ImageId view_image_id = slot_image_views[view.id].image_id;
{
bool is_continue = false;
for (size_t i = 0; i < 8; ++i)
is_continue |= (rt_active_mask & (1u << i)) && view_image_id == rt_image_id[i];
if (is_continue)
continue;
}
if (depth_active && view_image_id == rt_depth_image_id) {
return true;
}
}
return false;
}();
last_feedback_loop_serial = render_targets_serial;
last_feedback_texture_serial = texture_bindings_serial;
last_feedback_loop_result = requires_barrier;
if (requires_barrier) {
runtime.BarrierFeedbackLoop();
}
@@ -399,13 +403,19 @@ void TextureCache<P>::SynchronizeGraphicsDescriptors() {
const bool linked_tsc = maxwell3d->regs.sampler_binding == SamplerBinding::ViaHeaderBinding;
const u32 tic_limit = maxwell3d->regs.tex_header.limit;
const u32 tsc_limit = linked_tsc ? tic_limit : maxwell3d->regs.tex_sampler.limit;
bool bindings_changed = false;
if (channel_state->graphics_sampler_table.Synchronize(maxwell3d->regs.tex_sampler.Address(),
tsc_limit)) {
channel_state->graphics_sampler_ids.resize(tsc_limit + 1, CORRUPT_ID);
bindings_changed = true;
}
if (channel_state->graphics_image_table.Synchronize(maxwell3d->regs.tex_header.Address(),
tic_limit)) {
channel_state->graphics_image_view_ids.resize(tic_limit + 1, CORRUPT_ID);
bindings_changed = true;
}
if (bindings_changed) {
++texture_bindings_serial;
}
}
@@ -415,12 +425,18 @@ void TextureCache<P>::SynchronizeComputeDescriptors() {
const u32 tic_limit = kepler_compute->regs.tic.limit;
const u32 tsc_limit = linked_tsc ? tic_limit : kepler_compute->regs.tsc.limit;
const GPUVAddr tsc_gpu_addr = kepler_compute->regs.tsc.Address();
bool bindings_changed = false;
if (channel_state->compute_sampler_table.Synchronize(tsc_gpu_addr, tsc_limit)) {
channel_state->compute_sampler_ids.resize(tsc_limit + 1, CORRUPT_ID);
bindings_changed = true;
}
if (channel_state->compute_image_table.Synchronize(kepler_compute->regs.tic.Address(),
tic_limit)) {
channel_state->compute_image_view_ids.resize(tic_limit + 1, CORRUPT_ID);
bindings_changed = true;
}
if (bindings_changed) {
++texture_bindings_serial;
}
}
@@ -534,6 +550,7 @@ void TextureCache<P>::UpdateRenderTargets(bool is_clear) {
return;
}
const VideoCommon::RenderTargets previous_render_targets = render_targets;
const bool rescaled = RescaleRenderTargets();
if (is_rescaling != rescaled) {
flags[Dirty::RescaleViewports] = true;
@@ -549,6 +566,21 @@ void TextureCache<P>::UpdateRenderTargets(bool is_clear) {
PrepareImageView(depth_buffer_id, true, is_clear && IsFullClear(depth_buffer_id));
rt_active_mask = 0;
rt_image_id = {};
for (size_t i = 0; i < rt_image_id.size(); ++i) {
if (ImageViewId const view = render_targets.color_buffer_ids[i]; view) {
rt_active_mask |= 1u << i;
rt_image_id[i] = slot_image_views[view].image_id;
}
}
if (depth_buffer_id) {
rt_active_mask |= (1u << NUM_RT);
rt_depth_image_id = slot_image_views[depth_buffer_id].image_id;
} else {
rt_depth_image_id = ImageId{};
}
for (size_t index = 0; index < NUM_RT; ++index) {
render_targets.draw_buffers[index] = static_cast<u8>(maxwell3d->regs.rt_control.Map(index));
}
@@ -564,12 +596,22 @@ void TextureCache<P>::UpdateRenderTargets(bool is_clear) {
};
render_targets.is_rescaled = is_rescaling;
if (render_targets != previous_render_targets) {
++render_targets_serial;
}
flags[Dirty::DepthBiasGlobal] = true;
}
template <class P>
typename P::Framebuffer* TextureCache<P>::GetFramebuffer() {
return &slot_framebuffers[GetFramebufferId(render_targets)];
if (last_framebuffer_id && last_framebuffer_serial == render_targets_serial) {
return &slot_framebuffers[last_framebuffer_id];
}
const FramebufferId framebuffer_id = GetFramebufferId(render_targets);
last_framebuffer_id = framebuffer_id;
last_framebuffer_serial = render_targets_serial;
return &slot_framebuffers[framebuffer_id];
}
template <class P>
@@ -2614,6 +2656,10 @@ void TextureCache<P>::RemoveFramebuffers(std::span<const ImageViewId> removed_vi
if (it->first.Contains(removed_views)) {
auto framebuffer_id = it->second;
ASSERT(framebuffer_id);
if (framebuffer_id == last_framebuffer_id) {
last_framebuffer_id = {};
last_framebuffer_serial = 0;
}
sentenced_framebuffers.Push(std::move(slot_framebuffers[framebuffer_id]));
it = framebuffers.erase(it);
} else {
@@ -455,6 +455,16 @@ private:
std::deque<TextureCacheGPUMap> gpu_page_table_storage;
RenderTargets render_targets;
u64 render_targets_serial = 0;
u32 rt_active_mask = 0;
std::array<ImageId, 8> rt_image_id{};
ImageId rt_depth_image_id{};
u64 texture_bindings_serial = 0;
u64 last_feedback_loop_serial = 0;
u64 last_feedback_texture_serial = 0;
bool last_feedback_loop_result = false;
FramebufferId last_framebuffer_id{};
u64 last_framebuffer_serial = 0;
ankerl::unordered_dense::map<RenderTargets, FramebufferId> framebuffers;
ankerl::unordered_dense::map<u64, std::vector<ImageMapId>, Common::IdentityHash<u64>> page_table;
+46 -7
View File
@@ -924,9 +924,6 @@ bool Device::ShouldBoostClocks() const {
}
bool Device::HasTimelineSemaphore() const {
if (GetDriverID() == VK_DRIVER_ID_MESA_TURNIP) {
return false;
}
return features.timeline_semaphore.timelineSemaphore;
}
@@ -935,8 +932,18 @@ bool Device::GetSuitability(bool requires_swapchain) {
bool suitable = true;
// Configure properties.
if (!Settings::values.renderer_debug) {
features.features.robustBufferAccess = VK_FALSE;
if (extensions.robustness_2) {
features.robustness2.robustBufferAccess2 = VK_FALSE;
}
}
VkPhysicalDeviceVulkan12Features features_1_2{};
VkPhysicalDeviceVulkan13Features features_1_3{};
#ifdef VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_4_FEATURES
VkPhysicalDeviceVulkan14Features features_1_4{};
#endif
// Configure properties.
properties.properties = physical.GetProperties();
@@ -976,6 +983,11 @@ bool Device::GetSuitability(bool requires_swapchain) {
if (instance_version < VK_API_VERSION_1_3) {
FOR_EACH_VK_FEATURE_1_3(FEATURE_EXTENSION);
}
#ifdef VK_API_VERSION_1_4
if (instance_version < VK_API_VERSION_1_4) {
FOR_EACH_VK_FEATURE_1_4(FEATURE_EXTENSION);
}
#endif
FOR_EACH_VK_FEATURE_EXT(FEATURE_EXTENSION);
FOR_EACH_VK_EXTENSION(EXTENSION);
@@ -1011,11 +1023,16 @@ bool Device::GetSuitability(bool requires_swapchain) {
// Set next pointer.
void** next = &features2.pNext;
// Vulkan 1.2 and 1.3 features
// Vulkan 1.2, 1.3 and 1.4 features
if (instance_version >= VK_API_VERSION_1_2) {
features_1_2.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_2_FEATURES;
features_1_3.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_3_FEATURES;
#ifdef VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_4_FEATURES
features_1_4.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_VULKAN_1_4_FEATURES;
features_1_3.pNext = &features_1_4;
#endif
features_1_2.pNext = &features_1_3;
*next = &features_1_2;
@@ -1047,6 +1064,13 @@ bool Device::GetSuitability(bool requires_swapchain) {
} else {
FOR_EACH_VK_FEATURE_1_3(EXT_FEATURE);
}
#ifdef VK_API_VERSION_1_4
if (instance_version >= VK_API_VERSION_1_4) {
FOR_EACH_VK_FEATURE_1_4(FEATURE);
} else {
FOR_EACH_VK_FEATURE_1_4(EXT_FEATURE);
}
#endif
#undef EXT_FEATURE
#undef FEATURE
@@ -1484,16 +1508,31 @@ void Device::RemoveUnsuitableExtensions() {
RemoveExtensionFeatureIfUnsuitable(extensions.maintenance6, features.maintenance6,
VK_KHR_MAINTENANCE_6_EXTENSION_NAME);
// VK_KHR_maintenance7 (proposed for Vulkan 1.4, no features)
// VK_KHR_maintenance7
#ifdef VK_API_VERSION_1_4
extensions.maintenance7 = instance_version >= VK_API_VERSION_1_4 ||
loaded_extensions.contains(VK_KHR_MAINTENANCE_7_EXTENSION_NAME);
#else
extensions.maintenance7 = loaded_extensions.contains(VK_KHR_MAINTENANCE_7_EXTENSION_NAME);
#endif
RemoveExtensionIfUnsuitable(extensions.maintenance7, VK_KHR_MAINTENANCE_7_EXTENSION_NAME);
// VK_KHR_maintenance8 (proposed for Vulkan 1.4, no features)
// VK_KHR_maintenance8
#ifdef VK_API_VERSION_1_4
extensions.maintenance8 = instance_version >= VK_API_VERSION_1_4 ||
loaded_extensions.contains(VK_KHR_MAINTENANCE_8_EXTENSION_NAME);
#else
extensions.maintenance8 = loaded_extensions.contains(VK_KHR_MAINTENANCE_8_EXTENSION_NAME);
#endif
RemoveExtensionIfUnsuitable(extensions.maintenance8, VK_KHR_MAINTENANCE_8_EXTENSION_NAME);
// VK_KHR_maintenance9 (proposed for Vulkan 1.4, no features)
// VK_KHR_maintenance9
#ifdef VK_API_VERSION_1_4
extensions.maintenance9 = instance_version >= VK_API_VERSION_1_4 ||
loaded_extensions.contains(VK_KHR_MAINTENANCE_9_EXTENSION_NAME);
#else
extensions.maintenance9 = loaded_extensions.contains(VK_KHR_MAINTENANCE_9_EXTENSION_NAME);
#endif
RemoveExtensionIfUnsuitable(extensions.maintenance9, VK_KHR_MAINTENANCE_9_EXTENSION_NAME);
}
@@ -487,6 +487,11 @@ public:
return extensions.workgroup_memory_explicit_layout;
}
/// Returns true if the device supports VK_KHR_unified_image_layouts.
bool IsKhrUnifiedImageLayoutsSupported() const {
return extensions.unified_image_layouts;
}
/// Returns true if the device supports VK_KHR_image_format_list.
bool IsKhrImageFormatListSupported() const {
return extensions.image_format_list || instance_version >= VK_API_VERSION_1_2;
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
@@ -47,6 +47,14 @@ bool IsMicrosoftDozen(const char* device_name) {
return std::strstr(device_name, "Microsoft") != nullptr;
}
constexpr u32 MaxInstanceApiVersion() {
#ifdef VK_API_VERSION_1_4
return VK_API_VERSION_1_4;
#else
return VK_API_VERSION_1_3;
#endif
}
void SortPhysicalDevices(std::vector<VkPhysicalDevice>& devices, const InstanceDispatch& dld) {
// Sort by name, this will set a base and make GPUs with higher numbers appear first
// (e.g. GTX 1650 will intentionally be listed before a GTX 1080).
@@ -437,8 +445,8 @@ Instance Instance::Create(u32 version, Span<const char*> layers, Span<const char
#else
constexpr VkFlags ci_flags{};
#endif
// DO NOT TOUCH, breaks RNDA3!!
// Don't know why, but gloom + yellow line glitch appears
// Keep application and engine tags stable for driver behavior compatibility.
const u32 api_version = std::min(version, MaxInstanceApiVersion());
const VkApplicationInfo application_info{
.sType = VK_STRUCTURE_TYPE_APPLICATION_INFO,
.pNext = nullptr,
@@ -446,7 +454,7 @@ Instance Instance::Create(u32 version, Span<const char*> layers, Span<const char
.applicationVersion = VK_MAKE_VERSION(1, 3, 0),
.pEngineName = "yuzu Emulator",
.engineVersion = VK_MAKE_VERSION(1, 3, 0),
.apiVersion = VK_API_VERSION_1_3,
.apiVersion = api_version,
};
const VkInstanceCreateInfo ci{
.sType = VK_STRUCTURE_TYPE_INSTANCE_CREATE_INFO,