mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-10-06 14:09:58 +00:00
Port UMA part 5.1
This commit is contained in:
+1
@@ -31,6 +31,7 @@ enum class BooleanSetting(override val key: String) : AbstractBooleanSetting {
|
||||
ENABLE_BUFFER_HISTORY("enable_buffer_history"),
|
||||
USE_OPTIMIZED_VERTEX_BUFFERS("use_optimized_vertex_buffers"),
|
||||
ENABLE_GPU_BUFFER_READBACK("enable_gpu_buffer_readback"),
|
||||
USE_UNIFIED_MEMORY("use_unified_memory"),
|
||||
SYNC_MEMORY_OPERATIONS("sync_memory_operations"),
|
||||
BUFFER_REORDER_DISABLE("disable_buffer_reorder"),
|
||||
RENDERER_DEBUG("debug"),
|
||||
|
||||
+7
@@ -874,6 +874,13 @@ abstract class SettingsItem(
|
||||
descriptionId = R.string.enable_gpu_buffer_readback_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.USE_UNIFIED_MEMORY,
|
||||
titleId = R.string.use_unified_memory,
|
||||
descriptionId = R.string.use_unified_memory_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.USE_OPTIMIZED_VERTEX_BUFFERS,
|
||||
|
||||
+1
@@ -551,6 +551,7 @@ class SettingsFragmentPresenter(
|
||||
add(BooleanSetting.RENDERER_BARRIER_FEEDBACK_LOOPS.key)
|
||||
add(BooleanSetting.ENABLE_BUFFER_HISTORY.key)
|
||||
add(BooleanSetting.ENABLE_GPU_BUFFER_READBACK.key)
|
||||
add(BooleanSetting.USE_UNIFIED_MEMORY.key)
|
||||
add(BooleanSetting.USE_OPTIMIZED_VERTEX_BUFFERS.key)
|
||||
|
||||
add(HeaderSetting(R.string.hacks))
|
||||
|
||||
@@ -587,6 +587,8 @@
|
||||
<string name="enable_buffer_history">Enable buffer history</string>
|
||||
<string name="enable_buffer_history_description">Enables access to previous buffer states. This option may improve rendering quality and performance consistency in some games.</string>
|
||||
<string name="enable_gpu_buffer_readback">Enable GPU Buffer Readback</string>
|
||||
<string name="use_unified_memory">Unified Memory</string>
|
||||
<string name="use_unified_memory_description">Backs part of the emulated memory with GPU-shared buffers so textures are unswizzled directly from it, skipping a CPU copy. Needs 12 GB of RAM or more and a GPU with I/O coherency; it turns itself off otherwise. Takes effect on the next game launch.</string>
|
||||
<string name="enable_gpu_buffer_readback_description">Preserves GPU-modified buffer data by reading it back before uploads. Some games require this to render certain effects properly. May cause issues if the hardware cannot handle the additional workload.</string>
|
||||
<string name="use_optimized_vertex_buffers">Optimized Vertex Buffers</string>
|
||||
<string name="use_optimized_vertex_buffers_description">Enables optimized vertex buffer binding for improved performance. Requires Mesa 26.0+ Turnip drivers/ QCOM drivers. Will crash on older Turnip drivers (25.3 and below).</string>
|
||||
|
||||
+22
-28
@@ -4,34 +4,17 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <fstream>
|
||||
#include "common/heap_tracker.h"
|
||||
#include "common/logging.h"
|
||||
#include "common/memory_detect.h"
|
||||
#include "common/assert.h"
|
||||
|
||||
namespace Common {
|
||||
|
||||
namespace {
|
||||
|
||||
s64 GetMaxPermissibleResidentMapCount() {
|
||||
// Default value.
|
||||
s64 value = 65530;
|
||||
|
||||
// Try to read how many mappings we can make.
|
||||
std::ifstream s("/proc/sys/vm/max_map_count");
|
||||
s >> value;
|
||||
|
||||
// Print, for debug.
|
||||
LOG_INFO(HW_Memory, "Current maximum map count: {}", value);
|
||||
|
||||
// Allow 20000 maps for other code and to account for split inaccuracy.
|
||||
return std::max<s64>(value - 20000, 0);
|
||||
}
|
||||
|
||||
} // namespace
|
||||
|
||||
HeapTracker::HeapTracker(Common::HostMemory& buffer)
|
||||
: m_buffer(buffer), m_max_resident_map_count(GetMaxPermissibleResidentMapCount()) {}
|
||||
: m_buffer(buffer),
|
||||
m_has_hardware_buffer_backing(!buffer.BackingHardwareBuffers().empty()),
|
||||
m_max_resident_map_count(static_cast<s64>(GetPermissibleMapCount())) {}
|
||||
HeapTracker::~HeapTracker() = default;
|
||||
|
||||
void HeapTracker::Map(size_t virtual_offset, size_t host_offset, size_t length,
|
||||
@@ -85,7 +68,8 @@ void HeapTracker::Unmap(size_t virtual_offset, size_t size, bool is_separate_hea
|
||||
|
||||
// If resident, erase from resident map.
|
||||
if (item->is_resident) {
|
||||
ASSERT(--m_resident_map_count >= 0);
|
||||
m_resident_map_count -= this->HostMapCount(item->paddr, item->size);
|
||||
ASSERT(m_resident_map_count >= 0);
|
||||
m_resident_mappings.erase(m_resident_mappings.iterator_to(*item));
|
||||
}
|
||||
|
||||
@@ -191,7 +175,7 @@ bool HeapTracker::DeferredMapSeparateHeap(size_t virtual_offset) {
|
||||
|
||||
// This map is now resident.
|
||||
it->is_resident = true;
|
||||
m_resident_map_count++;
|
||||
m_resident_map_count += this->HostMapCount(it->paddr, it->size);
|
||||
m_resident_mappings.insert(*it);
|
||||
}
|
||||
|
||||
@@ -213,17 +197,17 @@ void HeapTracker::RebuildSeparateHeapAddressSpace() {
|
||||
// Despite being worse in theory, this has proven to be better in practice than more
|
||||
// regularly dumping a smaller amount, because it significantly reduces average case
|
||||
// lock contention.
|
||||
std::size_t const desired_count = (std::min)(m_resident_map_count, m_max_resident_map_count) / 2;
|
||||
std::size_t const evict_count = m_resident_map_count - desired_count;
|
||||
s64 const desired_count = (std::min)(m_resident_map_count, m_max_resident_map_count) / 2;
|
||||
auto it = m_resident_mappings.begin();
|
||||
|
||||
for (size_t i = 0; i < evict_count && it != m_resident_mappings.end(); i++) {
|
||||
while (m_resident_map_count > desired_count && it != m_resident_mappings.end()) {
|
||||
// Unmark and unmap.
|
||||
it->is_resident = false;
|
||||
m_buffer.Unmap(it->vaddr, it->size, false);
|
||||
|
||||
// Advance.
|
||||
ASSERT(--m_resident_map_count >= 0);
|
||||
m_resident_map_count -= this->HostMapCount(it->paddr, it->size);
|
||||
ASSERT(m_resident_map_count >= 0);
|
||||
it = m_resident_mappings.erase(it);
|
||||
}
|
||||
}
|
||||
@@ -245,6 +229,7 @@ void HeapTracker::SplitHeapMapLocked(VAddr offset) {
|
||||
// Cache the original values.
|
||||
auto* const left = std::addressof(*it);
|
||||
const size_t orig_size = left->size;
|
||||
const s64 orig_host_map_count = this->HostMapCount(left->paddr, orig_size);
|
||||
|
||||
// Adjust the left map.
|
||||
const size_t left_size = offset - left->vaddr;
|
||||
@@ -266,11 +251,20 @@ void HeapTracker::SplitHeapMapLocked(VAddr offset) {
|
||||
|
||||
// If resident, also insert into resident map.
|
||||
if (right->is_resident) {
|
||||
m_resident_map_count++;
|
||||
m_resident_map_count += this->HostMapCount(left->paddr, left->size) +
|
||||
this->HostMapCount(right->paddr, right->size) -
|
||||
orig_host_map_count;
|
||||
m_resident_mappings.insert(*right);
|
||||
}
|
||||
}
|
||||
|
||||
s64 HeapTracker::HostMapCount(PAddr paddr, size_t size) const {
|
||||
if (!m_has_hardware_buffer_backing) {
|
||||
return static_cast<s64>(size != 0);
|
||||
}
|
||||
return static_cast<s64>(m_buffer.BackingMapCount(paddr, size));
|
||||
}
|
||||
|
||||
HeapTracker::AddrTree::iterator HeapTracker::GetNearestHeapMapLocked(VAddr offset) {
|
||||
const SeparateHeapMap key{
|
||||
.vaddr = offset,
|
||||
|
||||
@@ -1,3 +1,6 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
@@ -82,10 +85,13 @@ private:
|
||||
|
||||
AddrTree::iterator GetNearestHeapMapLocked(VAddr offset);
|
||||
|
||||
s64 HostMapCount(PAddr paddr, size_t size) const;
|
||||
|
||||
void RebuildSeparateHeapAddressSpace();
|
||||
|
||||
private:
|
||||
Common::HostMemory& m_buffer;
|
||||
const bool m_has_hardware_buffer_backing;
|
||||
const s64 m_max_resident_map_count;
|
||||
|
||||
std::shared_mutex m_rebuild_lock{};
|
||||
|
||||
+341
-6
@@ -53,12 +53,42 @@
|
||||
|
||||
#include <mutex>
|
||||
#include <random>
|
||||
#include <vector>
|
||||
|
||||
#include "common/alignment.h"
|
||||
#include "common/assert.h"
|
||||
#include "common/free_region_manager.h"
|
||||
#include "common/host_memory.h"
|
||||
#include "common/logging.h"
|
||||
#include "common/memory_detect.h"
|
||||
#include "common/settings.h"
|
||||
|
||||
#ifdef __ANDROID__
|
||||
#include <dlfcn.h>
|
||||
#include <android/hardware_buffer.h>
|
||||
|
||||
namespace {
|
||||
|
||||
struct NativeHandle {
|
||||
int version;
|
||||
int numFds;
|
||||
int numInts;
|
||||
int data[1];
|
||||
};
|
||||
|
||||
using PFN_AHardwareBuffer_getNativeHandle = const NativeHandle* (*)(const AHardwareBuffer*);
|
||||
|
||||
PFN_AHardwareBuffer_getNativeHandle ResolveGetNativeHandle() {
|
||||
void* const lib = dlopen("libnativewindow.so", RTLD_NOW);
|
||||
if (lib == nullptr) {
|
||||
return nullptr;
|
||||
}
|
||||
return reinterpret_cast<PFN_AHardwareBuffer_getNativeHandle>(
|
||||
dlsym(lib, "AHardwareBuffer_getNativeHandle"));
|
||||
}
|
||||
|
||||
}
|
||||
#endif
|
||||
|
||||
#if defined(__ANDROID__) && __ANDROID_API__ < 30
|
||||
#include <sys/syscall.h>
|
||||
@@ -122,7 +152,7 @@ static void GetFuncAddress(Common::DynamicLibrary& dll, const char* name, T& pfn
|
||||
|
||||
class HostMemory::Impl {
|
||||
public:
|
||||
explicit Impl(size_t backing_size_, size_t virtual_size_)
|
||||
explicit Impl(size_t backing_size_, size_t virtual_size_, size_t)
|
||||
: backing_size{backing_size_}
|
||||
, virtual_size{virtual_size_}
|
||||
, process{GetCurrentProcess()}
|
||||
@@ -500,9 +530,10 @@ static int shm_open_anon(int flags, mode_t mode) {
|
||||
|
||||
class HostMemory::Impl {
|
||||
public:
|
||||
explicit Impl(size_t backing_size_, size_t virtual_size_)
|
||||
explicit Impl(size_t backing_size_, size_t virtual_size_, size_t preferred_offset_)
|
||||
: backing_size{backing_size_}
|
||||
, virtual_size{virtual_size_}
|
||||
, preferred_offset{preferred_offset_}
|
||||
{}
|
||||
|
||||
bool Init() {
|
||||
@@ -542,10 +573,15 @@ public:
|
||||
LOG_WARNING(Common_Memory, "Using private mappings instead of shared ones");
|
||||
backing_base = static_cast<u8*>(mmap(nullptr, backing_size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE, -1, 0));
|
||||
if (fd > 0) {
|
||||
fd = -1;
|
||||
close(fd);
|
||||
}
|
||||
fd = -1;
|
||||
} else {
|
||||
#ifdef __ANDROID__
|
||||
if (InitAhbBacking()) {
|
||||
return InitVirtual();
|
||||
}
|
||||
#endif
|
||||
backing_base = static_cast<u8*>(mmap(nullptr, backing_size, PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0));
|
||||
}
|
||||
if (backing_base == MAP_FAILED) {
|
||||
@@ -553,7 +589,10 @@ public:
|
||||
return false;
|
||||
}
|
||||
|
||||
// Virtual memory initialization
|
||||
return InitVirtual();
|
||||
}
|
||||
|
||||
bool InitVirtual() {
|
||||
virtual_base = virtual_map_base = static_cast<u8*>(ChooseVirtualBase(virtual_size));
|
||||
if (virtual_base == MAP_FAILED) {
|
||||
LOG_CRITICAL(HW_Memory, "mmap failed: {}", strerror(errno));
|
||||
@@ -566,6 +605,240 @@ public:
|
||||
return true;
|
||||
}
|
||||
|
||||
#ifdef __ANDROID__
|
||||
static AHardwareBuffer_Desc MakeBlobDesc(size_t len) {
|
||||
return AHardwareBuffer_Desc{
|
||||
.width = static_cast<u32>(len),
|
||||
.height = 1,
|
||||
.layers = 1,
|
||||
.format = AHARDWAREBUFFER_FORMAT_BLOB,
|
||||
.usage = AHARDWAREBUFFER_USAGE_CPU_READ_OFTEN |
|
||||
AHARDWAREBUFFER_USAGE_CPU_WRITE_OFTEN |
|
||||
AHARDWAREBUFFER_USAGE_GPU_DATA_BUFFER,
|
||||
.stride = 0,
|
||||
.rfu0 = 0,
|
||||
.rfu1 = 0,
|
||||
};
|
||||
}
|
||||
|
||||
static bool ProbeAhbBacking(PFN_AHardwareBuffer_getNativeHandle get_native_handle) {
|
||||
const AHardwareBuffer_Desc desc = MakeBlobDesc(PageAlignment * 2);
|
||||
AHardwareBuffer* buffer{};
|
||||
if (AHardwareBuffer_allocate(&desc, &buffer) != 0 || buffer == nullptr) {
|
||||
return false;
|
||||
}
|
||||
const NativeHandle* const handle = get_native_handle(buffer);
|
||||
if (handle == nullptr || handle->numFds < 1) {
|
||||
AHardwareBuffer_release(buffer);
|
||||
return false;
|
||||
}
|
||||
const int probe_fd = handle->data[0];
|
||||
bool ok = true;
|
||||
const auto try_map = [&](int prot, off_t offset) {
|
||||
if (!ok) {
|
||||
return;
|
||||
}
|
||||
void* const ptr = mmap(nullptr, PageAlignment, prot, MAP_SHARED, probe_fd, offset);
|
||||
if (ptr == MAP_FAILED) {
|
||||
ok = false;
|
||||
return;
|
||||
}
|
||||
munmap(ptr, PageAlignment);
|
||||
};
|
||||
try_map(PROT_READ | PROT_WRITE, 0);
|
||||
try_map(PROT_READ | PROT_WRITE, static_cast<off_t>(PageAlignment));
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
try_map(PROT_READ | PROT_EXEC, 0);
|
||||
#endif
|
||||
AHardwareBuffer_release(buffer);
|
||||
return ok;
|
||||
}
|
||||
|
||||
size_t ComputeAhbBudget(size_t window_size) const {
|
||||
const u64 total_physical = Common::GetMemInfo().TotalPhysicalMemory;
|
||||
constexpr u64 BaselineFootprint = 6ULL << 30;
|
||||
if (total_physical <= BaselineFootprint) {
|
||||
return 0;
|
||||
}
|
||||
const u64 permissible_maps = Common::GetPermissibleMapCount();
|
||||
if (permissible_maps == 0) {
|
||||
return 0;
|
||||
}
|
||||
u64 budget = (total_physical - BaselineFootprint) / 2;
|
||||
constexpr u64 MapSlotsPerWindow = 64;
|
||||
const u64 affordable_windows = permissible_maps / MapSlotsPerWindow;
|
||||
budget = (std::min)(budget, affordable_windows * window_size);
|
||||
const u64 available = Common::GetAvailablePhysicalMemory();
|
||||
if (available != 0) {
|
||||
budget = (std::min)(budget, available / 2);
|
||||
}
|
||||
budget = (std::min)(budget, static_cast<u64>(backing_size));
|
||||
budget = Common::AlignDown(budget, window_size);
|
||||
constexpr u64 MinimumBudget = 256ULL << 20;
|
||||
if (budget < MinimumBudget) {
|
||||
return 0;
|
||||
}
|
||||
return static_cast<size_t>(budget);
|
||||
}
|
||||
|
||||
bool InitAhbBacking() {
|
||||
if (!Settings::values.use_unified_memory.GetValue()) {
|
||||
return false;
|
||||
}
|
||||
static const PFN_AHardwareBuffer_getNativeHandle get_native_handle =
|
||||
ResolveGetNativeHandle();
|
||||
if (get_native_handle == nullptr) {
|
||||
return false;
|
||||
}
|
||||
constexpr size_t window_size = 256ULL << 20;
|
||||
const size_t budget = ComputeAhbBudget(window_size);
|
||||
if (budget == 0) {
|
||||
return false;
|
||||
}
|
||||
if (!ProbeAhbBacking(get_native_handle)) {
|
||||
return false;
|
||||
}
|
||||
const size_t aligned_backing = Common::AlignDown(backing_size, window_size);
|
||||
const size_t max_windows = (std::min)(budget, aligned_backing) / window_size;
|
||||
|
||||
std::vector<AHardwareBuffer*> buffers;
|
||||
std::vector<int> buffer_fds;
|
||||
const auto cleanup = [&] {
|
||||
for (AHardwareBuffer* buffer : buffers) {
|
||||
AHardwareBuffer_release(buffer);
|
||||
}
|
||||
buffers.clear();
|
||||
buffer_fds.clear();
|
||||
};
|
||||
for (size_t i = 0; i < max_windows; ++i) {
|
||||
const AHardwareBuffer_Desc desc = MakeBlobDesc(window_size);
|
||||
AHardwareBuffer* buffer{};
|
||||
if (AHardwareBuffer_allocate(&desc, &buffer) != 0 || buffer == nullptr) {
|
||||
break;
|
||||
}
|
||||
const NativeHandle* const handle = get_native_handle(buffer);
|
||||
if (handle == nullptr || handle->numFds < 1) {
|
||||
AHardwareBuffer_release(buffer);
|
||||
break;
|
||||
}
|
||||
const int buffer_fd = handle->data[0];
|
||||
const off_t buffer_len = lseek(buffer_fd, 0, SEEK_END);
|
||||
if (buffer_len < static_cast<off_t>(window_size)) {
|
||||
AHardwareBuffer_release(buffer);
|
||||
break;
|
||||
}
|
||||
buffers.push_back(buffer);
|
||||
buffer_fds.push_back(buffer_fd);
|
||||
}
|
||||
const size_t num_windows = buffers.size();
|
||||
if (num_windows == 0) {
|
||||
return false;
|
||||
}
|
||||
const size_t region_size = num_windows * window_size;
|
||||
const size_t region_base = Common::AlignDown(
|
||||
(std::min)(preferred_offset, aligned_backing - region_size), window_size);
|
||||
u8* const base = static_cast<u8*>(mmap(nullptr, backing_size, PROT_NONE,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1, 0));
|
||||
if (base == MAP_FAILED) {
|
||||
cleanup();
|
||||
return false;
|
||||
}
|
||||
const auto map_over_reservation = [&](size_t offset, size_t len, int map_fd,
|
||||
off_t map_offset) {
|
||||
if (len == 0) {
|
||||
return true;
|
||||
}
|
||||
if (mmap(base + offset, len, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_FIXED, map_fd,
|
||||
map_offset) == MAP_FAILED) {
|
||||
munmap(base, backing_size);
|
||||
cleanup();
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
};
|
||||
if (!map_over_reservation(0, region_base, fd, 0)) {
|
||||
return false;
|
||||
}
|
||||
for (size_t i = 0; i < num_windows; ++i) {
|
||||
if (!map_over_reservation(region_base + i * window_size, window_size, buffer_fds[i],
|
||||
0)) {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
const size_t tail_offset = region_base + region_size;
|
||||
if (!map_over_reservation(tail_offset, backing_size - tail_offset, fd,
|
||||
static_cast<off_t>(tail_offset))) {
|
||||
return false;
|
||||
}
|
||||
backing_base = base;
|
||||
ahb_windows = std::move(buffers);
|
||||
ahb_fds = std::move(buffer_fds);
|
||||
ahb_window_size = window_size;
|
||||
ahb_base = region_base;
|
||||
ahb_bytes = region_size;
|
||||
return true;
|
||||
}
|
||||
|
||||
void MapBackingRange(size_t virtual_offset, size_t host_offset, size_t length, int prot_flags) {
|
||||
while (length > 0) {
|
||||
int map_fd = fd;
|
||||
off_t map_offset = static_cast<off_t>(host_offset);
|
||||
size_t chunk = length;
|
||||
if (host_offset < ahb_base) {
|
||||
chunk = (std::min)(chunk, ahb_base - host_offset);
|
||||
} else if (host_offset < ahb_base + ahb_bytes) {
|
||||
const size_t relative = host_offset - ahb_base;
|
||||
const size_t window = relative / ahb_window_size;
|
||||
const size_t local = relative % ahb_window_size;
|
||||
map_fd = ahb_fds[window];
|
||||
map_offset = static_cast<off_t>(local);
|
||||
chunk = (std::min)(chunk, ahb_window_size - local);
|
||||
}
|
||||
void* const ret = mmap(virtual_base + virtual_offset, chunk, prot_flags,
|
||||
MAP_SHARED | MAP_FIXED, map_fd, map_offset);
|
||||
ASSERT_MSG(ret != MAP_FAILED, "mmap: {}", strerror(errno));
|
||||
virtual_offset += chunk;
|
||||
host_offset += chunk;
|
||||
length -= chunk;
|
||||
}
|
||||
}
|
||||
|
||||
size_t BackingMapCount(size_t host_offset, size_t length) const noexcept {
|
||||
if (length == 0) {
|
||||
return 0;
|
||||
}
|
||||
if (ahb_bytes == 0) {
|
||||
return 1;
|
||||
}
|
||||
size_t count = 0;
|
||||
while (length > 0) {
|
||||
size_t chunk = length;
|
||||
if (host_offset < ahb_base) {
|
||||
chunk = (std::min)(chunk, ahb_base - host_offset);
|
||||
} else if (host_offset < ahb_base + ahb_bytes) {
|
||||
const size_t local = (host_offset - ahb_base) % ahb_window_size;
|
||||
chunk = (std::min)(chunk, ahb_window_size - local);
|
||||
}
|
||||
host_offset += chunk;
|
||||
length -= chunk;
|
||||
++count;
|
||||
}
|
||||
return count;
|
||||
}
|
||||
|
||||
std::span<AHardwareBuffer* const> AhbWindows() const noexcept {
|
||||
return ahb_windows;
|
||||
}
|
||||
|
||||
size_t AhbWindowSize() const noexcept {
|
||||
return ahb_window_size;
|
||||
}
|
||||
|
||||
size_t AhbBase() const noexcept {
|
||||
return ahb_base;
|
||||
}
|
||||
#endif
|
||||
|
||||
~Impl() {
|
||||
Release();
|
||||
}
|
||||
@@ -586,6 +859,12 @@ public:
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
if (True(perms & MemoryPermission::Execute))
|
||||
prot_flags |= PROT_EXEC;
|
||||
#endif
|
||||
#ifdef __ANDROID__
|
||||
if (ahb_bytes != 0) {
|
||||
MapBackingRange(virtual_offset, host_offset, length, prot_flags);
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
int flags = (fd >= 0 ? MAP_SHARED : MAP_PRIVATE) | MAP_FIXED;
|
||||
void* ret = mmap(virtual_base + virtual_offset, length, prot_flags, flags, fd, host_offset);
|
||||
@@ -633,6 +912,7 @@ public:
|
||||
|
||||
const size_t backing_size; ///< Size of the backing memory in bytes
|
||||
const size_t virtual_size; ///< Size of the virtual address placeholder in bytes
|
||||
const size_t preferred_offset;
|
||||
|
||||
u8* backing_base{reinterpret_cast<u8*>(MAP_FAILED)};
|
||||
u8* virtual_base{reinterpret_cast<u8*>(MAP_FAILED)};
|
||||
@@ -655,6 +935,15 @@ private:
|
||||
int ret = close(fd);
|
||||
ASSERT_MSG(ret == 0, "close failed: {}", strerror(errno));
|
||||
}
|
||||
|
||||
#ifdef __ANDROID__
|
||||
for (AHardwareBuffer* buffer : ahb_windows) {
|
||||
AHardwareBuffer_release(buffer);
|
||||
}
|
||||
ahb_windows.clear();
|
||||
ahb_fds.clear();
|
||||
ahb_bytes = 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
void AdjustMap(size_t* virtual_offset, size_t* length) {
|
||||
@@ -680,11 +969,19 @@ private:
|
||||
|
||||
int fd{-1}; // memfd file descriptor, -1 is the error value of memfd_create
|
||||
FreeRegionManager free_manager{};
|
||||
|
||||
#ifdef __ANDROID__
|
||||
std::vector<AHardwareBuffer*> ahb_windows;
|
||||
std::vector<int> ahb_fds;
|
||||
size_t ahb_window_size{};
|
||||
size_t ahb_base{};
|
||||
size_t ahb_bytes{};
|
||||
#endif
|
||||
};
|
||||
|
||||
#endif // ^^^ POSIX ^^^
|
||||
|
||||
HostMemory::HostMemory(size_t backing_size_, size_t virtual_size_)
|
||||
HostMemory::HostMemory(size_t backing_size_, size_t virtual_size_, size_t preferred_offset_)
|
||||
: backing_size(backing_size_)
|
||||
, virtual_size(virtual_size_)
|
||||
{
|
||||
@@ -696,7 +993,9 @@ HostMemory::HostMemory(size_t backing_size_, size_t virtual_size_)
|
||||
#else
|
||||
// Try to allocate a fastmem arena.
|
||||
// The implementation will fail with std::bad_alloc on errors.
|
||||
impl = std::make_unique<HostMemory::Impl>(AlignUp(backing_size, PageAlignment), AlignUp(virtual_size, PageAlignment) + HugePageSize);
|
||||
impl = std::make_unique<HostMemory::Impl>(AlignUp(backing_size, PageAlignment),
|
||||
AlignUp(virtual_size, PageAlignment) + HugePageSize,
|
||||
preferred_offset_);
|
||||
if (impl->Init()) {
|
||||
backing_base = impl->backing_base;
|
||||
virtual_base = impl->virtual_base;
|
||||
@@ -766,6 +1065,42 @@ void HostMemory::ClearBackingRegion(size_t physical_offset, size_t length, u32 f
|
||||
std::memset(backing_base + physical_offset, fill_value, length);
|
||||
}
|
||||
|
||||
std::span<AHardwareBuffer* const> HostMemory::BackingHardwareBuffers() const noexcept {
|
||||
#ifdef __ANDROID__
|
||||
if (impl) {
|
||||
return impl->AhbWindows();
|
||||
}
|
||||
#endif
|
||||
return {};
|
||||
}
|
||||
|
||||
size_t HostMemory::BackingMapCount(size_t host_offset, size_t length) const noexcept {
|
||||
#ifdef __ANDROID__
|
||||
if (impl) {
|
||||
return impl->BackingMapCount(host_offset, length);
|
||||
}
|
||||
#endif
|
||||
return static_cast<size_t>(length != 0);
|
||||
}
|
||||
|
||||
size_t HostMemory::BackingHardwareBufferWindowSize() const noexcept {
|
||||
#ifdef __ANDROID__
|
||||
if (impl) {
|
||||
return impl->AhbWindowSize();
|
||||
}
|
||||
#endif
|
||||
return 0;
|
||||
}
|
||||
|
||||
size_t HostMemory::BackingHardwareBufferBase() const noexcept {
|
||||
#ifdef __ANDROID__
|
||||
if (impl) {
|
||||
return impl->AhbBase();
|
||||
}
|
||||
#endif
|
||||
return 0;
|
||||
}
|
||||
|
||||
void HostMemory::EnableDirectMappedAddress() {
|
||||
#if !(defined(__OPENORBIS__) || defined(__managarm__))
|
||||
if (impl) {
|
||||
|
||||
@@ -8,10 +8,13 @@
|
||||
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include <span>
|
||||
#include "common/common_funcs.h"
|
||||
#include "common/common_types.h"
|
||||
#include "common/virtual_buffer.h"
|
||||
|
||||
struct AHardwareBuffer;
|
||||
|
||||
namespace Common {
|
||||
|
||||
enum class MemoryPermission : u32 {
|
||||
@@ -28,7 +31,7 @@ DECLARE_ENUM_FLAG_OPERATORS(MemoryPermission)
|
||||
*/
|
||||
class HostMemory {
|
||||
public:
|
||||
explicit HostMemory(size_t backing_size_, size_t virtual_size_);
|
||||
explicit HostMemory(size_t backing_size_, size_t virtual_size_, size_t preferred_offset_ = 0);
|
||||
~HostMemory();
|
||||
|
||||
/**
|
||||
@@ -62,6 +65,18 @@ public:
|
||||
return backing_base;
|
||||
}
|
||||
|
||||
[[nodiscard]] size_t BackingSize() const noexcept {
|
||||
return backing_size;
|
||||
}
|
||||
|
||||
[[nodiscard]] size_t BackingMapCount(size_t host_offset, size_t length) const noexcept;
|
||||
|
||||
[[nodiscard]] std::span<AHardwareBuffer* const> BackingHardwareBuffers() const noexcept;
|
||||
|
||||
[[nodiscard]] size_t BackingHardwareBufferWindowSize() const noexcept;
|
||||
|
||||
[[nodiscard]] size_t BackingHardwareBufferBase() const noexcept;
|
||||
|
||||
[[nodiscard]] u8* VirtualBasePointer() noexcept {
|
||||
return virtual_base;
|
||||
}
|
||||
|
||||
@@ -1,3 +1,6 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
@@ -17,6 +20,10 @@
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
#include "common/memory_detect.h"
|
||||
|
||||
namespace Common {
|
||||
@@ -69,4 +76,61 @@ const MemoryInfo& GetMemInfo() {
|
||||
return mem_info;
|
||||
}
|
||||
|
||||
u64 GetPermissibleMapCount() {
|
||||
constexpr u64 DefaultMapCount = 65530;
|
||||
constexpr u64 ReservedMaps = 20000;
|
||||
u64 count = DefaultMapCount;
|
||||
#ifdef __linux__
|
||||
if (std::FILE* const file = std::fopen("/proc/sys/vm/max_map_count", "re")) {
|
||||
char line[32];
|
||||
if (std::fgets(line, sizeof(line), file) != nullptr) {
|
||||
const u64 parsed = std::strtoull(line, nullptr, 10);
|
||||
if (parsed != 0) {
|
||||
count = parsed;
|
||||
}
|
||||
}
|
||||
std::fclose(file);
|
||||
}
|
||||
#endif
|
||||
if (count <= ReservedMaps) {
|
||||
return 0;
|
||||
}
|
||||
return count - ReservedMaps;
|
||||
}
|
||||
|
||||
u64 GetAvailablePhysicalMemory() {
|
||||
#ifdef _WIN32
|
||||
MEMORYSTATUSEX memorystatus;
|
||||
memorystatus.dwLength = sizeof(memorystatus);
|
||||
if (GlobalMemoryStatusEx(&memorystatus) == 0) {
|
||||
return 0;
|
||||
}
|
||||
return memorystatus.ullAvailPhys;
|
||||
#elif defined(__linux__)
|
||||
static constexpr char AvailableKey[] = "MemAvailable:";
|
||||
if (std::FILE* const file = std::fopen("/proc/meminfo", "re")) {
|
||||
char line[256];
|
||||
u64 available = 0;
|
||||
while (std::fgets(line, sizeof(line), file) != nullptr) {
|
||||
if (std::strncmp(line, AvailableKey, sizeof(AvailableKey) - 1) != 0) {
|
||||
continue;
|
||||
}
|
||||
available = std::strtoull(line + sizeof(AvailableKey) - 1, nullptr, 10) * 1024;
|
||||
break;
|
||||
}
|
||||
std::fclose(file);
|
||||
if (available != 0) {
|
||||
return available;
|
||||
}
|
||||
}
|
||||
struct sysinfo meminfo;
|
||||
if (sysinfo(&meminfo) != 0) {
|
||||
return 0;
|
||||
}
|
||||
return static_cast<u64>(meminfo.freeram) * static_cast<u64>(meminfo.mem_unit);
|
||||
#else
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace Common
|
||||
|
||||
@@ -1,3 +1,6 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
@@ -18,4 +21,8 @@ struct MemoryInfo {
|
||||
*/
|
||||
[[nodiscard]] const MemoryInfo& GetMemInfo();
|
||||
|
||||
[[nodiscard]] u64 GetPermissibleMapCount();
|
||||
|
||||
[[nodiscard]] u64 GetAvailablePhysicalMemory();
|
||||
|
||||
} // namespace Common
|
||||
|
||||
@@ -654,6 +654,9 @@ struct Values {
|
||||
SwitchableSetting<bool> use_asynchronous_shaders{linkage, false, "use_asynchronous_shaders",
|
||||
Category::RendererHacks};
|
||||
|
||||
SwitchableSetting<bool> use_unified_memory{linkage, false, "use_unified_memory",
|
||||
Category::RendererHacks};
|
||||
|
||||
SwitchableSetting<ExtendedDynamicState> dyna_state{linkage,
|
||||
#if defined(__ANDROID__)
|
||||
ExtendedDynamicState::Disabled,
|
||||
|
||||
+4
-1
@@ -120,6 +120,7 @@ struct System::Impl {
|
||||
|
||||
is_multicore = Settings::values.use_multi_core.GetValue();
|
||||
extended_memory_layout = Settings::values.memory_layout_mode.GetValue() != Settings::MemoryLayout::Memory_4Gb;
|
||||
unified_memory = Settings::values.use_unified_memory.GetValue();
|
||||
|
||||
core_timing.SetMulticore(is_multicore);
|
||||
core_timing.Initialize([&system]() { system.RegisterHostThread(); });
|
||||
@@ -147,7 +148,8 @@ struct System::Impl {
|
||||
!device_memory.has_value() ||
|
||||
is_multicore != Settings::values.use_multi_core.GetValue() ||
|
||||
extended_memory_layout != (Settings::values.memory_layout_mode.GetValue() !=
|
||||
Settings::MemoryLayout::Memory_4Gb);
|
||||
Settings::MemoryLayout::Memory_4Gb) ||
|
||||
unified_memory != Settings::values.use_unified_memory.GetValue();
|
||||
|
||||
if (!must_reinitialize) {
|
||||
return;
|
||||
@@ -535,6 +537,7 @@ struct System::Impl {
|
||||
std::atomic_bool is_powered_on{};
|
||||
bool is_multicore : 1 = false;
|
||||
bool extended_memory_layout : 1 = false;
|
||||
bool unified_memory : 1 = false;
|
||||
bool exit_locked : 1 = false;
|
||||
bool exit_requested : 1 = false;
|
||||
|
||||
|
||||
@@ -1,3 +1,6 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
@@ -12,9 +15,21 @@ constexpr size_t VirtualReserveSize = 1ULL << 38;
|
||||
constexpr size_t VirtualReserveSize = 1ULL << 39;
|
||||
#endif
|
||||
|
||||
namespace {
|
||||
size_t ApplicationPoolOffset() {
|
||||
using Init = Kernel::Board::Nintendo::Nx::KSystemControl::Init;
|
||||
const size_t dram_size = Init::GetIntendedMemorySize();
|
||||
const size_t application_pool_size = Init::GetApplicationPoolSize();
|
||||
if (dram_size <= application_pool_size) {
|
||||
return 0;
|
||||
}
|
||||
return dram_size - application_pool_size;
|
||||
}
|
||||
}
|
||||
|
||||
DeviceMemory::DeviceMemory()
|
||||
: buffer{Kernel::Board::Nintendo::Nx::KSystemControl::Init::GetIntendedMemorySize(),
|
||||
VirtualReserveSize} {}
|
||||
VirtualReserveSize, ApplicationPoolOffset()} {}
|
||||
|
||||
DeviceMemory::~DeviceMemory() = default;
|
||||
|
||||
|
||||
@@ -20,6 +20,10 @@
|
||||
#include "common/scratch_buffer.h"
|
||||
#include "common/virtual_buffer.h"
|
||||
|
||||
namespace Common {
|
||||
class HostMemory;
|
||||
}
|
||||
|
||||
namespace Core {
|
||||
|
||||
constexpr size_t DEVICE_PAGEBITS = 12ULL;
|
||||
@@ -126,6 +130,10 @@ public:
|
||||
// New batch API to update multiple ranges with a single lock acquisition.
|
||||
void UpdatePagesCachedBatch(std::span<const std::pair<DAddr, size_t>> ranges, s32 delta);
|
||||
|
||||
const Common::HostMemory& GetHostMemory() const noexcept {
|
||||
return host_memory;
|
||||
}
|
||||
|
||||
private:
|
||||
struct TranslationEntry {
|
||||
DAddr guest_page{};
|
||||
@@ -171,6 +179,7 @@ private:
|
||||
std::unique_ptr<DeviceMemoryManagerAllocator<Traits>> impl;
|
||||
|
||||
const uintptr_t physical_base;
|
||||
const Common::HostMemory& host_memory;
|
||||
DeviceInterface* device_inter;
|
||||
|
||||
struct TrackedEntry {
|
||||
|
||||
@@ -171,6 +171,7 @@ struct DeviceMemoryManagerAllocator {
|
||||
template <typename Traits>
|
||||
DeviceMemoryManager<Traits>::DeviceMemoryManager(const DeviceMemory& device_memory_)
|
||||
: physical_base{uintptr_t(device_memory_.buffer.BackingBasePointer())}
|
||||
, host_memory{device_memory_.buffer}
|
||||
, device_inter{nullptr}
|
||||
, compressed_device_addr(1ULL << ((Settings::values.memory_layout_mode.GetValue() == Settings::MemoryLayout::Memory_4Gb ? physical_min_bits : physical_max_bits) - Memory::YUZU_PAGEBITS))
|
||||
, tracked_entries(device_as_size >> Memory::YUZU_PAGEBITS)
|
||||
|
||||
@@ -168,6 +168,8 @@ add_library(video_core STATIC
|
||||
renderer_vulkan/vk_fence_manager.h
|
||||
renderer_vulkan/vk_graphics_pipeline.cpp
|
||||
renderer_vulkan/vk_graphics_pipeline.h
|
||||
renderer_vulkan/vk_guest_memory.cpp
|
||||
renderer_vulkan/vk_guest_memory.h
|
||||
renderer_vulkan/vk_multi_range_buffer.cpp
|
||||
renderer_vulkan/vk_multi_range_buffer.h
|
||||
renderer_vulkan/vk_master_semaphore.cpp
|
||||
|
||||
@@ -591,8 +591,21 @@ void QueriesPrefixScanPass::Run(VkBuffer accumulation_buffer, VkBuffer dst_buffe
|
||||
|
||||
namespace {
|
||||
|
||||
void RecordUnswizzleBeginBarrier(Scheduler& scheduler, VkPipeline vk_pipeline, VkImage vk_image,
|
||||
VkImageAspectFlags aspect_mask, bool is_initialized) {
|
||||
template <typename Func>
|
||||
void RecordUnswizzle(Scheduler& scheduler, bool reorder, Func&& func) {
|
||||
if (reorder) {
|
||||
scheduler.RecordWithUploadBuffer(
|
||||
[func = std::forward<Func>(func)](vk::CommandBuffer, vk::CommandBuffer upload_cmdbuf) {
|
||||
func(upload_cmdbuf);
|
||||
});
|
||||
return;
|
||||
}
|
||||
scheduler.Record(std::forward<Func>(func));
|
||||
}
|
||||
|
||||
void RecordUnswizzleBeginBarrier(Scheduler& scheduler, bool reorder, VkPipeline vk_pipeline,
|
||||
VkImage vk_image, VkImageAspectFlags aspect_mask,
|
||||
bool is_initialized) {
|
||||
VkAccessFlags2 src_access = VK_ACCESS_2_NONE;
|
||||
VkImageLayout old_layout = VK_IMAGE_LAYOUT_UNDEFINED;
|
||||
VkPipelineStageFlags2 src_stage = VK_PIPELINE_STAGE_2_TOP_OF_PIPE_BIT;
|
||||
@@ -601,8 +614,8 @@ void RecordUnswizzleBeginBarrier(Scheduler& scheduler, VkPipeline vk_pipeline, V
|
||||
old_layout = VK_IMAGE_LAYOUT_GENERAL;
|
||||
src_stage = vk::PIPELINE_STAGE_IMAGE_USERS;
|
||||
}
|
||||
scheduler.Record([vk_pipeline, vk_image, aspect_mask, src_access, old_layout,
|
||||
src_stage](vk::CommandBuffer cmdbuf) {
|
||||
RecordUnswizzle(scheduler, reorder, [vk_pipeline, vk_image, aspect_mask, src_access,
|
||||
old_layout, src_stage](vk::CommandBuffer cmdbuf) {
|
||||
const VkImageMemoryBarrier2 image_barrier{
|
||||
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2,
|
||||
.pNext = nullptr,
|
||||
@@ -628,9 +641,9 @@ void RecordUnswizzleBeginBarrier(Scheduler& scheduler, VkPipeline vk_pipeline, V
|
||||
});
|
||||
}
|
||||
|
||||
void RecordUnswizzleEndBarrier(Scheduler& scheduler, VkImage vk_image,
|
||||
void RecordUnswizzleEndBarrier(Scheduler& scheduler, bool reorder, VkImage vk_image,
|
||||
VkImageAspectFlags aspect_mask) {
|
||||
scheduler.Record([vk_image, aspect_mask](vk::CommandBuffer cmdbuf) {
|
||||
RecordUnswizzle(scheduler, reorder, [vk_image, aspect_mask](vk::CommandBuffer cmdbuf) {
|
||||
const VkImageMemoryBarrier2 image_barrier{
|
||||
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER_2,
|
||||
.pNext = nullptr,
|
||||
@@ -672,16 +685,19 @@ ASTCDecoderPass::ASTCDecoderPass(const Device& device_, Scheduler& scheduler_,
|
||||
ASTCDecoderPass::~ASTCDecoderPass() = default;
|
||||
|
||||
void ASTCDecoderPass::Assemble(Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles) {
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles,
|
||||
bool reorder) {
|
||||
using namespace VideoCommon::Accelerated;
|
||||
const std::array<u32, 2> block_dims{
|
||||
VideoCore::Surface::DefaultBlockWidth(image.info.format),
|
||||
VideoCore::Surface::DefaultBlockHeight(image.info.format),
|
||||
};
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
if (!reorder) {
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
}
|
||||
const VkImageAspectFlags aspect_mask = image.AspectMask();
|
||||
const VkImage vk_image = image.Handle();
|
||||
RecordUnswizzleBeginBarrier(scheduler, *pipeline, vk_image, aspect_mask,
|
||||
RecordUnswizzleBeginBarrier(scheduler, reorder, *pipeline, vk_image, aspect_mask,
|
||||
image.ExchangeInitialization());
|
||||
for (const VideoCommon::SwizzleParameters& swizzle : swizzles) {
|
||||
const size_t input_offset = swizzle.buffer_offset + map.offset;
|
||||
@@ -700,8 +716,9 @@ void ASTCDecoderPass::Assemble(Image& image, const StagingBufferRef& map,
|
||||
ASSERT(params.origin == (std::array<u32, 3>{0, 0, 0}));
|
||||
ASSERT(params.destination == (std::array<s32, 3>{0, 0, 0}));
|
||||
ASSERT(params.bytes_per_block_log2 == 4);
|
||||
scheduler.Record([this, num_dispatches_x, num_dispatches_y, num_dispatches_z, block_dims,
|
||||
params, descriptor_data](vk::CommandBuffer cmdbuf) {
|
||||
RecordUnswizzle(scheduler, reorder,
|
||||
[this, num_dispatches_x, num_dispatches_y, num_dispatches_z, block_dims,
|
||||
params, descriptor_data](vk::CommandBuffer cmdbuf) {
|
||||
const AstcPushConstants uniforms{
|
||||
.blocks_dims = block_dims,
|
||||
.layer_stride = params.layer_stride,
|
||||
@@ -717,7 +734,7 @@ void ASTCDecoderPass::Assemble(Image& image, const StagingBufferRef& map,
|
||||
cmdbuf.Dispatch(num_dispatches_x, num_dispatches_y, num_dispatches_z);
|
||||
});
|
||||
}
|
||||
RecordUnswizzleEndBarrier(scheduler, vk_image, aspect_mask);
|
||||
RecordUnswizzleEndBarrier(scheduler, reorder, vk_image, aspect_mask);
|
||||
}
|
||||
|
||||
BlockLinearUnswizzleImage2DPass::BlockLinearUnswizzleImage2DPass(
|
||||
@@ -736,12 +753,14 @@ BlockLinearUnswizzleImage2DPass::~BlockLinearUnswizzleImage2DPass() = default;
|
||||
|
||||
void BlockLinearUnswizzleImage2DPass::Unswizzle(
|
||||
Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles) {
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles, bool reorder) {
|
||||
using namespace VideoCommon::Accelerated;
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
if (!reorder) {
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
}
|
||||
const VkImageAspectFlags aspect_mask = image.AspectMask();
|
||||
const VkImage vk_image = image.Handle();
|
||||
RecordUnswizzleBeginBarrier(scheduler, *pipeline, vk_image, aspect_mask,
|
||||
RecordUnswizzleBeginBarrier(scheduler, reorder, *pipeline, vk_image, aspect_mask,
|
||||
image.ExchangeInitialization());
|
||||
for (const VideoCommon::SwizzleParameters& swizzle : swizzles) {
|
||||
const size_t input_offset = swizzle.buffer_offset + map.offset;
|
||||
@@ -756,8 +775,9 @@ void BlockLinearUnswizzleImage2DPass::Unswizzle(
|
||||
const void* const descriptor_data{compute_pass_descriptor_queue.UpdateData()};
|
||||
|
||||
const auto params = MakeBlockLinearSwizzle2DParams(swizzle, image.info);
|
||||
scheduler.Record([this, num_dispatches_x, num_dispatches_y, num_dispatches_z, params,
|
||||
descriptor_data](vk::CommandBuffer cmdbuf) {
|
||||
RecordUnswizzle(scheduler, reorder,
|
||||
[this, num_dispatches_x, num_dispatches_y, num_dispatches_z, params,
|
||||
descriptor_data](vk::CommandBuffer cmdbuf) {
|
||||
const VkDescriptorSet set = descriptor_allocator.Commit();
|
||||
device.GetLogical().UpdateDescriptorSet(set, *descriptor_template, descriptor_data);
|
||||
cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_COMPUTE, *layout, 0, set, {});
|
||||
@@ -765,7 +785,7 @@ void BlockLinearUnswizzleImage2DPass::Unswizzle(
|
||||
cmdbuf.Dispatch(num_dispatches_x, num_dispatches_y, num_dispatches_z);
|
||||
});
|
||||
}
|
||||
RecordUnswizzleEndBarrier(scheduler, vk_image, aspect_mask);
|
||||
RecordUnswizzleEndBarrier(scheduler, reorder, vk_image, aspect_mask);
|
||||
}
|
||||
|
||||
BlockLinearUnswizzleImage3DPass::BlockLinearUnswizzleImage3DPass(
|
||||
@@ -783,12 +803,14 @@ BlockLinearUnswizzleImage3DPass::~BlockLinearUnswizzleImage3DPass() = default;
|
||||
|
||||
void BlockLinearUnswizzleImage3DPass::Unswizzle(
|
||||
Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles) {
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles, bool reorder) {
|
||||
using namespace VideoCommon::Accelerated;
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
if (!reorder) {
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
}
|
||||
const VkImageAspectFlags aspect_mask = image.AspectMask();
|
||||
const VkImage vk_image = image.Handle();
|
||||
RecordUnswizzleBeginBarrier(scheduler, *pipeline, vk_image, aspect_mask,
|
||||
RecordUnswizzleBeginBarrier(scheduler, reorder, *pipeline, vk_image, aspect_mask,
|
||||
image.ExchangeInitialization());
|
||||
for (const VideoCommon::SwizzleParameters& swizzle : swizzles) {
|
||||
const size_t input_offset = swizzle.buffer_offset + map.offset;
|
||||
@@ -815,8 +837,9 @@ void BlockLinearUnswizzleImage3DPass::Unswizzle(
|
||||
.block_depth = p.block_depth,
|
||||
.block_depth_mask = p.block_depth_mask,
|
||||
};
|
||||
scheduler.Record([this, num_dispatches_x, num_dispatches_y, num_dispatches_z, params,
|
||||
descriptor_data](vk::CommandBuffer cmdbuf) {
|
||||
RecordUnswizzle(scheduler, reorder,
|
||||
[this, num_dispatches_x, num_dispatches_y, num_dispatches_z, params,
|
||||
descriptor_data](vk::CommandBuffer cmdbuf) {
|
||||
const VkDescriptorSet set = descriptor_allocator.Commit();
|
||||
device.GetLogical().UpdateDescriptorSet(set, *descriptor_template, descriptor_data);
|
||||
cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_COMPUTE, *layout, 0, set, {});
|
||||
@@ -824,7 +847,7 @@ void BlockLinearUnswizzleImage3DPass::Unswizzle(
|
||||
cmdbuf.Dispatch(num_dispatches_x, num_dispatches_y, num_dispatches_z);
|
||||
});
|
||||
}
|
||||
RecordUnswizzleEndBarrier(scheduler, vk_image, aspect_mask);
|
||||
RecordUnswizzleEndBarrier(scheduler, reorder, vk_image, aspect_mask);
|
||||
}
|
||||
|
||||
PitchUnswizzlePass::PitchUnswizzlePass(const Device& device_, Scheduler& scheduler_,
|
||||
@@ -841,13 +864,16 @@ PitchUnswizzlePass::PitchUnswizzlePass(const Device& device_, Scheduler& schedul
|
||||
PitchUnswizzlePass::~PitchUnswizzlePass() = default;
|
||||
|
||||
void PitchUnswizzlePass::Unswizzle(Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles) {
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles,
|
||||
bool reorder) {
|
||||
const u32 bytes_per_block = VideoCore::Surface::BytesPerBlock(image.info.format);
|
||||
ASSERT(std::has_single_bit(bytes_per_block));
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
if (!reorder) {
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
}
|
||||
const VkImageAspectFlags aspect_mask = image.AspectMask();
|
||||
const VkImage vk_image = image.Handle();
|
||||
RecordUnswizzleBeginBarrier(scheduler, *pipeline, vk_image, aspect_mask,
|
||||
RecordUnswizzleBeginBarrier(scheduler, reorder, *pipeline, vk_image, aspect_mask,
|
||||
image.ExchangeInitialization());
|
||||
for (const VideoCommon::SwizzleParameters& swizzle : swizzles) {
|
||||
const size_t input_offset = swizzle.buffer_offset + map.offset;
|
||||
@@ -866,8 +892,9 @@ void PitchUnswizzlePass::Unswizzle(Image& image, const StagingBufferRef& map,
|
||||
.bytes_per_block_log2 = static_cast<u32>(std::countr_zero(bytes_per_block)),
|
||||
.pitch = image.info.pitch,
|
||||
};
|
||||
scheduler.Record([this, num_dispatches_x, num_dispatches_y, params,
|
||||
descriptor_data](vk::CommandBuffer cmdbuf) {
|
||||
RecordUnswizzle(scheduler, reorder,
|
||||
[this, num_dispatches_x, num_dispatches_y, params,
|
||||
descriptor_data](vk::CommandBuffer cmdbuf) {
|
||||
const VkDescriptorSet set = descriptor_allocator.Commit();
|
||||
device.GetLogical().UpdateDescriptorSet(set, *descriptor_template, descriptor_data);
|
||||
cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_COMPUTE, *layout, 0, set, {});
|
||||
@@ -875,7 +902,7 @@ void PitchUnswizzlePass::Unswizzle(Image& image, const StagingBufferRef& map,
|
||||
cmdbuf.Dispatch(num_dispatches_x, num_dispatches_y, 1);
|
||||
});
|
||||
}
|
||||
RecordUnswizzleEndBarrier(scheduler, vk_image, aspect_mask);
|
||||
RecordUnswizzleEndBarrier(scheduler, reorder, vk_image, aspect_mask);
|
||||
}
|
||||
|
||||
} // namespace Vulkan
|
||||
|
||||
@@ -144,7 +144,7 @@ public:
|
||||
~ASTCDecoderPass();
|
||||
|
||||
void Assemble(Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles);
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles, bool reorder);
|
||||
|
||||
private:
|
||||
Scheduler& scheduler;
|
||||
@@ -162,7 +162,7 @@ public:
|
||||
~PitchUnswizzlePass();
|
||||
|
||||
void Unswizzle(Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles);
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles, bool reorder);
|
||||
|
||||
private:
|
||||
Scheduler& scheduler;
|
||||
@@ -179,7 +179,7 @@ public:
|
||||
~BlockLinearUnswizzleImage3DPass();
|
||||
|
||||
void Unswizzle(Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles);
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles, bool reorder);
|
||||
|
||||
private:
|
||||
Scheduler& scheduler;
|
||||
@@ -196,7 +196,7 @@ public:
|
||||
~BlockLinearUnswizzleImage2DPass();
|
||||
|
||||
void Unswizzle(Image& image, const StagingBufferRef& map,
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles);
|
||||
std::span<const VideoCommon::SwizzleParameters> swizzles, bool reorder);
|
||||
|
||||
private:
|
||||
Scheduler& scheduler;
|
||||
|
||||
@@ -0,0 +1,176 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <bit>
|
||||
#include <cstring>
|
||||
#include <span>
|
||||
|
||||
#include "common/host_memory.h"
|
||||
#include "video_core/renderer_vulkan/vk_guest_memory.h"
|
||||
#include "video_core/renderer_vulkan/vk_scheduler.h"
|
||||
#include "video_core/vulkan_common/vulkan_device.h"
|
||||
#include "video_core/vulkan_common/vulkan_memory_allocator.h"
|
||||
|
||||
namespace Vulkan {
|
||||
namespace {
|
||||
constexpr size_t PROBE_WORDS = 1024;
|
||||
constexpr size_t PROBE_SIZE = PROBE_WORDS * sizeof(u32);
|
||||
constexpr u32 PROBE_CPU_STRIDE = 0x9E3779B9U;
|
||||
constexpr u32 PROBE_GPU_VALUE = 0x5AC3E10FU;
|
||||
}
|
||||
|
||||
GuestMemory::GuestMemory(const Device& device, MemoryAllocator& memory_allocator,
|
||||
Scheduler& scheduler, const Common::HostMemory& host_memory) {
|
||||
Import(device, host_memory);
|
||||
if (!windows.empty() && !IsCoherent(memory_allocator, scheduler)) {
|
||||
windows.clear();
|
||||
}
|
||||
}
|
||||
|
||||
std::optional<GuestMemory::Range> GuestMemory::Find(const u8* pointer, size_t size) const {
|
||||
if (windows.empty() || pointer < base) {
|
||||
return std::nullopt;
|
||||
}
|
||||
const size_t offset = static_cast<size_t>(pointer - base);
|
||||
const size_t index = offset / window_size;
|
||||
const size_t local_offset = offset % window_size;
|
||||
if (index >= windows.size() || window_size - local_offset < size) {
|
||||
return std::nullopt;
|
||||
}
|
||||
return Range{*windows[index].buffer, local_offset};
|
||||
}
|
||||
|
||||
void GuestMemory::Import([[maybe_unused]] const Device& device,
|
||||
[[maybe_unused]] const Common::HostMemory& host_memory) {
|
||||
#ifdef __ANDROID__
|
||||
const std::span<AHardwareBuffer* const> hardware_buffers =
|
||||
host_memory.BackingHardwareBuffers();
|
||||
window_size = host_memory.BackingHardwareBufferWindowSize();
|
||||
if (hardware_buffers.empty() || window_size == 0 || !device.IsAhbImportSupported()) {
|
||||
return;
|
||||
}
|
||||
base = const_cast<u8*>(host_memory.BackingBasePointer()) +
|
||||
host_memory.BackingHardwareBufferBase();
|
||||
const vk::Device& logical = device.GetLogical();
|
||||
try {
|
||||
for (AHardwareBuffer* const hardware_buffer : hardware_buffers) {
|
||||
const VkAndroidHardwareBufferPropertiesANDROID properties =
|
||||
logical.GetAndroidHardwareBufferProperties(hardware_buffer);
|
||||
const VkExternalMemoryBufferCreateInfo external_info{
|
||||
.sType = VK_STRUCTURE_TYPE_EXTERNAL_MEMORY_BUFFER_CREATE_INFO,
|
||||
.pNext = nullptr,
|
||||
.handleTypes = VK_EXTERNAL_MEMORY_HANDLE_TYPE_ANDROID_HARDWARE_BUFFER_BIT_ANDROID,
|
||||
};
|
||||
Window& window = windows.emplace_back();
|
||||
window.buffer = logical.CreateExternalBuffer(VkBufferCreateInfo{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
|
||||
.pNext = &external_info,
|
||||
.flags = 0,
|
||||
.size = window_size,
|
||||
.usage = VK_BUFFER_USAGE_STORAGE_BUFFER_BIT | VK_BUFFER_USAGE_TRANSFER_SRC_BIT |
|
||||
VK_BUFFER_USAGE_TRANSFER_DST_BIT,
|
||||
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
});
|
||||
const u32 type_bits =
|
||||
logical.GetBufferMemoryRequirements(*window.buffer).memoryTypeBits &
|
||||
properties.memoryTypeBits;
|
||||
if (type_bits == 0) {
|
||||
windows.clear();
|
||||
return;
|
||||
}
|
||||
const VkImportAndroidHardwareBufferInfoANDROID import_info{
|
||||
.sType = VK_STRUCTURE_TYPE_IMPORT_ANDROID_HARDWARE_BUFFER_INFO_ANDROID,
|
||||
.pNext = nullptr,
|
||||
.buffer = hardware_buffer,
|
||||
};
|
||||
const VkMemoryDedicatedAllocateInfo dedicated_info{
|
||||
.sType = VK_STRUCTURE_TYPE_MEMORY_DEDICATED_ALLOCATE_INFO,
|
||||
.pNext = &import_info,
|
||||
.image = VK_NULL_HANDLE,
|
||||
.buffer = *window.buffer,
|
||||
};
|
||||
window.memory = logical.AllocateMemory(VkMemoryAllocateInfo{
|
||||
.sType = VK_STRUCTURE_TYPE_MEMORY_ALLOCATE_INFO,
|
||||
.pNext = &dedicated_info,
|
||||
.allocationSize = properties.allocationSize,
|
||||
.memoryTypeIndex = static_cast<u32>(std::countr_zero(type_bits)),
|
||||
});
|
||||
logical.BindBufferMemory(*window.buffer, *window.memory, 0);
|
||||
}
|
||||
} catch (const vk::Exception&) {
|
||||
windows.clear();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
bool GuestMemory::IsCoherent(MemoryAllocator& memory_allocator, Scheduler& scheduler) const {
|
||||
const VkDeviceSize probe_offset = window_size - PROBE_SIZE;
|
||||
u8* const probe = base + (windows.size() - 1) * window_size + probe_offset;
|
||||
std::array<u8, PROBE_SIZE> saved;
|
||||
std::memcpy(saved.data(), probe, PROBE_SIZE);
|
||||
std::array<u32, PROBE_WORDS> pattern;
|
||||
for (size_t index = 0; index < PROBE_WORDS; ++index) {
|
||||
pattern[index] = static_cast<u32>(index + 1) * PROBE_CPU_STRIDE;
|
||||
}
|
||||
std::memcpy(probe, pattern.data(), PROBE_SIZE);
|
||||
|
||||
vk::Buffer readback = memory_allocator.CreateBuffer(
|
||||
VkBufferCreateInfo{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_CREATE_INFO,
|
||||
.pNext = nullptr,
|
||||
.flags = 0,
|
||||
.size = PROBE_SIZE,
|
||||
.usage = VK_BUFFER_USAGE_TRANSFER_DST_BIT,
|
||||
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
|
||||
.queueFamilyIndexCount = 0,
|
||||
.pQueueFamilyIndices = nullptr,
|
||||
},
|
||||
MemoryUsage::Download);
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
scheduler.Record([src = *windows.back().buffer, dst = *readback,
|
||||
probe_offset](vk::CommandBuffer cmdbuf) {
|
||||
static constexpr VkMemoryBarrier2 READ_BEFORE_WRITE{
|
||||
.sType = VK_STRUCTURE_TYPE_MEMORY_BARRIER_2,
|
||||
.pNext = nullptr,
|
||||
.srcStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT,
|
||||
.srcAccessMask = VK_ACCESS_2_TRANSFER_READ_BIT,
|
||||
.dstStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT,
|
||||
.dstAccessMask = VK_ACCESS_2_TRANSFER_WRITE_BIT,
|
||||
};
|
||||
static constexpr VkMemoryBarrier2 WRITE_TO_HOST{
|
||||
.sType = VK_STRUCTURE_TYPE_MEMORY_BARRIER_2,
|
||||
.pNext = nullptr,
|
||||
.srcStageMask = VK_PIPELINE_STAGE_2_TRANSFER_BIT,
|
||||
.srcAccessMask = VK_ACCESS_2_TRANSFER_WRITE_BIT,
|
||||
.dstStageMask = VK_PIPELINE_STAGE_2_HOST_BIT,
|
||||
.dstAccessMask = VK_ACCESS_2_HOST_READ_BIT,
|
||||
};
|
||||
const VkBufferCopy2 copy{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_COPY_2,
|
||||
.pNext = nullptr,
|
||||
.srcOffset = probe_offset,
|
||||
.dstOffset = 0,
|
||||
.size = PROBE_SIZE,
|
||||
};
|
||||
cmdbuf.CopyBuffer(src, dst, copy);
|
||||
cmdbuf.PipelineBarrier(READ_BEFORE_WRITE);
|
||||
cmdbuf.FillBuffer(src, probe_offset, PROBE_SIZE, PROBE_GPU_VALUE);
|
||||
cmdbuf.PipelineBarrier(WRITE_TO_HOST);
|
||||
});
|
||||
scheduler.Finish();
|
||||
|
||||
const bool gpu_sees_cpu =
|
||||
std::memcmp(readback.Mapped().data(), pattern.data(), PROBE_SIZE) == 0;
|
||||
std::array<u32, PROBE_WORDS> written;
|
||||
std::memcpy(written.data(), probe, PROBE_SIZE);
|
||||
const bool cpu_sees_gpu =
|
||||
std::ranges::all_of(written, [](u32 word) { return word == PROBE_GPU_VALUE; });
|
||||
std::memcpy(probe, saved.data(), PROBE_SIZE);
|
||||
return gpu_sees_cpu && cpu_sees_gpu;
|
||||
}
|
||||
|
||||
}
|
||||
@@ -0,0 +1,53 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <optional>
|
||||
#include <vector>
|
||||
|
||||
#include "common/common_types.h"
|
||||
#include "video_core/vulkan_common/vulkan_wrapper.h"
|
||||
|
||||
namespace Common {
|
||||
class HostMemory;
|
||||
}
|
||||
|
||||
namespace Vulkan {
|
||||
|
||||
class Device;
|
||||
class MemoryAllocator;
|
||||
class Scheduler;
|
||||
|
||||
class GuestMemory {
|
||||
public:
|
||||
struct Range {
|
||||
VkBuffer buffer;
|
||||
VkDeviceSize offset;
|
||||
};
|
||||
|
||||
explicit GuestMemory(const Device& device, MemoryAllocator& memory_allocator,
|
||||
Scheduler& scheduler, const Common::HostMemory& host_memory);
|
||||
|
||||
[[nodiscard]] std::optional<Range> Find(const u8* pointer, size_t size) const;
|
||||
|
||||
[[nodiscard]] bool Empty() const noexcept {
|
||||
return windows.empty();
|
||||
}
|
||||
|
||||
private:
|
||||
struct Window {
|
||||
vk::DeviceMemory memory;
|
||||
vk::ExternalBuffer buffer;
|
||||
};
|
||||
|
||||
void Import(const Device& device, const Common::HostMemory& host_memory);
|
||||
|
||||
[[nodiscard]] bool IsCoherent(MemoryAllocator& memory_allocator, Scheduler& scheduler) const;
|
||||
|
||||
std::vector<Window> windows;
|
||||
u8* base{};
|
||||
size_t window_size{};
|
||||
};
|
||||
|
||||
}
|
||||
@@ -227,6 +227,9 @@ RasterizerVulkan::RasterizerVulkan(Core::Frontend::EmuWindow& emu_window_, Tegra
|
||||
accelerate_dma(buffer_cache, texture_cache, scheduler),
|
||||
fence_manager(*this, gpu, texture_cache, buffer_cache, query_cache, device, scheduler) {
|
||||
scheduler.SetQueryCache(query_cache);
|
||||
if (Settings::values.use_unified_memory.GetValue()) {
|
||||
texture_cache_runtime.ImportGuestMemory(device_memory.GetHostMemory());
|
||||
}
|
||||
}
|
||||
|
||||
RasterizerVulkan::~RasterizerVulkan() {
|
||||
|
||||
@@ -3206,20 +3206,40 @@ void TextureCacheRuntime::AccelerateImageUpload(
|
||||
if (is_rescaled) {
|
||||
image.ScaleDown(true);
|
||||
}
|
||||
const bool reorder = !is_rescaled && True(image.flags & ImageFlagBits::ReorderableUpload);
|
||||
if (IsPixelFormatASTC(image.info.format)) {
|
||||
astc_decoder_pass->Assemble(image, map, swizzles);
|
||||
astc_decoder_pass->Assemble(image, map, swizzles, reorder);
|
||||
} else if (image.info.type == ImageType::e3D) {
|
||||
bl_unswizzle_3d_pass->Unswizzle(image, map, swizzles);
|
||||
bl_unswizzle_3d_pass->Unswizzle(image, map, swizzles, reorder);
|
||||
} else if (image.info.type == ImageType::Linear) {
|
||||
pitch_unswizzle_pass->Unswizzle(image, map, swizzles);
|
||||
pitch_unswizzle_pass->Unswizzle(image, map, swizzles, reorder);
|
||||
} else {
|
||||
bl_unswizzle_2d_pass->Unswizzle(image, map, swizzles);
|
||||
bl_unswizzle_2d_pass->Unswizzle(image, map, swizzles, reorder);
|
||||
}
|
||||
if (is_rescaled) {
|
||||
image.ScaleUp();
|
||||
}
|
||||
}
|
||||
|
||||
void TextureCacheRuntime::ImportGuestMemory(const Common::HostMemory& host_memory) {
|
||||
guest_memory.emplace(device, memory_allocator, scheduler, host_memory);
|
||||
if (guest_memory->Empty()) {
|
||||
guest_memory.reset();
|
||||
}
|
||||
}
|
||||
|
||||
std::optional<StagingBufferRef> TextureCacheRuntime::GuestMemorySource(const u8* pointer,
|
||||
size_t size) const {
|
||||
if (pointer == nullptr) {
|
||||
return std::nullopt;
|
||||
}
|
||||
const std::optional<GuestMemory::Range> range = guest_memory->Find(pointer, size);
|
||||
if (!range || range->offset % device.GetStorageBufferAlignment() != 0) {
|
||||
return std::nullopt;
|
||||
}
|
||||
return StagingBufferRef{.buffer = range->buffer, .offset = range->offset};
|
||||
}
|
||||
|
||||
void TextureCacheRuntime::TransitionImageLayout(Image& image) {
|
||||
if (image.ExchangeInitialization()) {
|
||||
return;
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
|
||||
#include "shader_recompiler/shader_info.h"
|
||||
#include "video_core/renderer_vulkan/vk_compute_pass.h"
|
||||
#include "video_core/renderer_vulkan/vk_guest_memory.h"
|
||||
#include "video_core/renderer_vulkan/vk_render_pass_cache.h"
|
||||
#include "video_core/renderer_vulkan/vk_staging_buffer_pool.h"
|
||||
#include "video_core/texture_cache/image_view_base.h"
|
||||
@@ -93,6 +94,15 @@ public:
|
||||
void AccelerateImageUpload(Image&, const StagingBufferRef&,
|
||||
std::span<const VideoCommon::SwizzleParameters>);
|
||||
|
||||
void ImportGuestMemory(const Common::HostMemory& host_memory);
|
||||
|
||||
[[nodiscard]] bool HasGuestMemory() const noexcept {
|
||||
return guest_memory.has_value();
|
||||
}
|
||||
|
||||
[[nodiscard]] std::optional<StagingBufferRef> GuestMemorySource(const u8* pointer,
|
||||
size_t size) const;
|
||||
|
||||
void InsertUploadMemoryBarrier() {}
|
||||
|
||||
void TransitionImageLayout(Image& image);
|
||||
@@ -153,6 +163,7 @@ public:
|
||||
std::optional<BlockLinearUnswizzleImage2DPass> bl_unswizzle_2d_pass;
|
||||
std::optional<BlockLinearUnswizzleImage3DPass> bl_unswizzle_3d_pass;
|
||||
std::optional<PitchUnswizzlePass> pitch_unswizzle_pass;
|
||||
std::optional<GuestMemory> guest_memory;
|
||||
const Settings::ResolutionScalingInfo& resolution;
|
||||
std::array<std::vector<VkFormat>, VideoCore::Surface::MaxPixelFormat> view_formats;
|
||||
std::bitset<VideoCore::Surface::MaxPixelFormat> host_copy_formats;
|
||||
|
||||
@@ -1033,6 +1033,24 @@ void TextureCache<P>::RefreshContents(Image& image, ImageId image_id) {
|
||||
QueueAsyncDecode(image, image_id);
|
||||
return;
|
||||
}
|
||||
if constexpr (requires { runtime.GuestMemorySource(nullptr, size_t{}); }) {
|
||||
if (True(image.flags & ImageFlagBits::AcceleratedUpload) && runtime.HasGuestMemory() &&
|
||||
gpu_memory->IsContinuousRange(image.gpu_addr, image.guest_size_bytes)) {
|
||||
const std::optional<DAddr> device_addr = gpu_memory->GpuToCpuAddress(image.gpu_addr);
|
||||
if (device_addr) {
|
||||
const auto source = runtime.GuestMemorySource(
|
||||
device_memory.GetSpan(*device_addr, image.guest_size_bytes),
|
||||
image.guest_size_bytes);
|
||||
if (source) {
|
||||
gpu_memory->FlushRegion(image.gpu_addr, image.guest_size_bytes,
|
||||
VideoCommon::CacheType::NoTextureCache);
|
||||
runtime.AccelerateImageUpload(
|
||||
image, *source, FixSmallVectorADL(FullUploadSwizzles(image.info)));
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
if constexpr (requires { image.UploadHostMemory(unswizzle_data_buffer, {}); }) {
|
||||
if (True(image.flags & ImageFlagBits::ReorderableUpload) && image.CanUploadHostMemory()) {
|
||||
Tegra::Memory::GpuGuestMemory<u8, Tegra::Memory::GuestMemoryFlags::UnsafeRead>
|
||||
|
||||
@@ -934,6 +934,15 @@ bool Device::GetSuitability(bool requires_swapchain) {
|
||||
|
||||
FOR_EACH_VK_FEATURE_EXT(FEATURE_EXTENSION);
|
||||
FOR_EACH_VK_EXTENSION(EXTENSION);
|
||||
#ifdef __ANDROID__
|
||||
if (supported_extensions.contains(
|
||||
VK_ANDROID_EXTERNAL_MEMORY_ANDROID_HARDWARE_BUFFER_EXTENSION_NAME) &&
|
||||
supported_extensions.contains(VK_EXT_QUEUE_FAMILY_FOREIGN_EXTENSION_NAME)) {
|
||||
loaded_extensions.insert(VK_ANDROID_EXTERNAL_MEMORY_ANDROID_HARDWARE_BUFFER_EXTENSION_NAME);
|
||||
loaded_extensions.insert(VK_EXT_QUEUE_FAMILY_FOREIGN_EXTENSION_NAME);
|
||||
has_ahb_import = true;
|
||||
}
|
||||
#endif
|
||||
|
||||
extensions.depth_stencil_resolve =
|
||||
extensions.depth_stencil_resolve &&
|
||||
|
||||
@@ -993,6 +993,10 @@ FN_MAX_LIMIT_LIST
|
||||
return extensions.host_image_copy;
|
||||
}
|
||||
|
||||
bool IsAhbImportSupported() const {
|
||||
return has_ahb_import;
|
||||
}
|
||||
|
||||
bool HasNullDescriptor() const {
|
||||
return features.robustness2.nullDescriptor;
|
||||
}
|
||||
@@ -1208,6 +1212,7 @@ private:
|
||||
bool is_warp_potentially_bigger{}; ///< Host warp size can be bigger than guest.
|
||||
bool is_integrated{}; ///< Is GPU an iGPU.
|
||||
bool is_uma{};
|
||||
bool has_ahb_import{};
|
||||
bool has_broken_compute{}; ///< Compute shaders can cause crashes
|
||||
bool has_broken_cube_compatibility{}; ///< Has broken cube compatibility bit
|
||||
bool has_broken_float16_math{};
|
||||
|
||||
@@ -85,6 +85,12 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
|
||||
X(vkAcquireNextImageKHR);
|
||||
X(vkAllocateCommandBuffers);
|
||||
X(vkAllocateDescriptorSets);
|
||||
X(vkAllocateMemory);
|
||||
X(vkBindBufferMemory);
|
||||
X(vkFreeMemory);
|
||||
#ifdef __ANDROID__
|
||||
X(vkGetAndroidHardwareBufferPropertiesANDROID);
|
||||
#endif
|
||||
X(vkBeginCommandBuffer);
|
||||
X(vkCmdBeginConditionalRenderingEXT);
|
||||
X(vkCmdBeginQuery);
|
||||
@@ -337,6 +343,10 @@ void Destroy(VkDevice device, VkBuffer handle, const DeviceDispatch& dld) noexce
|
||||
dld.vkDestroyBuffer(device, handle, nullptr);
|
||||
}
|
||||
|
||||
void Destroy(VkDevice device, VkDeviceMemory handle, const DeviceDispatch& dld) noexcept {
|
||||
dld.vkFreeMemory(device, handle, nullptr);
|
||||
}
|
||||
|
||||
void Destroy(VkDevice device, VkBufferView handle, const DeviceDispatch& dld) noexcept {
|
||||
dld.vkDestroyBufferView(device, handle, nullptr);
|
||||
}
|
||||
@@ -616,6 +626,51 @@ BufferView Device::CreateBufferView(const VkBufferViewCreateInfo& ci) const {
|
||||
return BufferView(object, handle, *dld);
|
||||
}
|
||||
|
||||
ExternalBuffer Device::CreateExternalBuffer(const VkBufferCreateInfo& ci) const {
|
||||
VkBuffer object;
|
||||
Check(dld->vkCreateBuffer(handle, &ci, nullptr, &object));
|
||||
return ExternalBuffer(object, handle, *dld);
|
||||
}
|
||||
|
||||
DeviceMemory Device::AllocateMemory(const VkMemoryAllocateInfo& ai) const {
|
||||
VkDeviceMemory object;
|
||||
Check(dld->vkAllocateMemory(handle, &ai, nullptr, &object));
|
||||
return DeviceMemory(object, handle, *dld);
|
||||
}
|
||||
|
||||
void Device::BindBufferMemory(VkBuffer buffer, VkDeviceMemory memory, VkDeviceSize offset) const {
|
||||
Check(dld->vkBindBufferMemory(handle, buffer, memory, offset));
|
||||
}
|
||||
|
||||
VkMemoryRequirements Device::GetBufferMemoryRequirements(VkBuffer buffer) const noexcept {
|
||||
const VkBufferMemoryRequirementsInfo2 info{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_REQUIREMENTS_INFO_2,
|
||||
.pNext = nullptr,
|
||||
.buffer = buffer,
|
||||
};
|
||||
VkMemoryRequirements2 requirements{
|
||||
.sType = VK_STRUCTURE_TYPE_MEMORY_REQUIREMENTS_2,
|
||||
.pNext = nullptr,
|
||||
.memoryRequirements = {},
|
||||
};
|
||||
dld->vkGetBufferMemoryRequirements2(handle, &info, &requirements);
|
||||
return requirements.memoryRequirements;
|
||||
}
|
||||
|
||||
#ifdef __ANDROID__
|
||||
VkAndroidHardwareBufferPropertiesANDROID Device::GetAndroidHardwareBufferProperties(
|
||||
const AHardwareBuffer* buffer) const {
|
||||
VkAndroidHardwareBufferPropertiesANDROID properties{
|
||||
.sType = VK_STRUCTURE_TYPE_ANDROID_HARDWARE_BUFFER_PROPERTIES_ANDROID,
|
||||
.pNext = nullptr,
|
||||
.allocationSize = 0,
|
||||
.memoryTypeBits = 0,
|
||||
};
|
||||
Check(dld->vkGetAndroidHardwareBufferPropertiesANDROID(handle, buffer, &properties));
|
||||
return properties;
|
||||
}
|
||||
#endif
|
||||
|
||||
ImageView Device::CreateImageView(const VkImageViewCreateInfo& ci) const {
|
||||
VkImageView object;
|
||||
Check(dld->vkCreateImageView(handle, &ci, nullptr, &object));
|
||||
|
||||
@@ -234,6 +234,12 @@ struct DeviceDispatch : InstanceDispatch {
|
||||
PFN_vkAcquireNextImageKHR vkAcquireNextImageKHR{};
|
||||
PFN_vkAllocateCommandBuffers vkAllocateCommandBuffers{};
|
||||
PFN_vkAllocateDescriptorSets vkAllocateDescriptorSets{};
|
||||
PFN_vkAllocateMemory vkAllocateMemory{};
|
||||
PFN_vkBindBufferMemory vkBindBufferMemory{};
|
||||
PFN_vkFreeMemory vkFreeMemory{};
|
||||
#ifdef __ANDROID__
|
||||
PFN_vkGetAndroidHardwareBufferPropertiesANDROID vkGetAndroidHardwareBufferPropertiesANDROID{};
|
||||
#endif
|
||||
PFN_vkBeginCommandBuffer vkBeginCommandBuffer{};
|
||||
PFN_vkCmdBeginConditionalRenderingEXT vkCmdBeginConditionalRenderingEXT{};
|
||||
PFN_vkCmdBeginQuery vkCmdBeginQuery{};
|
||||
@@ -387,6 +393,7 @@ void Destroy(VkDevice, const InstanceDispatch&) noexcept;
|
||||
|
||||
void Destroy(VkDevice, VkBuffer, const DeviceDispatch&) noexcept;
|
||||
void Destroy(VkDevice, VkBufferView, const DeviceDispatch&) noexcept;
|
||||
void Destroy(VkDevice, VkDeviceMemory, const DeviceDispatch&) noexcept;
|
||||
void Destroy(VkDevice, VkCommandPool, const DeviceDispatch&) noexcept;
|
||||
void Destroy(VkDevice, VkDescriptorPool, const DeviceDispatch&) noexcept;
|
||||
void Destroy(VkDevice, VkDescriptorSetLayout, const DeviceDispatch&) noexcept;
|
||||
@@ -646,6 +653,8 @@ using Pipeline = Handle<VkPipeline, VkDevice, DeviceDispatch>;
|
||||
using PipelineLayout = Handle<VkPipelineLayout, VkDevice, DeviceDispatch>;
|
||||
using QueryPool = Handle<VkQueryPool, VkDevice, DeviceDispatch>;
|
||||
using Sampler = Handle<VkSampler, VkDevice, DeviceDispatch>;
|
||||
using DeviceMemory = Handle<VkDeviceMemory, VkDevice, DeviceDispatch>;
|
||||
using ExternalBuffer = Handle<VkBuffer, VkDevice, DeviceDispatch>;
|
||||
using SurfaceKHR = Handle<VkSurfaceKHR, VkInstance, InstanceDispatch>;
|
||||
|
||||
using DescriptorSets = PoolAllocations<VkDescriptorSet, VkDescriptorPool>;
|
||||
@@ -1000,6 +1009,19 @@ public:
|
||||
|
||||
[[nodiscard]] BufferView CreateBufferView(const VkBufferViewCreateInfo& ci) const;
|
||||
|
||||
[[nodiscard]] ExternalBuffer CreateExternalBuffer(const VkBufferCreateInfo& ci) const;
|
||||
|
||||
[[nodiscard]] DeviceMemory AllocateMemory(const VkMemoryAllocateInfo& ai) const;
|
||||
|
||||
void BindBufferMemory(VkBuffer buffer, VkDeviceMemory memory, VkDeviceSize offset) const;
|
||||
|
||||
[[nodiscard]] VkMemoryRequirements GetBufferMemoryRequirements(VkBuffer buffer) const noexcept;
|
||||
|
||||
#ifdef __ANDROID__
|
||||
[[nodiscard]] VkAndroidHardwareBufferPropertiesANDROID GetAndroidHardwareBufferProperties(
|
||||
const AHardwareBuffer* buffer) const;
|
||||
#endif
|
||||
|
||||
[[nodiscard]] ImageView CreateImageView(const VkImageViewCreateInfo& ci) const;
|
||||
|
||||
[[nodiscard]] Semaphore CreateSemaphore() const;
|
||||
|
||||
Reference in New Issue
Block a user