mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-09-29 11:53:24 +00:00
Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| 3cfe9f04e1 | |||
| a074268eb0 |
@@ -218,8 +218,6 @@ object NativeLibrary {
|
||||
|
||||
external fun logSettings()
|
||||
|
||||
external fun refreshThreadPolicies()
|
||||
|
||||
external fun getDebugKnobAt(index: Int): Boolean
|
||||
|
||||
/**
|
||||
|
||||
@@ -1451,7 +1451,6 @@ class EmulationFragment : Fragment(), SurfaceHolder.Callback {
|
||||
|
||||
override fun onResume() {
|
||||
super.onResume()
|
||||
NativeLibrary.refreshThreadPolicies()
|
||||
val b = _binding ?: return
|
||||
updateStatsPosition(IntSetting.PERF_OVERLAY_POSITION.getInt())
|
||||
updateSocPosition(IntSetting.SOC_OVERLAY_POSITION.getInt())
|
||||
|
||||
@@ -50,7 +50,6 @@ extern "C" {
|
||||
#include "common/scope_exit.h"
|
||||
#include "common/settings.h"
|
||||
#include "common/string_util.h"
|
||||
#include "common/thread.h"
|
||||
#include "frontend_common/play_time_manager.h"
|
||||
#include "core/constants.h"
|
||||
#include "core/core.h"
|
||||
@@ -1183,10 +1182,6 @@ void Java_org_yuzu_yuzu_1emu_NativeLibrary_logSettings(JNIEnv* env, jobject jobj
|
||||
Settings::LogSettings();
|
||||
}
|
||||
|
||||
void Java_org_yuzu_yuzu_1emu_NativeLibrary_refreshThreadPolicies(JNIEnv* env, jobject jobj) {
|
||||
Common::RefreshThreadPolicies();
|
||||
}
|
||||
|
||||
jboolean Java_org_yuzu_yuzu_1emu_NativeLibrary_getDebugKnobAt(JNIEnv* env, jobject jobj, jint index) {
|
||||
return static_cast<jboolean>(Settings::getDebugKnobAt(static_cast<u8>(index)));
|
||||
}
|
||||
|
||||
+129
-200
@@ -78,9 +78,15 @@ struct NativeHandle {
|
||||
};
|
||||
|
||||
using PFN_AHardwareBuffer_getNativeHandle = const NativeHandle* (*)(const AHardwareBuffer*);
|
||||
using PFN_AHardwareBuffer_isSupported = int (*)(const AHardwareBuffer_Desc*);
|
||||
|
||||
void* NativeWindowLibrary() {
|
||||
static void* const lib = dlopen("libnativewindow.so", RTLD_NOW);
|
||||
return lib;
|
||||
}
|
||||
|
||||
PFN_AHardwareBuffer_getNativeHandle ResolveGetNativeHandle() {
|
||||
void* const lib = dlopen("libnativewindow.so", RTLD_NOW);
|
||||
void* const lib = NativeWindowLibrary();
|
||||
if (lib == nullptr) {
|
||||
return nullptr;
|
||||
}
|
||||
@@ -88,6 +94,15 @@ PFN_AHardwareBuffer_getNativeHandle ResolveGetNativeHandle() {
|
||||
dlsym(lib, "AHardwareBuffer_getNativeHandle"));
|
||||
}
|
||||
|
||||
PFN_AHardwareBuffer_isSupported ResolveIsSupported() {
|
||||
void* const lib = NativeWindowLibrary();
|
||||
if (lib == nullptr) {
|
||||
return nullptr;
|
||||
}
|
||||
return reinterpret_cast<PFN_AHardwareBuffer_isSupported>(
|
||||
dlsym(lib, "AHardwareBuffer_isSupported"));
|
||||
}
|
||||
|
||||
} // namespace
|
||||
#endif
|
||||
|
||||
@@ -160,7 +175,7 @@ static void GetFuncAddress(Common::DynamicLibrary& dll, const char* name, T& pfn
|
||||
|
||||
class HostMemory::Impl {
|
||||
public:
|
||||
explicit Impl(size_t backing_size_, size_t virtual_size_, size_t)
|
||||
explicit Impl(size_t backing_size_, size_t virtual_size_)
|
||||
: backing_size{backing_size_}
|
||||
, virtual_size{virtual_size_}
|
||||
, process{GetCurrentProcess()}
|
||||
@@ -266,10 +281,6 @@ public:
|
||||
UNREACHABLE();
|
||||
}
|
||||
|
||||
bool IsBackingShared() const noexcept {
|
||||
return true;
|
||||
}
|
||||
|
||||
const size_t backing_size; ///< Size of the backing memory in bytes
|
||||
const size_t virtual_size; ///< Size of the virtual address placeholder in bytes
|
||||
|
||||
@@ -542,15 +553,19 @@ static int shm_open_anon(int flags, mode_t mode) {
|
||||
|
||||
class HostMemory::Impl {
|
||||
public:
|
||||
explicit Impl(size_t backing_size_, size_t virtual_size_, size_t preferred_offset_)
|
||||
explicit Impl(size_t backing_size_, size_t virtual_size_)
|
||||
: backing_size{backing_size_}
|
||||
, virtual_size{virtual_size_}
|
||||
, preferred_offset{preferred_offset_}
|
||||
{}
|
||||
|
||||
bool Init() {
|
||||
long page_size = sysconf(_SC_PAGESIZE);
|
||||
ASSERT_MSG(page_size == 0x1000, "page size {:#x} is incompatible with 4K paging", page_size);
|
||||
#ifdef __ANDROID__
|
||||
if (InitAhbBacking()) {
|
||||
return InitVirtual();
|
||||
}
|
||||
#endif
|
||||
// Backing memory initialization
|
||||
#if defined(__sun__) || defined(__HAIKU__) || defined(__NetBSD__) || defined(__DragonFly__)
|
||||
fd = shm_open_anon(O_RDWR | O_CREAT | O_EXCL | O_NOFOLLOW, 0600);
|
||||
@@ -585,15 +600,10 @@ public:
|
||||
LOG_WARNING(Common_Memory, "Using private mappings instead of shared ones");
|
||||
backing_base = static_cast<u8*>(mmap(nullptr, backing_size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE, -1, 0));
|
||||
if (fd > 0) {
|
||||
fd = -1;
|
||||
close(fd);
|
||||
}
|
||||
fd = -1;
|
||||
} else {
|
||||
#ifdef __ANDROID__
|
||||
if (InitAhbBacking()) {
|
||||
return InitVirtual();
|
||||
}
|
||||
#endif
|
||||
backing_base = static_cast<u8*>(mmap(nullptr, backing_size, PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0));
|
||||
}
|
||||
if (backing_base == MAP_FAILED) {
|
||||
@@ -670,46 +680,6 @@ public:
|
||||
return ok;
|
||||
}
|
||||
|
||||
size_t ComputeAhbBudget(size_t window_size) const {
|
||||
const u64 total_physical = Common::GetMemInfo().TotalPhysicalMemory;
|
||||
if (total_physical == 0) {
|
||||
LOG_WARNING(HW_Memory, "Host memory size is unknown, not committing hardware buffers");
|
||||
return 0;
|
||||
}
|
||||
constexpr u64 MinimumTotalPhysical = 7ULL << 30;
|
||||
if (total_physical < MinimumTotalPhysical) {
|
||||
LOG_INFO(HW_Memory,
|
||||
"Skipping hardware buffer backing, {} MiB of RAM is below the {} MiB minimum",
|
||||
total_physical >> 20, MinimumTotalPhysical >> 20);
|
||||
return 0;
|
||||
}
|
||||
const u64 max_map_count = Common::GetMaxMapCount();
|
||||
constexpr u64 ReservedMaps = 24576;
|
||||
if (max_map_count == 0 || max_map_count <= ReservedMaps) {
|
||||
LOG_WARNING(HW_Memory,
|
||||
"Skipping hardware buffer backing, vm.max_map_count is unknown or too low");
|
||||
return 0;
|
||||
}
|
||||
u64 budget = total_physical / 6;
|
||||
budget = (std::min)(budget, (max_map_count - ReservedMaps) * PageAlignment);
|
||||
const u64 available = Common::GetAvailablePhysicalMemory();
|
||||
if (available != 0) {
|
||||
constexpr u64 Headroom = 2ULL << 30;
|
||||
budget = (std::min)(budget, available > Headroom ? available - Headroom : 0);
|
||||
}
|
||||
budget = (std::min)(budget, static_cast<u64>(backing_size));
|
||||
budget = Common::AlignDown(budget, window_size);
|
||||
constexpr u64 MinimumBudget = 256ULL << 20;
|
||||
if (budget < MinimumBudget) {
|
||||
LOG_INFO(HW_Memory,
|
||||
"Skipping hardware buffer backing, only {} MiB could be committed on a {} MiB "
|
||||
"system with {} MiB available and vm.max_map_count {}",
|
||||
budget >> 20, total_physical >> 20, available >> 20, max_map_count);
|
||||
return 0;
|
||||
}
|
||||
return static_cast<size_t>(budget);
|
||||
}
|
||||
|
||||
bool InitAhbBacking() {
|
||||
if (!Settings::values.use_unified_memory.GetValue()) {
|
||||
return false;
|
||||
@@ -720,130 +690,106 @@ public:
|
||||
LOG_WARNING(HW_Memory, "AHardwareBuffer_getNativeHandle is not available");
|
||||
return false;
|
||||
}
|
||||
constexpr size_t window_size = 64ULL << 20;
|
||||
const AHardwareBuffer_Desc window_desc = MakeBlobDesc(window_size);
|
||||
if (AHardwareBuffer_isSupported(&window_desc) == 0) {
|
||||
LOG_WARNING(HW_Memory, "Allocator rejects {} MiB hardware buffer windows",
|
||||
window_size >> 20);
|
||||
static const PFN_AHardwareBuffer_isSupported is_supported = ResolveIsSupported();
|
||||
if (is_supported == nullptr) {
|
||||
LOG_WARNING(HW_Memory, "AHardwareBuffer_isSupported is not available");
|
||||
return false;
|
||||
}
|
||||
const size_t budget = ComputeAhbBudget(window_size);
|
||||
if (budget == 0) {
|
||||
const u64 total_physical = Common::GetMemInfo().TotalPhysicalMemory;
|
||||
if (total_physical != 0 && backing_size > total_physical / 2) {
|
||||
LOG_WARNING(HW_Memory,
|
||||
"Hardware buffer backing would commit {} MiB on a {} MiB system, keeping "
|
||||
"lazily committed memory",
|
||||
backing_size >> 20, total_physical >> 20);
|
||||
return false;
|
||||
}
|
||||
if (!ProbeAhbBacking(get_native_handle)) {
|
||||
return false;
|
||||
}
|
||||
const size_t aligned_backing = Common::AlignDown(backing_size, window_size);
|
||||
const size_t region_size = (std::min)(budget, aligned_backing);
|
||||
const size_t region_base = Common::AlignDown(
|
||||
(std::min)(preferred_offset, aligned_backing - region_size), window_size);
|
||||
const size_t num_windows = region_size / window_size;
|
||||
|
||||
std::vector<AHardwareBuffer*> buffers;
|
||||
std::vector<int> buffer_fds;
|
||||
const auto cleanup = [&] {
|
||||
for (AHardwareBuffer* buffer : buffers) {
|
||||
AHardwareBuffer_release(buffer);
|
||||
const auto try_window_size = [&](size_t window_size) -> bool {
|
||||
const size_t num_windows = (backing_size + window_size - 1) / window_size;
|
||||
std::vector<AHardwareBuffer*> buffers;
|
||||
std::vector<int> buffer_fds;
|
||||
const auto cleanup = [&] {
|
||||
for (AHardwareBuffer* buffer : buffers) {
|
||||
AHardwareBuffer_release(buffer);
|
||||
}
|
||||
buffers.clear();
|
||||
buffer_fds.clear();
|
||||
};
|
||||
for (size_t i = 0; i < num_windows; ++i) {
|
||||
const size_t len = (std::min)(window_size, backing_size - i * window_size);
|
||||
const AHardwareBuffer_Desc desc = MakeBlobDesc(len);
|
||||
AHardwareBuffer* buffer{};
|
||||
if (AHardwareBuffer_allocate(&desc, &buffer) != 0 || buffer == nullptr) {
|
||||
LOG_WARNING(HW_Memory, "Hardware buffer allocation failed for window {}", i);
|
||||
cleanup();
|
||||
return false;
|
||||
}
|
||||
buffers.push_back(buffer);
|
||||
const NativeHandle* const handle = get_native_handle(buffer);
|
||||
if (handle == nullptr || handle->numFds < 1) {
|
||||
LOG_WARNING(HW_Memory, "Hardware buffer has no mappable file descriptor");
|
||||
cleanup();
|
||||
return false;
|
||||
}
|
||||
const int buffer_fd = handle->data[0];
|
||||
const off_t buffer_len = lseek(buffer_fd, 0, SEEK_END);
|
||||
if (buffer_len < static_cast<off_t>(len)) {
|
||||
LOG_WARNING(HW_Memory, "Hardware buffer descriptor smaller than requested");
|
||||
cleanup();
|
||||
return false;
|
||||
}
|
||||
buffer_fds.push_back(buffer_fd);
|
||||
}
|
||||
buffers.clear();
|
||||
buffer_fds.clear();
|
||||
};
|
||||
for (size_t i = 0; i < num_windows; ++i) {
|
||||
const AHardwareBuffer_Desc desc = MakeBlobDesc(window_size);
|
||||
AHardwareBuffer* buffer{};
|
||||
if (AHardwareBuffer_allocate(&desc, &buffer) != 0 || buffer == nullptr) {
|
||||
LOG_WARNING(HW_Memory, "Hardware buffer allocation failed for window {} of {}", i,
|
||||
num_windows);
|
||||
u8* const base =
|
||||
static_cast<u8*>(mmap(nullptr, backing_size, PROT_NONE,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1, 0));
|
||||
if (base == MAP_FAILED) {
|
||||
cleanup();
|
||||
return false;
|
||||
}
|
||||
buffers.push_back(buffer);
|
||||
const NativeHandle* const handle = get_native_handle(buffer);
|
||||
if (handle == nullptr || handle->numFds < 1) {
|
||||
LOG_WARNING(HW_Memory, "Hardware buffer has no mappable file descriptor");
|
||||
cleanup();
|
||||
return false;
|
||||
}
|
||||
const int buffer_fd = handle->data[0];
|
||||
const off_t buffer_len = lseek(buffer_fd, 0, SEEK_END);
|
||||
if (buffer_len < static_cast<off_t>(window_size)) {
|
||||
LOG_WARNING(HW_Memory, "Hardware buffer descriptor smaller than requested");
|
||||
cleanup();
|
||||
return false;
|
||||
}
|
||||
buffer_fds.push_back(buffer_fd);
|
||||
}
|
||||
u8* const base = static_cast<u8*>(mmap(nullptr, backing_size, PROT_NONE,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1, 0));
|
||||
if (base == MAP_FAILED) {
|
||||
LOG_WARNING(HW_Memory, "Failed to reserve backing address space: {}", strerror(errno));
|
||||
cleanup();
|
||||
return false;
|
||||
}
|
||||
const auto map_over_reservation = [&](size_t offset, size_t len, int map_fd,
|
||||
off_t map_offset) {
|
||||
if (len == 0) {
|
||||
return true;
|
||||
}
|
||||
if (mmap(base + offset, len, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_FIXED, map_fd,
|
||||
map_offset) == MAP_FAILED) {
|
||||
LOG_WARNING(HW_Memory, "Backing mmap failed: {}", strerror(errno));
|
||||
munmap(base, backing_size);
|
||||
cleanup();
|
||||
return false;
|
||||
for (size_t i = 0; i < num_windows; ++i) {
|
||||
const size_t len = (std::min)(window_size, backing_size - i * window_size);
|
||||
if (mmap(base + i * window_size, len, PROT_READ | PROT_WRITE,
|
||||
MAP_SHARED | MAP_FIXED, buffer_fds[i], 0) == MAP_FAILED) {
|
||||
LOG_WARNING(HW_Memory, "Hardware buffer mmap failed: {}", strerror(errno));
|
||||
munmap(base, backing_size);
|
||||
cleanup();
|
||||
return false;
|
||||
}
|
||||
}
|
||||
backing_base = base;
|
||||
ahb_windows = std::move(buffers);
|
||||
ahb_fds = std::move(buffer_fds);
|
||||
ahb_window_size = window_size;
|
||||
ahb_backing = true;
|
||||
committed_backing_size.store(backing_size, std::memory_order_relaxed);
|
||||
LOG_INFO(HW_Memory,
|
||||
"Guest memory backed by {} hardware buffer windows of {} MiB, {} MiB committed",
|
||||
ahb_windows.size(), window_size >> 20, backing_size >> 20);
|
||||
return true;
|
||||
};
|
||||
if (!map_over_reservation(0, region_base, fd, 0)) {
|
||||
return false;
|
||||
}
|
||||
for (size_t i = 0; i < num_windows; ++i) {
|
||||
if (!map_over_reservation(region_base + i * window_size, window_size, buffer_fds[i],
|
||||
0)) {
|
||||
return false;
|
||||
static constexpr size_t candidate_window_sizes[] = {
|
||||
1024ULL << 20,
|
||||
512ULL << 20,
|
||||
256ULL << 20,
|
||||
128ULL << 20,
|
||||
};
|
||||
for (const size_t candidate : candidate_window_sizes) {
|
||||
const AHardwareBuffer_Desc window_desc = MakeBlobDesc(candidate);
|
||||
if (is_supported(&window_desc) == 0) {
|
||||
LOG_DEBUG(HW_Memory, "Allocator rejects {} MiB hardware buffer windows",
|
||||
candidate >> 20);
|
||||
continue;
|
||||
}
|
||||
}
|
||||
const size_t tail_offset = region_base + region_size;
|
||||
if (!map_over_reservation(tail_offset, backing_size - tail_offset, fd,
|
||||
static_cast<off_t>(tail_offset))) {
|
||||
return false;
|
||||
}
|
||||
backing_base = base;
|
||||
ahb_windows = std::move(buffers);
|
||||
ahb_fds = std::move(buffer_fds);
|
||||
ahb_window_size = window_size;
|
||||
ahb_base = region_base;
|
||||
ahb_bytes = region_size;
|
||||
committed_backing_size.store(region_size, std::memory_order_relaxed);
|
||||
LOG_INFO(HW_Memory,
|
||||
"Guest memory {:#x}-{:#x} backed by {} hardware buffer windows, {} MiB committed",
|
||||
region_base, region_base + region_size, ahb_windows.size(), region_size >> 20);
|
||||
return true;
|
||||
}
|
||||
|
||||
void MapBackingRange(size_t virtual_offset, size_t host_offset, size_t length, int prot_flags) {
|
||||
while (length > 0) {
|
||||
int map_fd = fd;
|
||||
off_t map_offset = static_cast<off_t>(host_offset);
|
||||
size_t chunk = length;
|
||||
if (host_offset < ahb_base) {
|
||||
chunk = (std::min)(chunk, ahb_base - host_offset);
|
||||
} else if (host_offset < ahb_base + ahb_bytes) {
|
||||
const size_t relative = host_offset - ahb_base;
|
||||
const size_t window = relative / ahb_window_size;
|
||||
const size_t local = relative % ahb_window_size;
|
||||
map_fd = ahb_fds[window];
|
||||
map_offset = static_cast<off_t>(local);
|
||||
chunk = (std::min)(chunk, ahb_window_size - local);
|
||||
if (try_window_size(candidate)) {
|
||||
return true;
|
||||
}
|
||||
void* const ret = mmap(virtual_base + virtual_offset, chunk, prot_flags,
|
||||
MAP_SHARED | MAP_FIXED, map_fd, map_offset);
|
||||
ASSERT_MSG(ret != MAP_FAILED, "mmap: {}", strerror(errno));
|
||||
virtual_offset += chunk;
|
||||
host_offset += chunk;
|
||||
length -= chunk;
|
||||
LOG_WARNING(HW_Memory, "Could not back guest memory with {} MiB windows",
|
||||
candidate >> 20);
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
std::span<AHardwareBuffer* const> AhbWindows() const noexcept {
|
||||
@@ -851,11 +797,7 @@ public:
|
||||
}
|
||||
|
||||
size_t AhbWindowSize() const noexcept {
|
||||
return ahb_bytes != 0 ? ahb_window_size : 0;
|
||||
}
|
||||
|
||||
size_t AhbBase() const noexcept {
|
||||
return ahb_base;
|
||||
return ahb_backing ? ahb_window_size : 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
@@ -881,8 +823,22 @@ public:
|
||||
prot_flags |= PROT_EXEC;
|
||||
#endif
|
||||
#ifdef __ANDROID__
|
||||
if (ahb_bytes != 0) {
|
||||
MapBackingRange(virtual_offset, host_offset, length, prot_flags);
|
||||
if (ahb_backing) {
|
||||
size_t voff = virtual_offset;
|
||||
size_t hoff = host_offset;
|
||||
size_t remaining = length;
|
||||
while (remaining > 0) {
|
||||
const size_t window = hoff / ahb_window_size;
|
||||
const size_t local = hoff % ahb_window_size;
|
||||
const size_t chunk = (std::min)(remaining, ahb_window_size - local);
|
||||
void* const ret =
|
||||
mmap(virtual_base + voff, chunk, prot_flags, MAP_SHARED | MAP_FIXED,
|
||||
ahb_fds[window], static_cast<off_t>(local));
|
||||
ASSERT_MSG(ret != MAP_FAILED, "mmap: {}", strerror(errno));
|
||||
voff += chunk;
|
||||
hoff += chunk;
|
||||
remaining -= chunk;
|
||||
}
|
||||
return;
|
||||
}
|
||||
#endif
|
||||
@@ -930,18 +886,8 @@ public:
|
||||
virtual_base = nullptr;
|
||||
}
|
||||
|
||||
bool IsBackingShared() const noexcept {
|
||||
#ifdef __ANDROID__
|
||||
if (ahb_bytes != 0) {
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
return fd >= 0;
|
||||
}
|
||||
|
||||
const size_t backing_size; ///< Size of the backing memory in bytes
|
||||
const size_t virtual_size; ///< Size of the virtual address placeholder in bytes
|
||||
const size_t preferred_offset;
|
||||
|
||||
u8* backing_base{reinterpret_cast<u8*>(MAP_FAILED)};
|
||||
u8* virtual_base{reinterpret_cast<u8*>(MAP_FAILED)};
|
||||
@@ -971,9 +917,9 @@ private:
|
||||
}
|
||||
ahb_windows.clear();
|
||||
ahb_fds.clear();
|
||||
if (ahb_bytes != 0) {
|
||||
if (ahb_backing) {
|
||||
committed_backing_size.store(0, std::memory_order_relaxed);
|
||||
ahb_bytes = 0;
|
||||
ahb_backing = false;
|
||||
}
|
||||
#endif
|
||||
}
|
||||
@@ -1003,17 +949,16 @@ private:
|
||||
FreeRegionManager free_manager{};
|
||||
|
||||
#ifdef __ANDROID__
|
||||
bool ahb_backing{};
|
||||
std::vector<AHardwareBuffer*> ahb_windows;
|
||||
std::vector<int> ahb_fds;
|
||||
size_t ahb_window_size{};
|
||||
size_t ahb_base{};
|
||||
size_t ahb_bytes{};
|
||||
#endif
|
||||
};
|
||||
|
||||
#endif // ^^^ POSIX ^^^
|
||||
|
||||
HostMemory::HostMemory(size_t backing_size_, size_t virtual_size_, size_t preferred_offset_)
|
||||
HostMemory::HostMemory(size_t backing_size_, size_t virtual_size_)
|
||||
: backing_size(backing_size_)
|
||||
, virtual_size(virtual_size_)
|
||||
{
|
||||
@@ -1025,7 +970,7 @@ HostMemory::HostMemory(size_t backing_size_, size_t virtual_size_, size_t prefer
|
||||
#else
|
||||
// Try to allocate a fastmem arena.
|
||||
// The implementation will fail with std::bad_alloc on errors.
|
||||
impl = std::make_unique<HostMemory::Impl>(AlignUp(backing_size, PageAlignment), AlignUp(virtual_size, PageAlignment) + HugePageSize, preferred_offset_);
|
||||
impl = std::make_unique<HostMemory::Impl>(AlignUp(backing_size, PageAlignment), AlignUp(virtual_size, PageAlignment) + HugePageSize);
|
||||
if (impl->Init()) {
|
||||
backing_base = impl->backing_base;
|
||||
virtual_base = impl->virtual_base;
|
||||
@@ -1111,22 +1056,6 @@ size_t HostMemory::BackingHardwareBufferWindowSize() const noexcept {
|
||||
#endif
|
||||
}
|
||||
|
||||
bool HostMemory::IsBackingShared() const noexcept {
|
||||
#if defined(__OPENORBIS__) || defined(__managarm__)
|
||||
return false;
|
||||
#else
|
||||
return impl && impl->IsBackingShared();
|
||||
#endif
|
||||
}
|
||||
|
||||
size_t HostMemory::BackingHardwareBufferBase() const noexcept {
|
||||
#ifdef __ANDROID__
|
||||
return impl ? impl->AhbBase() : 0;
|
||||
#else
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
void HostMemory::EnableDirectMappedAddress() {
|
||||
#if !(defined(__OPENORBIS__) || defined(__managarm__))
|
||||
if (impl) {
|
||||
|
||||
@@ -33,7 +33,7 @@ DECLARE_ENUM_FLAG_OPERATORS(MemoryPermission)
|
||||
*/
|
||||
class HostMemory {
|
||||
public:
|
||||
explicit HostMemory(size_t backing_size_, size_t virtual_size_, size_t preferred_offset_ = 0);
|
||||
explicit HostMemory(size_t backing_size_, size_t virtual_size_);
|
||||
~HostMemory();
|
||||
|
||||
/**
|
||||
@@ -75,10 +75,6 @@ public:
|
||||
|
||||
[[nodiscard]] size_t BackingHardwareBufferWindowSize() const noexcept;
|
||||
|
||||
[[nodiscard]] size_t BackingHardwareBufferBase() const noexcept;
|
||||
|
||||
[[nodiscard]] bool IsBackingShared() const noexcept;
|
||||
|
||||
[[nodiscard]] u8* VirtualBasePointer() noexcept {
|
||||
return virtual_base;
|
||||
}
|
||||
|
||||
@@ -17,10 +17,6 @@
|
||||
#endif
|
||||
#endif
|
||||
|
||||
#include <cstdio>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
|
||||
#include "common/memory_detect.h"
|
||||
|
||||
namespace Common {
|
||||
@@ -73,55 +69,4 @@ const MemoryInfo& GetMemInfo() {
|
||||
return mem_info;
|
||||
}
|
||||
|
||||
u64 GetAvailablePhysicalMemory() {
|
||||
#ifdef _WIN32
|
||||
MEMORYSTATUSEX memorystatus;
|
||||
memorystatus.dwLength = sizeof(memorystatus);
|
||||
if (GlobalMemoryStatusEx(&memorystatus)) {
|
||||
return memorystatus.ullAvailPhys;
|
||||
}
|
||||
return 0;
|
||||
#elif defined(__linux__)
|
||||
if (std::FILE* const file = std::fopen("/proc/meminfo", "re")) {
|
||||
char line[256];
|
||||
u64 available = 0;
|
||||
while (std::fgets(line, sizeof(line), file) != nullptr) {
|
||||
if (std::strncmp(line, "MemAvailable:", 13) == 0) {
|
||||
available = std::strtoull(line + 13, nullptr, 10) * 1024ULL;
|
||||
break;
|
||||
}
|
||||
}
|
||||
std::fclose(file);
|
||||
if (available != 0) {
|
||||
return available;
|
||||
}
|
||||
}
|
||||
struct sysinfo info;
|
||||
if (sysinfo(&info) == 0) {
|
||||
const u64 unit = info.mem_unit != 0 ? info.mem_unit : 1ULL;
|
||||
return (static_cast<u64>(info.freeram) + static_cast<u64>(info.bufferram)) * unit;
|
||||
}
|
||||
return 0;
|
||||
#else
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
u64 GetMaxMapCount() {
|
||||
#ifdef __linux__
|
||||
if (std::FILE* const file = std::fopen("/proc/sys/vm/max_map_count", "re")) {
|
||||
char line[32];
|
||||
u64 count = 0;
|
||||
if (std::fgets(line, sizeof(line), file) != nullptr) {
|
||||
count = std::strtoull(line, nullptr, 10);
|
||||
}
|
||||
std::fclose(file);
|
||||
return count;
|
||||
}
|
||||
return 0;
|
||||
#else
|
||||
return 0;
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace Common
|
||||
|
||||
@@ -18,8 +18,4 @@ struct MemoryInfo {
|
||||
*/
|
||||
[[nodiscard]] const MemoryInfo& GetMemInfo();
|
||||
|
||||
[[nodiscard]] u64 GetAvailablePhysicalMemory();
|
||||
|
||||
[[nodiscard]] u64 GetMaxMapCount();
|
||||
|
||||
} // namespace Common
|
||||
|
||||
+75
-228
@@ -43,251 +43,103 @@
|
||||
#ifdef __ANDROID__
|
||||
#include <sys/resource.h>
|
||||
#include <algorithm>
|
||||
#include <cstdlib>
|
||||
#include <cstring>
|
||||
#include <fstream>
|
||||
#include <mutex>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
namespace {
|
||||
constexpr int ANDROID_THREAD_PRIORITY_AUDIO = -16;
|
||||
constexpr int ANDROID_THREAD_PRIORITY_URGENT_DISPLAY = -8;
|
||||
constexpr int ANDROID_THREAD_PRIORITY_DISPLAY = -4;
|
||||
constexpr int ANDROID_THREAD_PRIORITY_DEFAULT = 0;
|
||||
constexpr int ANDROID_THREAD_PRIORITY_BACKGROUND = 10;
|
||||
[[maybe_unused]] constexpr int ANDROID_THREAD_PRIORITY_URGENT_AUDIO = -19;
|
||||
[[maybe_unused]] constexpr int ANDROID_THREAD_PRIORITY_AUDIO = -16;
|
||||
[[maybe_unused]] constexpr int ANDROID_THREAD_PRIORITY_URGENT_DISPLAY = -8;
|
||||
[[maybe_unused]] constexpr int ANDROID_THREAD_PRIORITY_DISPLAY = -4;
|
||||
[[maybe_unused]] constexpr int ANDROID_THREAD_PRIORITY_FOREGROUND = -2;
|
||||
[[maybe_unused]] constexpr int ANDROID_THREAD_PRIORITY_MORE_FAVORABLE = -1;
|
||||
[[maybe_unused]] constexpr int ANDROID_THREAD_PRIORITY_DEFAULT = 0;
|
||||
[[maybe_unused]] constexpr int ANDROID_THREAD_PRIORITY_LESS_FAVORABLE = 1;
|
||||
[[maybe_unused]] constexpr int ANDROID_THREAD_PRIORITY_BACKGROUND = 10;
|
||||
[[maybe_unused]] constexpr int ANDROID_THREAD_PRIORITY_LOWEST = 19;
|
||||
|
||||
constexpr size_t ANDROID_MINIMUM_PERFORMANCE_CORES = 4;
|
||||
|
||||
enum class CoreGroup {
|
||||
Unrestricted,
|
||||
Performance,
|
||||
Efficiency,
|
||||
};
|
||||
cpu_set_t ComputePerformanceCoreMask() {
|
||||
cpu_set_t mask;
|
||||
CPU_ZERO(&mask);
|
||||
|
||||
struct CoreTopology {
|
||||
cpu_set_t allowed;
|
||||
cpu_set_t performance;
|
||||
cpu_set_t efficiency;
|
||||
bool separated;
|
||||
bool initialized;
|
||||
};
|
||||
|
||||
struct ThreadPolicy {
|
||||
pid_t tid;
|
||||
CoreGroup group;
|
||||
int nice_value;
|
||||
bool has_nice;
|
||||
};
|
||||
|
||||
std::mutex g_topology_mutex;
|
||||
CoreTopology g_topology{};
|
||||
|
||||
std::mutex g_policy_mutex;
|
||||
|
||||
std::vector<ThreadPolicy>& Policies() {
|
||||
static auto* const policies = new std::vector<ThreadPolicy>();
|
||||
return *policies;
|
||||
}
|
||||
|
||||
struct PolicyRegistration {
|
||||
~PolicyRegistration() {
|
||||
const pid_t tid = gettid();
|
||||
std::scoped_lock lock{g_policy_mutex};
|
||||
std::erase_if(Policies(), [tid](const ThreadPolicy& policy) { return policy.tid == tid; });
|
||||
CPU_ZERO(&allowed);
|
||||
if (sched_getaffinity(gettid(), sizeof(allowed), &allowed) != 0) {
|
||||
return mask;
|
||||
}
|
||||
};
|
||||
|
||||
thread_local PolicyRegistration t_policy_registration;
|
||||
|
||||
int PossibleCpuCount() {
|
||||
std::ifstream file("/sys/devices/system/cpu/possible");
|
||||
std::string list;
|
||||
if (file && std::getline(file, list) && !list.empty()) {
|
||||
int highest = -1;
|
||||
const char* cursor = list.c_str();
|
||||
while (*cursor != '\0') {
|
||||
char* end = nullptr;
|
||||
const long value = std::strtol(cursor, &end, 10);
|
||||
if (end == cursor) {
|
||||
break;
|
||||
}
|
||||
highest = (std::max)(highest, static_cast<int>(value));
|
||||
cursor = end;
|
||||
while (*cursor == '-' || *cursor == ',') {
|
||||
++cursor;
|
||||
}
|
||||
}
|
||||
if (highest >= 0) {
|
||||
return (std::min)(highest + 1, CPU_SETSIZE);
|
||||
}
|
||||
}
|
||||
const long configured = sysconf(_SC_NPROCESSORS_CONF);
|
||||
if (configured > 0) {
|
||||
return static_cast<int>((std::min<long>)(configured, CPU_SETSIZE));
|
||||
}
|
||||
return static_cast<int>((std::min<unsigned>)(std::thread::hardware_concurrency(), CPU_SETSIZE));
|
||||
}
|
||||
|
||||
long ReadCpuScalar(int cpu, const char* node) {
|
||||
long value = 0;
|
||||
std::ifstream file("/sys/devices/system/cpu/cpu" + std::to_string(cpu) + "/" + node);
|
||||
if (!file || !(file >> value) || value <= 0) {
|
||||
return 0;
|
||||
}
|
||||
return value;
|
||||
}
|
||||
|
||||
std::vector<std::pair<long, int>> CollectCoreWeights(const cpu_set_t& allowed, int total,
|
||||
const char* node, bool require_all) {
|
||||
std::vector<std::pair<long, int>> cores;
|
||||
const int total = static_cast<int>(std::thread::hardware_concurrency());
|
||||
for (int cpu = 0; cpu < total; ++cpu) {
|
||||
if (!CPU_ISSET(cpu, &allowed)) {
|
||||
continue;
|
||||
}
|
||||
const long weight = ReadCpuScalar(cpu, node);
|
||||
if (weight <= 0) {
|
||||
if (require_all) {
|
||||
return {};
|
||||
}
|
||||
LOG_WARNING(Common, "Could not read {} for CPU {}, treating it as an efficiency core",
|
||||
node, cpu);
|
||||
continue;
|
||||
long max_frequency = 0;
|
||||
std::ifstream file("/sys/devices/system/cpu/cpu" + std::to_string(cpu) +
|
||||
"/cpufreq/cpuinfo_max_freq");
|
||||
if (!file || !(file >> max_frequency) || max_frequency <= 0) {
|
||||
CPU_ZERO(&mask);
|
||||
return mask;
|
||||
}
|
||||
cores.emplace_back(weight, cpu);
|
||||
}
|
||||
return cores;
|
||||
}
|
||||
|
||||
void ComputeTopologyLocked() {
|
||||
g_topology.initialized = true;
|
||||
g_topology.separated = false;
|
||||
CPU_ZERO(&g_topology.allowed);
|
||||
CPU_ZERO(&g_topology.performance);
|
||||
CPU_ZERO(&g_topology.efficiency);
|
||||
|
||||
if (sched_getaffinity(getpid(), sizeof(g_topology.allowed), &g_topology.allowed) != 0) {
|
||||
LOG_WARNING(Common, "Could not query process CPU affinity: {}",
|
||||
::Common::GetLastErrorMsg());
|
||||
return;
|
||||
}
|
||||
|
||||
const int total = PossibleCpuCount();
|
||||
auto cores = CollectCoreWeights(g_topology.allowed, total, "cpu_capacity", true);
|
||||
if (cores.empty()) {
|
||||
cores = CollectCoreWeights(g_topology.allowed, total, "cpufreq/cpuinfo_max_freq", false);
|
||||
cores.emplace_back(max_frequency, cpu);
|
||||
}
|
||||
if (cores.empty()) {
|
||||
LOG_WARNING(Common, "Could not determine CPU topology, thread placement is disabled");
|
||||
return;
|
||||
return mask;
|
||||
}
|
||||
|
||||
std::sort(cores.begin(), cores.end(),
|
||||
[](const auto& lhs, const auto& rhs) { return lhs.first > rhs.first; });
|
||||
|
||||
const size_t allowed_count = static_cast<size_t>(CPU_COUNT(&g_topology.allowed));
|
||||
const size_t maximum =
|
||||
allowed_count > 2 * ANDROID_MINIMUM_PERFORMANCE_CORES
|
||||
? allowed_count - ANDROID_MINIMUM_PERFORMANCE_CORES
|
||||
: ANDROID_MINIMUM_PERFORMANCE_CORES;
|
||||
|
||||
size_t taken = 0;
|
||||
long cluster_weight = cores.front().first;
|
||||
for (const auto& [weight, cpu] : cores) {
|
||||
if (weight != cluster_weight) {
|
||||
long cluster_frequency = cores.front().first;
|
||||
for (const auto& [frequency, cpu] : cores) {
|
||||
if (frequency != cluster_frequency) {
|
||||
if (taken >= ANDROID_MINIMUM_PERFORMANCE_CORES) {
|
||||
break;
|
||||
}
|
||||
cluster_weight = weight;
|
||||
cluster_frequency = frequency;
|
||||
}
|
||||
if (taken >= maximum) {
|
||||
break;
|
||||
}
|
||||
CPU_SET(cpu, &g_topology.performance);
|
||||
CPU_SET(cpu, &mask);
|
||||
++taken;
|
||||
}
|
||||
if (taken == 0) {
|
||||
return;
|
||||
return mask;
|
||||
}
|
||||
|
||||
const cpu_set_t& PerformanceCoreMask() {
|
||||
static const cpu_set_t mask = ComputePerformanceCoreMask();
|
||||
return mask;
|
||||
}
|
||||
|
||||
cpu_set_t ComputeEfficiencyCoreMask() {
|
||||
cpu_set_t mask;
|
||||
CPU_ZERO(&mask);
|
||||
|
||||
const cpu_set_t& performance = PerformanceCoreMask();
|
||||
if (CPU_COUNT(&performance) == 0) {
|
||||
return mask;
|
||||
}
|
||||
|
||||
cpu_set_t allowed;
|
||||
CPU_ZERO(&allowed);
|
||||
if (sched_getaffinity(gettid(), sizeof(allowed), &allowed) != 0) {
|
||||
return mask;
|
||||
}
|
||||
|
||||
const int total = static_cast<int>(std::thread::hardware_concurrency());
|
||||
for (int cpu = 0; cpu < total; ++cpu) {
|
||||
if (CPU_ISSET(cpu, &g_topology.allowed) && !CPU_ISSET(cpu, &g_topology.performance)) {
|
||||
CPU_SET(cpu, &g_topology.efficiency);
|
||||
if (CPU_ISSET(cpu, &allowed) && !CPU_ISSET(cpu, &performance)) {
|
||||
CPU_SET(cpu, &mask);
|
||||
}
|
||||
}
|
||||
|
||||
g_topology.separated = CPU_COUNT(&g_topology.efficiency) > 0;
|
||||
LOG_INFO(Common, "CPU topology: {} performance cores, {} efficiency cores, separation {}",
|
||||
CPU_COUNT(&g_topology.performance), CPU_COUNT(&g_topology.efficiency),
|
||||
g_topology.separated ? "enabled" : "unavailable");
|
||||
return mask;
|
||||
}
|
||||
|
||||
void EnsureTopologyLocked() {
|
||||
if (!g_topology.initialized) {
|
||||
ComputeTopologyLocked();
|
||||
}
|
||||
}
|
||||
|
||||
void RefreshTopologyLocked() {
|
||||
if (!g_topology.initialized) {
|
||||
ComputeTopologyLocked();
|
||||
return;
|
||||
}
|
||||
cpu_set_t current;
|
||||
CPU_ZERO(¤t);
|
||||
if (sched_getaffinity(getpid(), sizeof(current), ¤t) != 0) {
|
||||
return;
|
||||
}
|
||||
if (std::memcmp(¤t, &g_topology.allowed, sizeof(current)) != 0) {
|
||||
ComputeTopologyLocked();
|
||||
}
|
||||
}
|
||||
|
||||
bool ApplyCoreGroupLocked(pid_t tid, CoreGroup group) {
|
||||
if (!g_topology.separated || group == CoreGroup::Unrestricted) {
|
||||
return false;
|
||||
}
|
||||
const cpu_set_t& mask =
|
||||
group == CoreGroup::Performance ? g_topology.performance : g_topology.efficiency;
|
||||
if (CPU_COUNT(&mask) == 0) {
|
||||
return false;
|
||||
}
|
||||
if (sched_setaffinity(tid, sizeof(mask), &mask) != 0) {
|
||||
LOG_WARNING(Common, "Could not restrict thread {} to its core group: {}", tid,
|
||||
::Common::GetLastErrorMsg());
|
||||
return false;
|
||||
}
|
||||
return true;
|
||||
}
|
||||
|
||||
ThreadPolicy& AcquirePolicyLocked(pid_t tid) {
|
||||
auto& policies = Policies();
|
||||
for (auto& policy : policies) {
|
||||
if (policy.tid == tid) {
|
||||
return policy;
|
||||
}
|
||||
}
|
||||
return policies.emplace_back(ThreadPolicy{tid, CoreGroup::Unrestricted, 0, false});
|
||||
}
|
||||
|
||||
void SetCurrentThreadCoreGroup(CoreGroup group) {
|
||||
const pid_t tid = gettid();
|
||||
{
|
||||
std::scoped_lock lock{g_topology_mutex};
|
||||
EnsureTopologyLocked();
|
||||
ApplyCoreGroupLocked(tid, group);
|
||||
}
|
||||
(void)&t_policy_registration;
|
||||
std::scoped_lock lock{g_policy_mutex};
|
||||
AcquirePolicyLocked(tid).group = group;
|
||||
}
|
||||
|
||||
void RememberCurrentThreadNice(pid_t tid, int nice_value) {
|
||||
(void)&t_policy_registration;
|
||||
std::scoped_lock lock{g_policy_mutex};
|
||||
ThreadPolicy& policy = AcquirePolicyLocked(tid);
|
||||
policy.nice_value = nice_value;
|
||||
policy.has_nice = true;
|
||||
const cpu_set_t& EfficiencyCoreMask() {
|
||||
static const cpu_set_t mask = ComputeEfficiencyCoreMask();
|
||||
return mask;
|
||||
}
|
||||
} // Anonymous namespace
|
||||
#endif
|
||||
@@ -301,6 +153,7 @@ void RememberCurrentThreadNice(pid_t tid, int nice_value) {
|
||||
#endif
|
||||
#include "common/x64/rdtsc.h"
|
||||
#endif
|
||||
#include "core/core_timing.h"
|
||||
|
||||
namespace Common {
|
||||
|
||||
@@ -341,13 +194,10 @@ void SetCurrentThreadPriority(ThreadPriority new_priority) {
|
||||
default: return ANDROID_THREAD_PRIORITY_DEFAULT;
|
||||
}
|
||||
}();
|
||||
const pid_t tid = gettid();
|
||||
if (setpriority(PRIO_PROCESS, static_cast<id_t>(tid), nice_value) != 0) {
|
||||
LOG_WARNING(Common, "Could not set thread nice value to {}: {}", nice_value,
|
||||
GetLastErrorMsg());
|
||||
return;
|
||||
if (setpriority(PRIO_PROCESS, static_cast<id_t>(gettid()), nice_value) != 0) {
|
||||
LOG_DEBUG(Common, "Could not set thread nice value to {}: {}", nice_value,
|
||||
GetLastErrorMsg());
|
||||
}
|
||||
RememberCurrentThreadNice(tid, nice_value);
|
||||
#else
|
||||
pthread_t this_thread = pthread_self();
|
||||
const auto scheduling_type = SCHED_OTHER;
|
||||
@@ -404,27 +254,24 @@ void SetCurrentThreadName(const char* name) {
|
||||
|
||||
void SetCurrentThreadToPerformanceCores() {
|
||||
#if defined(__ANDROID__)
|
||||
SetCurrentThreadCoreGroup(CoreGroup::Performance);
|
||||
const cpu_set_t& mask = PerformanceCoreMask();
|
||||
if (CPU_COUNT(&mask) == 0) {
|
||||
return;
|
||||
}
|
||||
if (sched_setaffinity(gettid(), sizeof(mask), &mask) != 0) {
|
||||
LOG_DEBUG(Common, "Could not restrict thread to performance cores: {}", GetLastErrorMsg());
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
void SetCurrentThreadToEfficiencyCores() {
|
||||
#if defined(__ANDROID__)
|
||||
SetCurrentThreadCoreGroup(CoreGroup::Efficiency);
|
||||
#endif
|
||||
}
|
||||
|
||||
void RefreshThreadPolicies() {
|
||||
#if defined(__ANDROID__)
|
||||
std::scoped_lock topology_lock{g_topology_mutex};
|
||||
RefreshTopologyLocked();
|
||||
|
||||
std::scoped_lock policy_lock{g_policy_mutex};
|
||||
for (const auto& policy : Policies()) {
|
||||
if (policy.has_nice) {
|
||||
setpriority(PRIO_PROCESS, static_cast<id_t>(policy.tid), policy.nice_value);
|
||||
}
|
||||
ApplyCoreGroupLocked(policy.tid, policy.group);
|
||||
const cpu_set_t& mask = EfficiencyCoreMask();
|
||||
if (CPU_COUNT(&mask) == 0) {
|
||||
return;
|
||||
}
|
||||
if (sched_setaffinity(gettid(), sizeof(mask), &mask) != 0) {
|
||||
LOG_DEBUG(Common, "Could not restrict thread to efficiency cores: {}", GetLastErrorMsg());
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
@@ -102,13 +102,11 @@ enum class ThreadPriority : u32 {
|
||||
enum class ThreadPlacement : u32 {
|
||||
Default = 0,
|
||||
Background = 1,
|
||||
Efficiency = 2,
|
||||
};
|
||||
|
||||
void SetCurrentThreadPriority(ThreadPriority new_priority);
|
||||
void SetCurrentThreadName(const char* name);
|
||||
void SetCurrentThreadToPerformanceCores();
|
||||
void SetCurrentThreadToEfficiencyCores();
|
||||
void RefreshThreadPolicies();
|
||||
|
||||
} // namespace Common
|
||||
|
||||
@@ -42,10 +42,8 @@ public:
|
||||
: workers_queued{num_workers}, thread_name{std::move(name)} {
|
||||
const auto lambda = [this, func, placement](std::stop_token stop_token) {
|
||||
Common::SetCurrentThreadName(thread_name.c_str());
|
||||
if (placement != ThreadPlacement::Default) {
|
||||
if (placement == ThreadPlacement::Background) {
|
||||
Common::SetCurrentThreadPriority(ThreadPriority::Low);
|
||||
}
|
||||
if (placement == ThreadPlacement::Efficiency) {
|
||||
Common::SetCurrentThreadToEfficiencyCores();
|
||||
}
|
||||
{
|
||||
|
||||
@@ -58,8 +58,7 @@ void CoreTiming::Initialize(std::function<void()>&& on_thread_init_) {
|
||||
if (is_multicore) {
|
||||
timer_thread = std::jthread([this](std::stop_token stop_token) {
|
||||
Common::SetCurrentThreadName("HostTiming");
|
||||
Common::SetCurrentThreadPriority(Common::ThreadPriority::VeryHigh);
|
||||
Common::SetCurrentThreadToPerformanceCores();
|
||||
Common::SetCurrentThreadPriority(Common::ThreadPriority::High);
|
||||
on_thread_init();
|
||||
has_started = true;
|
||||
|
||||
|
||||
@@ -12,18 +12,9 @@ constexpr size_t VirtualReserveSize = 1ULL << 38;
|
||||
constexpr size_t VirtualReserveSize = 1ULL << 39;
|
||||
#endif
|
||||
|
||||
namespace {
|
||||
size_t ApplicationPoolOffset() {
|
||||
using Init = Kernel::Board::Nintendo::Nx::KSystemControl::Init;
|
||||
const size_t dram_size = Init::GetIntendedMemorySize();
|
||||
const size_t application_pool_size = Init::GetApplicationPoolSize();
|
||||
return dram_size > application_pool_size ? dram_size - application_pool_size : 0;
|
||||
}
|
||||
}
|
||||
|
||||
DeviceMemory::DeviceMemory()
|
||||
: buffer{Kernel::Board::Nintendo::Nx::KSystemControl::Init::GetIntendedMemorySize(),
|
||||
VirtualReserveSize, ApplicationPoolOffset()} {}
|
||||
VirtualReserveSize} {}
|
||||
|
||||
DeviceMemory::~DeviceMemory() = default;
|
||||
|
||||
|
||||
@@ -117,14 +117,6 @@ public:
|
||||
return ahb_window_size;
|
||||
}
|
||||
|
||||
size_t GetBackingHardwareBufferBase() const noexcept {
|
||||
return ahb_base;
|
||||
}
|
||||
|
||||
bool IsBackingShared() const noexcept {
|
||||
return backing_is_shared;
|
||||
}
|
||||
|
||||
PAddr GetPhysicalRawAddressFromDAddr(DAddr address) const {
|
||||
PAddr subbits = PAddr(address & page_mask);
|
||||
auto paddr = tracked_entries[(address >> page_bits)].compressed_physical_ptr;
|
||||
@@ -208,8 +200,6 @@ private:
|
||||
const size_t physical_size;
|
||||
const std::span<AHardwareBuffer* const> ahb_windows;
|
||||
const size_t ahb_window_size;
|
||||
const size_t ahb_base;
|
||||
const bool backing_is_shared;
|
||||
DeviceInterface* device_inter;
|
||||
|
||||
struct TrackedEntry {
|
||||
|
||||
@@ -174,8 +174,6 @@ DeviceMemoryManager<Traits>::DeviceMemoryManager(const DeviceMemory& device_memo
|
||||
, physical_size{device_memory_.buffer.BackingSize()}
|
||||
, ahb_windows{device_memory_.buffer.BackingHardwareBuffers()}
|
||||
, ahb_window_size{device_memory_.buffer.BackingHardwareBufferWindowSize()}
|
||||
, ahb_base{device_memory_.buffer.BackingHardwareBufferBase()}
|
||||
, backing_is_shared{device_memory_.buffer.IsBackingShared()}
|
||||
, device_inter{nullptr}
|
||||
, compressed_device_addr(1ULL << ((Settings::values.memory_layout_mode.GetValue() == Settings::MemoryLayout::Memory_4Gb ? physical_min_bits : physical_max_bits) - Memory::YUZU_PAGEBITS))
|
||||
, tracked_entries(device_as_size >> Memory::YUZU_PAGEBITS)
|
||||
|
||||
@@ -375,42 +375,39 @@ NvResult nvhost_as_gpu::MapBufferEx(IoctlMapBufferEx& params) {
|
||||
mapping_map.insert_or_assign(params.offset, Mapping(params.handle, device_address, params.offset, size, false, big_page, false));
|
||||
}
|
||||
|
||||
map_buffer_offsets.insert(params.offset);
|
||||
|
||||
return NvResult::Success;
|
||||
}
|
||||
|
||||
NvResult nvhost_as_gpu::UnmapBuffer(IoctlUnmapBuffer& params) {
|
||||
LOG_DEBUG(Service_NVDRV, "called, offset={:#X}", params.offset);
|
||||
|
||||
std::scoped_lock lock(mutex);
|
||||
if (auto const offset_it = map_buffer_offsets.find(params.offset); offset_it != map_buffer_offsets.end()) {
|
||||
LOG_DEBUG(Service_NVDRV, "called, offset={:#X}", params.offset);
|
||||
if (!vm.initialised) {
|
||||
return NvResult::BadValue;
|
||||
}
|
||||
|
||||
if (!vm.initialised) {
|
||||
return NvResult::BadValue;
|
||||
auto const it = mapping_map.find(params.offset);
|
||||
auto const mapping = it->second;
|
||||
if (!mapping.fixed) {
|
||||
auto& allocator{mapping.big_page ? *vm.big_page_allocator : *vm.small_page_allocator};
|
||||
u32 page_size_bits{mapping.big_page ? vm.big_page_size_bits : VM::PAGE_SIZE_BITS};
|
||||
allocator.Free(u32(mapping.offset >> page_size_bits), u32(mapping.size >> page_size_bits));
|
||||
}
|
||||
|
||||
// Sparse mappings shouldn't be fully unmapped, just returned to their sparse state
|
||||
// Only FreeSpace can unmap them fully
|
||||
if (mapping.sparse_alloc) {
|
||||
gmmu->MapSparse(params.offset, mapping.size, mapping.big_page);
|
||||
} else {
|
||||
gmmu->Unmap(params.offset, mapping.size);
|
||||
}
|
||||
|
||||
nvmap.UnpinHandle(mapping.handle);
|
||||
mapping_map.erase(params.offset);
|
||||
map_buffer_offsets.erase(params.offset);
|
||||
}
|
||||
|
||||
auto const it = mapping_map.find(params.offset);
|
||||
if (it == mapping_map.end()) {
|
||||
LOG_WARNING(Service_NVDRV, "Couldn't find region to unmap at {:#X}", params.offset);
|
||||
return NvResult::Success;
|
||||
}
|
||||
|
||||
auto const mapping = it->second;
|
||||
if (!mapping.fixed) {
|
||||
auto& allocator{mapping.big_page ? *vm.big_page_allocator : *vm.small_page_allocator};
|
||||
u32 page_size_bits{mapping.big_page ? vm.big_page_size_bits : VM::PAGE_SIZE_BITS};
|
||||
allocator.Free(u32(mapping.offset >> page_size_bits), u32(mapping.size >> page_size_bits));
|
||||
}
|
||||
|
||||
// Sparse mappings shouldn't be fully unmapped, just returned to their sparse state
|
||||
// Only FreeSpace can unmap them fully
|
||||
if (mapping.sparse_alloc) {
|
||||
gmmu->MapSparse(params.offset, mapping.size, mapping.big_page);
|
||||
} else {
|
||||
gmmu->Unmap(params.offset, mapping.size);
|
||||
}
|
||||
|
||||
nvmap.UnpinHandle(mapping.handle);
|
||||
mapping_map.erase(it);
|
||||
|
||||
return NvResult::Success;
|
||||
}
|
||||
|
||||
|
||||
@@ -13,6 +13,7 @@
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <optional>
|
||||
#include <ankerl/unordered_dense.h>
|
||||
#include <vector>
|
||||
|
||||
#include "common/address_space.h"
|
||||
@@ -112,6 +113,8 @@ private:
|
||||
};
|
||||
static_assert(sizeof(IoctlRemapEntry) == 20, "IoctlRemapEntry is incorrect size");
|
||||
|
||||
ankerl::unordered_dense::set<s64_le> map_buffer_offsets{};
|
||||
|
||||
struct IoctlMapBufferEx {
|
||||
MappingFlags flags{}; // bit0: fixed_offset, bit2: cacheable
|
||||
u32_le kind{}; // -1 is default
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <algorithm>
|
||||
#include <cstring>
|
||||
#include "common/assert.h"
|
||||
#include "common/logging.h"
|
||||
@@ -265,7 +264,7 @@ NvResult nvhost_ctrl_gpu::ZCullGetInfo(IoctlNvgpuGpuZcullGetInfoArgs& params) {
|
||||
}
|
||||
|
||||
NvResult nvhost_ctrl_gpu::ZBCSetTable(IoctlZbcSetTable& params) {
|
||||
if (params.type == 0 || params.type > supported_types) {
|
||||
if (params.type > supported_types) {
|
||||
LOG_ERROR(Service_NVDRV, "ZBCSetTable: invalid type {:#X}", params.type);
|
||||
return NvResult::BadParameter;
|
||||
}
|
||||
@@ -280,61 +279,42 @@ NvResult nvhost_ctrl_gpu::ZBCSetTable(IoctlZbcSetTable& params) {
|
||||
color_entry.format = params.format;
|
||||
color_entry.ref_cnt = 1u;
|
||||
|
||||
const auto color_end = zbc_colors.begin() + zbc_used_color_entries;
|
||||
auto color_it = std::find_if(zbc_colors.begin(), color_end,
|
||||
[&](const ZbcColorEntry& color_in_question) {
|
||||
return color_entry.format == color_in_question.format &&
|
||||
color_entry.color_ds == color_in_question.color_ds &&
|
||||
color_entry.color_l2 == color_in_question.color_l2;
|
||||
});
|
||||
auto color_it = std::ranges::find_if(zbc_colors,
|
||||
[&](const ZbcColorEntry& color_in_question) {
|
||||
return color_entry.format == color_in_question.format &&
|
||||
color_entry.color_ds == color_in_question.color_ds &&
|
||||
color_entry.color_l2 == color_in_question.color_l2;
|
||||
});
|
||||
|
||||
if (color_it != color_end) {
|
||||
if (color_it != zbc_colors.end()) {
|
||||
++color_it->ref_cnt;
|
||||
LOG_DEBUG(Service_NVDRV, "ZBCSetTable: reused color entry fmt={:#X}, ref_cnt={:#X}",
|
||||
params.format, color_it->ref_cnt);
|
||||
break;
|
||||
} else {
|
||||
zbc_colors.push_back(color_entry);
|
||||
LOG_DEBUG(Service_NVDRV, "ZBCSetTable: added color entry fmt={:#X}, index={:#X}",
|
||||
params.format, zbc_colors.size() - 1);
|
||||
}
|
||||
|
||||
if (zbc_used_color_entries >= zbc_table_size) {
|
||||
LOG_WARNING(Service_NVDRV, "ZBCSetTable: color table is full, fmt={:#X}",
|
||||
params.format);
|
||||
return NvResult::InsufficientMemory;
|
||||
}
|
||||
|
||||
zbc_colors[zbc_used_color_entries] = color_entry;
|
||||
LOG_DEBUG(Service_NVDRV, "ZBCSetTable: added color entry fmt={:#X}, index={:#X}",
|
||||
params.format, zbc_used_color_entries);
|
||||
++zbc_used_color_entries;
|
||||
break;
|
||||
}
|
||||
case ZBCTypes::depth: {
|
||||
ZbcDepthEntry depth_entry{params.depth, params.format, 1u};
|
||||
|
||||
const auto depth_end = zbc_depths.begin() + zbc_used_depth_entries;
|
||||
auto depth_it = std::find_if(zbc_depths.begin(), depth_end,
|
||||
[&](const ZbcDepthEntry& depth_entry_in_question) {
|
||||
return depth_entry.format == depth_entry_in_question.format &&
|
||||
depth_entry.depth == depth_entry_in_question.depth;
|
||||
});
|
||||
auto depth_it = std::ranges::find_if(zbc_depths,
|
||||
[&](const ZbcDepthEntry& depth_entry_in_question) {
|
||||
return depth_entry.format == depth_entry_in_question.format &&
|
||||
depth_entry.depth == depth_entry_in_question.depth;
|
||||
});
|
||||
|
||||
if (depth_it != depth_end) {
|
||||
if (depth_it != zbc_depths.end()) {
|
||||
++depth_it->ref_cnt;
|
||||
LOG_DEBUG(Service_NVDRV, "ZBCSetTable: reused depth entry fmt={:#X}, ref_cnt={:#X}",
|
||||
depth_entry.format, depth_it->ref_cnt);
|
||||
break;
|
||||
} else {
|
||||
zbc_depths.push_back(depth_entry);
|
||||
LOG_DEBUG(Service_NVDRV, "ZBCSetTable: added depth entry fmt={:#X}, index={:#X}",
|
||||
depth_entry.format, zbc_depths.size() - 1);
|
||||
}
|
||||
|
||||
if (zbc_used_depth_entries >= zbc_table_size) {
|
||||
LOG_WARNING(Service_NVDRV, "ZBCSetTable: depth table is full, fmt={:#X}",
|
||||
depth_entry.format);
|
||||
return NvResult::InsufficientMemory;
|
||||
}
|
||||
|
||||
zbc_depths[zbc_used_depth_entries] = depth_entry;
|
||||
LOG_DEBUG(Service_NVDRV, "ZBCSetTable: added depth entry fmt={:#X}, index={:#X}",
|
||||
depth_entry.format, zbc_used_depth_entries);
|
||||
++zbc_used_depth_entries;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -349,34 +329,35 @@ NvResult nvhost_ctrl_gpu::ZBCQueryTable(IoctlZbcQueryTable& params) {
|
||||
|
||||
std::scoped_lock lk(zbc_mutex);
|
||||
|
||||
if (params.type == 0) {
|
||||
params.index_size = zbc_table_size;
|
||||
return NvResult::Success;
|
||||
}
|
||||
|
||||
if (params.index_size >= zbc_table_size) {
|
||||
LOG_ERROR(Service_NVDRV, "ZBCQueryTable: invalid index {:#X}", params.index_size);
|
||||
return NvResult::BadParameter;
|
||||
}
|
||||
|
||||
switch (static_cast<ZBCTypes>(params.type)) {
|
||||
case ZBCTypes::color: {
|
||||
if (params.index_size >= zbc_colors.size()) {
|
||||
LOG_ERROR(Service_NVDRV, "ZBCQueryTable: invalid color index {:#X}", params.index_size);
|
||||
return NvResult::BadParameter;
|
||||
}
|
||||
|
||||
const auto& colors = zbc_colors[params.index_size];
|
||||
std::copy_n(colors.color_ds.begin(), colors.color_ds.size(), std::begin(params.color_ds));
|
||||
std::copy_n(colors.color_l2.begin(), colors.color_l2.size(), std::begin(params.color_l2));
|
||||
params.depth = 0;
|
||||
params.ref_cnt = colors.ref_cnt;
|
||||
params.format = colors.format;
|
||||
params.index_size = static_cast<u32>(zbc_colors.size());
|
||||
break;
|
||||
}
|
||||
case ZBCTypes::depth: {
|
||||
if (params.index_size >= zbc_depths.size()) {
|
||||
LOG_ERROR(Service_NVDRV, "ZBCQueryTable: invalid depth index {:#X}", params.index_size);
|
||||
return NvResult::BadParameter;
|
||||
}
|
||||
|
||||
const auto& depth_entry = zbc_depths[params.index_size];
|
||||
std::fill(std::begin(params.color_ds), std::end(params.color_ds), 0);
|
||||
std::fill(std::begin(params.color_l2), std::end(params.color_l2), 0);
|
||||
params.depth = depth_entry.depth;
|
||||
params.ref_cnt = depth_entry.ref_cnt;
|
||||
params.format = depth_entry.format;
|
||||
break;
|
||||
params.index_size = static_cast<u32>(zbc_depths.size());
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -6,7 +6,7 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <vector>
|
||||
|
||||
#include "common/common_funcs.h"
|
||||
#include "common/common_types.h"
|
||||
@@ -212,13 +212,9 @@ private:
|
||||
Kernel::KEvent* unknown_event;
|
||||
|
||||
// ZBC Tables
|
||||
static constexpr u32 zbc_table_size = 15u;
|
||||
|
||||
std::mutex zbc_mutex{};
|
||||
std::array<ZbcColorEntry, zbc_table_size> zbc_colors{};
|
||||
std::array<ZbcDepthEntry, zbc_table_size> zbc_depths{};
|
||||
u32 zbc_used_color_entries{};
|
||||
u32 zbc_used_depth_entries{};
|
||||
std::vector<ZbcColorEntry> zbc_colors{};
|
||||
std::vector<ZbcDepthEntry> zbc_depths{};
|
||||
const u32 supported_types = 2u;
|
||||
};
|
||||
|
||||
|
||||
@@ -174,9 +174,7 @@ NvResult nvhost_gpu::SetChannelPriority(IoctlChannelSetPriority& params) {
|
||||
case ChannelPriority::Low: channel_timeslice = 1300; break;
|
||||
case ChannelPriority::Medium: channel_timeslice = 2600; break;
|
||||
case ChannelPriority::High: channel_timeslice = 5200; break;
|
||||
default:
|
||||
LOG_WARNING(Service_NVDRV, "unknown channel priority {:#X}", channel_priority);
|
||||
break;
|
||||
default : return NvResult::BadParameter;
|
||||
}
|
||||
|
||||
return NvResult::Success;
|
||||
@@ -280,20 +278,18 @@ NvResult nvhost_gpu::AllocateObjectContext(IoctlAllocObjCtx& params) {
|
||||
params.flags = allowed_mask;
|
||||
}
|
||||
|
||||
params.obj_id = 0;
|
||||
|
||||
s32_le ctx_class_number_index =
|
||||
s32_le ctx_class_number_index =
|
||||
GetObjectContextClassNumberIndex(static_cast<CtxClasses>(params.class_num));
|
||||
if (ctx_class_number_index < 0) {
|
||||
LOG_WARNING(Service_NVDRV, "Untracked class number for object context: {:#X}",
|
||||
params.class_num);
|
||||
return NvResult::Success;
|
||||
LOG_ERROR(Service_NVDRV, "Invalid class number for object context: {:#X}",
|
||||
params.class_num);
|
||||
return NvResult::BadParameter;
|
||||
}
|
||||
|
||||
if (ctxObjs[ctx_class_number_index].has_value()) {
|
||||
LOG_DEBUG(Service_NVDRV, "Object context for class {:#X} already allocated on this channel",
|
||||
params.class_num);
|
||||
return NvResult::Success;
|
||||
LOG_WARNING(Service_NVDRV, "Object context for class {:#X} already allocated on this channel",
|
||||
params.class_num);
|
||||
return NvResult::AlreadyAllocated;
|
||||
}
|
||||
|
||||
// Defer actual hardware context binding until channel is initialized.
|
||||
@@ -439,6 +435,10 @@ NvResult nvhost_gpu::ChannelSetTimeout(IoctlChannelSetTimeout& params) {
|
||||
NvResult nvhost_gpu::ChannelSetTimeslice(IoctlSetTimeslice& params) {
|
||||
LOG_INFO(Service_NVDRV, "called, timeslice={:#X}", params.timeslice);
|
||||
|
||||
if (params.timeslice < 1000 || params.timeslice > 5000) {
|
||||
return NvResult::BadParameter;
|
||||
}
|
||||
|
||||
channel_timeslice = params.timeslice;
|
||||
|
||||
return NvResult::Success;
|
||||
|
||||
@@ -20,23 +20,33 @@ BufferQueueCore::~BufferQueueCore() = default;
|
||||
void BufferQueueCore::PushHistory(u64 frame_number, s64 queue_time, s64 presentation_time, BufferState state) {
|
||||
std::lock_guard lk(buffer_history_mutex);
|
||||
|
||||
buffer_history_pos = (buffer_history_pos + 1) % BUFFER_HISTORY_SIZE;
|
||||
buffer_history[buffer_history_pos] = BufferHistoryInfo{
|
||||
auto it = buffer_history_map.find(frame_number);
|
||||
if (it != buffer_history_map.end()) {
|
||||
it->second.state = state;
|
||||
return;
|
||||
}
|
||||
|
||||
buffer_history_map.emplace(frame_number, BufferHistoryInfo{
|
||||
frame_number,
|
||||
queue_time,
|
||||
presentation_time,
|
||||
state
|
||||
};
|
||||
});
|
||||
buffer_history_order.push_back(frame_number);
|
||||
|
||||
if (buffer_history_order.size() > BUFFER_HISTORY_SIZE) {
|
||||
u64 oldest_frame = buffer_history_order.front();
|
||||
buffer_history_order.pop_front();
|
||||
buffer_history_map.erase(oldest_frame);
|
||||
}
|
||||
}
|
||||
|
||||
void BufferQueueCore::UpdateHistory(u64 frame_number, BufferState state) {
|
||||
std::lock_guard lk(buffer_history_mutex);
|
||||
|
||||
for (auto& entry : buffer_history) {
|
||||
if (entry.frame_number == frame_number) {
|
||||
entry.state = state;
|
||||
return;
|
||||
}
|
||||
auto it = buffer_history_map.find(frame_number);
|
||||
if (it != buffer_history_map.end()) {
|
||||
it->second.state = state;
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -9,13 +9,14 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <condition_variable>
|
||||
#include <deque>
|
||||
#include <list>
|
||||
#include <memory>
|
||||
#include <mutex>
|
||||
#include <set>
|
||||
#include <vector>
|
||||
#include <unordered_map>
|
||||
#include <algorithm>
|
||||
|
||||
#include "core/hle/service/nvnflinger/buffer_item.h"
|
||||
@@ -27,15 +28,12 @@
|
||||
|
||||
namespace Service::android {
|
||||
|
||||
#pragma pack(push, 1)
|
||||
struct BufferHistoryInfo {
|
||||
u64 frame_number;
|
||||
s64 queue_time;
|
||||
s64 presentation_time;
|
||||
BufferState state;
|
||||
u64 frame_number{};
|
||||
s64 queue_time{};
|
||||
s64 presentation_time{};
|
||||
BufferState state{};
|
||||
};
|
||||
#pragma pack(pop)
|
||||
static_assert(sizeof(BufferHistoryInfo) == 0x1C, "BufferHistoryInfo must be 28 bytes");
|
||||
|
||||
class IConsumerListener;
|
||||
class IProducerListener;
|
||||
@@ -90,9 +88,9 @@ private:
|
||||
bool buffer_has_been_queued{};
|
||||
u64 frame_counter{};
|
||||
|
||||
std::array<BufferHistoryInfo, BUFFER_HISTORY_SIZE> buffer_history{};
|
||||
u32 buffer_history_pos{BUFFER_HISTORY_SIZE - 1};
|
||||
std::unordered_map<u64, BufferHistoryInfo> buffer_history_map{};
|
||||
mutable std::mutex buffer_history_mutex{};
|
||||
std::deque<u64> buffer_history_order;
|
||||
|
||||
u32 transform_hint{};
|
||||
bool is_allocating{};
|
||||
|
||||
@@ -507,8 +507,6 @@ Status BufferQueueProducer::QueueBuffer(s32 slot, const QueueBufferInput& input,
|
||||
|
||||
sticky_transform = sticky_transform_;
|
||||
|
||||
const bool track_history = Settings::values.enable_buffer_history.GetValue();
|
||||
|
||||
if (core->queue.empty()) {
|
||||
core->queue.push_back(item);
|
||||
listener_available = core->consumer_listener;
|
||||
@@ -516,7 +514,7 @@ Status BufferQueueProducer::QueueBuffer(s32 slot, const QueueBufferInput& input,
|
||||
auto front = core->queue.begin();
|
||||
if (front->is_droppable && core->StillTracking(*front)) {
|
||||
slots[front->slot].buffer_state = BufferState::Free;
|
||||
if (track_history) {
|
||||
if (Settings::values.enable_buffer_history.GetValue()) {
|
||||
core->UpdateHistory(front->frame_number, BufferState::Free);
|
||||
}
|
||||
slots[front->slot].frame_number = 0;
|
||||
@@ -531,7 +529,7 @@ Status BufferQueueProducer::QueueBuffer(s32 slot, const QueueBufferInput& input,
|
||||
}
|
||||
}
|
||||
|
||||
if (track_history) {
|
||||
if (Settings::values.enable_buffer_history.GetValue()) {
|
||||
core->PushHistory(core->frame_counter, slots[slot].queue_time, slots[slot].presentation_time, BufferState::Queued);
|
||||
}
|
||||
|
||||
@@ -904,31 +902,26 @@ void BufferQueueProducer::Transact(u32 code, std::span<const u8> parcel_data,
|
||||
|
||||
const s32 request = parcel_in.Read<s32>();
|
||||
if (request <= 0) {
|
||||
status = Status::BadValue;
|
||||
parcel_out.Write(Status::BadValue);
|
||||
parcel_out.Write<s32>(0);
|
||||
break;
|
||||
}
|
||||
|
||||
constexpr u32 history_size = BufferQueueCore::BUFFER_HISTORY_SIZE;
|
||||
std::array<BufferHistoryInfo, history_size> snapshot{};
|
||||
s32 count{};
|
||||
std::vector<BufferHistoryInfo> snapshot;
|
||||
|
||||
{
|
||||
std::scoped_lock lk(core->buffer_history_mutex);
|
||||
|
||||
const u32 newest = core->buffer_history_pos;
|
||||
for (u32 i = 0; i < history_size; ++i) {
|
||||
const auto& entry = core->buffer_history[(newest + history_size - i) % history_size];
|
||||
if (entry.frame_number == 0) {
|
||||
break;
|
||||
}
|
||||
|
||||
snapshot[count] = entry;
|
||||
++count;
|
||||
for (auto& [frame, info] : core->buffer_history_map) {
|
||||
snapshot.push_back(info);
|
||||
}
|
||||
}
|
||||
|
||||
const s32 limit = (std::min)(request, count);
|
||||
std::sort(snapshot.begin(), snapshot.end(), [](auto& a, auto& b){
|
||||
return a.frame_number > b.frame_number;
|
||||
});
|
||||
|
||||
const s32 limit = std::min(request, (s32)snapshot.size());
|
||||
parcel_out.Write(Status::NoError);
|
||||
parcel_out.Write<s32>(limit);
|
||||
for (s32 i = 0; i < limit; ++i) {
|
||||
parcel_out.Write(snapshot[i]);
|
||||
|
||||
@@ -5,7 +5,6 @@
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include "common/settings.h"
|
||||
#include "common/thread.h"
|
||||
#include "core/core.h"
|
||||
#include "core/core_timing.h"
|
||||
#include "core/hle/service/vi/conductor.h"
|
||||
@@ -77,7 +76,6 @@ void Conductor::ProcessVsync() {
|
||||
|
||||
void Conductor::VsyncThread(std::stop_token token) {
|
||||
Common::SetCurrentThreadName("VSyncThread");
|
||||
Common::SetCurrentThreadPriority(Common::ThreadPriority::High);
|
||||
|
||||
while (!token.stop_requested()) {
|
||||
m_signal.Wait();
|
||||
|
||||
+36
-3
@@ -17,12 +17,28 @@ enum class Mode : u64 {
|
||||
Attr,
|
||||
};
|
||||
|
||||
enum class SZ : u64 {
|
||||
U8,
|
||||
U16,
|
||||
U32,
|
||||
F32
|
||||
};
|
||||
|
||||
enum class Shift : u64 {
|
||||
Default,
|
||||
U16,
|
||||
B32,
|
||||
};
|
||||
|
||||
IR::U32 scaleIndex(IR::IREmitter& ir, IR::U32 index, Shift shift) {
|
||||
switch (shift) {
|
||||
case Shift::Default: return index;
|
||||
case Shift::U16: return ir.ShiftLeftLogical(index, ir.Imm32(1));
|
||||
case Shift::B32: return ir.ShiftLeftLogical(index, ir.Imm32(2));
|
||||
default: UNREACHABLE();
|
||||
}
|
||||
}
|
||||
|
||||
} // Anonymous namespace
|
||||
|
||||
void TranslatorVisitor::ISBERD(u64 insn) {
|
||||
@@ -37,6 +53,7 @@ void TranslatorVisitor::ISBERD(u64 insn) {
|
||||
BitField<31, 1, u64> skew;
|
||||
BitField<32, 1, u64> o;
|
||||
BitField<33, 2, Mode> mode;
|
||||
BitField<36, 4, SZ> sz;
|
||||
BitField<47, 2, Shift> shift;
|
||||
} const isberd{insn};
|
||||
|
||||
@@ -46,15 +63,31 @@ void TranslatorVisitor::ISBERD(u64 insn) {
|
||||
if (isberd.o != 0) {
|
||||
throw NotImplementedException("ISBERD O");
|
||||
}
|
||||
if (isberd.sz.Value() > SZ::F32) {
|
||||
throw NotImplementedException("ISBERD SZ {}",
|
||||
static_cast<u64>(isberd.sz.Value()));
|
||||
}
|
||||
if (isberd.shift.Value() > Shift::B32) {
|
||||
throw NotImplementedException("ISBERD Shift {}",
|
||||
static_cast<u64>(isberd.shift.Value()));
|
||||
}
|
||||
|
||||
switch (isberd.mode.Value()) {
|
||||
case Mode::Default:
|
||||
X(isberd.dest_reg.Value(), X(isberd.src_reg.Value()));
|
||||
return;
|
||||
case Mode::Attr:
|
||||
LOG_DEBUG(Shader, "(STUBBED) ISBERD Mode Attr");
|
||||
X(isberd.dest_reg.Value(), X(isberd.src_reg.Value()));
|
||||
case Mode::Attr: {
|
||||
IR::U32 offset{};
|
||||
if (isberd.src_reg_num.Value() == 0xFF) {
|
||||
offset = ir.Imm32(isberd.imm.Value());
|
||||
} else {
|
||||
const IR::U32 index{
|
||||
scaleIndex(ir, X(isberd.src_reg.Value()), isberd.shift.Value())};
|
||||
offset = ir.IAdd(index, ir.Imm32(isberd.imm.Value()));
|
||||
}
|
||||
X(isberd.dest_reg.Value(), ir.BitCast<IR::U32>(ir.GetAttributeIndexed(offset)));
|
||||
return;
|
||||
}
|
||||
default:
|
||||
throw NotImplementedException("ISBERD Mode {}",
|
||||
static_cast<u64>(isberd.mode.Value()));
|
||||
|
||||
@@ -92,11 +92,6 @@ u64 BufferCache<P>::ReclaimMemory(u64 target_bytes, bool allow_download) {
|
||||
return freed;
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::ReclaimDeferredResources(u64 completed_sync_point) {
|
||||
sentenced_buffers.Reclaim(completed_sync_point);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::EnsureHeadroom(bool allow_download) {
|
||||
if (reclaim_stalled) {
|
||||
@@ -144,9 +139,9 @@ void BufferCache<P>::TickFrame() {
|
||||
|
||||
usage_refresh_countdown = 0;
|
||||
reclaim_stalled = false;
|
||||
ReclaimDeferredResources(runtime.CompletedSyncPoint());
|
||||
EnsureHeadroom(true);
|
||||
++frame_tick;
|
||||
sentenced_buffers.Reclaim(runtime.CompletedSyncPoint());
|
||||
|
||||
for (auto& buffer : async_buffers_death_ring) {
|
||||
runtime.FreeDeferredStagingBuffer(buffer);
|
||||
@@ -675,11 +670,7 @@ void BufferCache<P>::AccumulateFlushes() {
|
||||
|
||||
template <class P>
|
||||
bool BufferCache<P>::ShouldWaitAsyncFlushes() const noexcept {
|
||||
if (async_buffers.empty()) {
|
||||
return false;
|
||||
}
|
||||
return async_buffers.front().has_value() ||
|
||||
!pending_downloads.front().unified_copies.empty();
|
||||
return (!async_buffers.empty() && async_buffers.front().has_value());
|
||||
}
|
||||
|
||||
template <class P>
|
||||
@@ -687,7 +678,6 @@ void BufferCache<P>::CommitAsyncFlushesHigh() {
|
||||
AccumulateFlushes();
|
||||
|
||||
if (committed_gpu_modified_ranges.empty()) {
|
||||
pending_downloads.emplace_back();
|
||||
async_buffers.emplace_back(std::optional<Async_Buffer>{});
|
||||
return;
|
||||
}
|
||||
@@ -747,83 +737,27 @@ void BufferCache<P>::CommitAsyncFlushesHigh() {
|
||||
}
|
||||
committed_gpu_modified_ranges.clear();
|
||||
if (downloads.empty()) {
|
||||
pending_downloads.emplace_back();
|
||||
async_buffers.emplace_back(std::optional<Async_Buffer>{});
|
||||
return;
|
||||
}
|
||||
|
||||
struct QueuedUnifiedCopy {
|
||||
u64 window;
|
||||
BufferId buffer_id;
|
||||
boost::container::small_vector<BufferCopy, 16> copies;
|
||||
};
|
||||
|
||||
AsyncDownloadBatch batch;
|
||||
boost::container::small_vector<std::pair<BufferCopy, BufferId>, 16> staging_downloads;
|
||||
boost::container::small_vector<QueuedUnifiedCopy, 4> unified_copy_queue;
|
||||
boost::container::small_vector<u64, 4> window_ids;
|
||||
UnifiedWindowGroups groups;
|
||||
u64 staging_size_bytes = 0;
|
||||
for (auto& [copy, buffer_id] : downloads) {
|
||||
Buffer& buffer = slot_buffers[buffer_id];
|
||||
const DAddr orig_device_addr = buffer.CpuAddr() + copy.src_offset;
|
||||
bool unified = false;
|
||||
if constexpr (USE_UNIFIED_MEMORY) {
|
||||
if (runtime.HasUnifiedMemory()) {
|
||||
window_ids.clear();
|
||||
groups.clear();
|
||||
unified = ResolveUnifiedWindows(orig_device_addr, copy.src_offset, copy.size,
|
||||
window_ids, groups);
|
||||
}
|
||||
}
|
||||
BufferCopy record{copy};
|
||||
record.src_offset = static_cast<size_t>(orig_device_addr);
|
||||
if (unified) {
|
||||
async_downloads.Add(orig_device_addr, copy.size);
|
||||
buffer.MarkUsage(copy.src_offset, copy.size);
|
||||
for (size_t i = 0; i < window_ids.size(); ++i) {
|
||||
unified_copy_queue.push_back(
|
||||
QueuedUnifiedCopy{window_ids[i], buffer_id, std::move(groups[i])});
|
||||
}
|
||||
batch.unified_copies.push_back(record);
|
||||
continue;
|
||||
}
|
||||
copy.dst_offset = staging_size_bytes;
|
||||
constexpr u64 align = 64ULL;
|
||||
staging_size_bytes += (copy.size + align - 1) & ~(align - 1ULL);
|
||||
staging_downloads.push_back({copy, buffer_id});
|
||||
}
|
||||
|
||||
std::optional<Async_Buffer> download_staging;
|
||||
if (!staging_downloads.empty()) {
|
||||
download_staging = runtime.DownloadStagingBuffer(staging_size_bytes, true);
|
||||
}
|
||||
auto download_staging = runtime.DownloadStagingBuffer(total_size_bytes, true);
|
||||
boost::container::small_vector<BufferCopy, 4> normalized_copies;
|
||||
runtime.PreCopyBarrier();
|
||||
for (auto& [copy, buffer_id] : staging_downloads) {
|
||||
copy.dst_offset += download_staging->offset;
|
||||
for (auto& [copy, buffer_id] : downloads) {
|
||||
copy.dst_offset += download_staging.offset;
|
||||
const std::array copies{copy};
|
||||
BufferCopy second_copy{copy};
|
||||
Buffer& buffer = slot_buffers[buffer_id];
|
||||
BufferCopy record{copy};
|
||||
record.src_offset = static_cast<size_t>(buffer.CpuAddr()) + copy.src_offset;
|
||||
const DAddr orig_device_addr = static_cast<DAddr>(record.src_offset);
|
||||
second_copy.src_offset = static_cast<size_t>(buffer.CpuAddr()) + copy.src_offset;
|
||||
const DAddr orig_device_addr = static_cast<DAddr>(second_copy.src_offset);
|
||||
async_downloads.Add(orig_device_addr, copy.size);
|
||||
buffer.MarkUsage(copy.src_offset, copy.size);
|
||||
runtime.CopyBuffer(download_staging->buffer, buffer, copies, false);
|
||||
batch.staging_copies.push_back(record);
|
||||
}
|
||||
if constexpr (USE_UNIFIED_MEMORY) {
|
||||
for (const auto& queued : unified_copy_queue) {
|
||||
const std::span<const BufferCopy> group_span(queued.copies.data(),
|
||||
queued.copies.size());
|
||||
runtime.CopyToUnifiedMemory(queued.window, slot_buffers[queued.buffer_id], group_span);
|
||||
}
|
||||
if (!unified_copy_queue.empty()) {
|
||||
runtime.UnifiedMemoryHostBarrier();
|
||||
}
|
||||
runtime.CopyBuffer(download_staging.buffer, buffer, copies, false);
|
||||
normalized_copies.push_back(second_copy);
|
||||
}
|
||||
runtime.PostCopyBarrier();
|
||||
pending_downloads.emplace_back(std::move(batch));
|
||||
async_buffers.emplace_back(std::move(download_staging));
|
||||
pending_downloads.emplace_back(std::move(normalized_copies));
|
||||
async_buffers.emplace_back(download_staging);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
@@ -849,32 +783,27 @@ void BufferCache<P>::PopAsyncBuffers() {
|
||||
if (async_buffers.empty()) {
|
||||
return;
|
||||
}
|
||||
auto& batch = pending_downloads.front();
|
||||
auto& async_buffer = async_buffers.front();
|
||||
if (async_buffer.has_value()) {
|
||||
const u8* base = async_buffer->mapped_span.data();
|
||||
const size_t base_offset = async_buffer->offset;
|
||||
for (const auto& copy : batch.staging_copies) {
|
||||
const DAddr device_addr = static_cast<DAddr>(copy.src_offset);
|
||||
const u64 dst_offset = copy.dst_offset - base_offset;
|
||||
const u8* read_mapped_memory = base + dst_offset;
|
||||
async_downloads.ForEachInRange(
|
||||
device_addr, copy.size, [&](DAddr start, DAddr end, s32) {
|
||||
writebacks.push_back(
|
||||
{start, &read_mapped_memory[start - device_addr], end - start});
|
||||
});
|
||||
async_downloads.Subtract(device_addr, copy.size, [&](DAddr start, DAddr end) {
|
||||
gpu_modified_ranges.Subtract(start, end - start);
|
||||
});
|
||||
}
|
||||
async_buffers_death_ring.emplace_back(*async_buffer);
|
||||
if (!async_buffers.front().has_value()) {
|
||||
async_buffers.pop_front();
|
||||
return;
|
||||
}
|
||||
for (const auto& copy : batch.unified_copies) {
|
||||
auto& downloads = pending_downloads.front();
|
||||
auto& async_buffer = async_buffers.front();
|
||||
const u8* base = async_buffer->mapped_span.data();
|
||||
const size_t base_offset = async_buffer->offset;
|
||||
for (const auto& copy : downloads) {
|
||||
const DAddr device_addr = static_cast<DAddr>(copy.src_offset);
|
||||
const u64 dst_offset = copy.dst_offset - base_offset;
|
||||
const u8* read_mapped_memory = base + dst_offset;
|
||||
async_downloads.ForEachInRange(device_addr, copy.size, [&](DAddr start, DAddr end, s32) {
|
||||
writebacks.push_back(
|
||||
{start, &read_mapped_memory[start - device_addr], end - start});
|
||||
});
|
||||
async_downloads.Subtract(device_addr, copy.size, [&](DAddr start, DAddr end) {
|
||||
gpu_modified_ranges.Subtract(start, end - start);
|
||||
});
|
||||
}
|
||||
async_buffers_death_ring.emplace_back(*async_buffer);
|
||||
async_buffers.pop_front();
|
||||
pending_downloads.pop_front();
|
||||
}
|
||||
@@ -1886,18 +1815,17 @@ void BufferCache<P>::ImmediateUploadMemory([[maybe_unused]] Buffer& buffer,
|
||||
}
|
||||
|
||||
template <class P>
|
||||
bool BufferCache<P>::ResolveUnifiedWindows(
|
||||
[[maybe_unused]] DAddr device_addr, [[maybe_unused]] u64 buffer_offset,
|
||||
[[maybe_unused]] u64 size, [[maybe_unused]] boost::container::small_vector<u64, 4>& window_ids,
|
||||
[[maybe_unused]] UnifiedWindowGroups& groups) {
|
||||
bool BufferCache<P>::TryUnifiedDownloadMemory([[maybe_unused]] Buffer& buffer,
|
||||
[[maybe_unused]] std::span<BufferCopy> copies) {
|
||||
if constexpr (USE_UNIFIED_MEMORY) {
|
||||
const u8* const physical_base = device_memory.GetPhysicalBase();
|
||||
const u64 unified_base = runtime.UnifiedMemoryBase();
|
||||
const u64 unified_size = runtime.UnifiedMemorySize();
|
||||
const u64 window_size = runtime.UnifiedMemoryWindowSize();
|
||||
if (window_size == 0) {
|
||||
return false;
|
||||
}
|
||||
boost::container::small_vector<u64, 4> window_ids;
|
||||
boost::container::small_vector<boost::container::small_vector<BufferCopy, 16>, 4> groups;
|
||||
const auto group_for = [&](u64 window) -> boost::container::small_vector<BufferCopy, 16>& {
|
||||
for (size_t i = 0; i < window_ids.size(); ++i) {
|
||||
if (window_ids[i] == window) {
|
||||
@@ -1908,68 +1836,51 @@ bool BufferCache<P>::ResolveUnifiedWindows(
|
||||
groups.emplace_back();
|
||||
return groups.back();
|
||||
};
|
||||
u64 downloaded = 0;
|
||||
while (downloaded < size) {
|
||||
const DAddr page_addr = device_addr + downloaded;
|
||||
const u8* const ptr = device_memory.GetPointer<u8>(page_addr);
|
||||
if (ptr == nullptr) {
|
||||
return false;
|
||||
}
|
||||
const u64 page_offset = page_addr & Core::DEVICE_PAGEMASK;
|
||||
u64 chunk = (std::min)(size - downloaded,
|
||||
static_cast<u64>(Core::DEVICE_PAGESIZE) - page_offset);
|
||||
const u64 phys_offset = static_cast<u64>(ptr - physical_base);
|
||||
if (phys_offset < unified_base || phys_offset - unified_base + chunk > unified_size) {
|
||||
return false;
|
||||
}
|
||||
const u64 relative = phys_offset - unified_base;
|
||||
const u64 window = relative / window_size;
|
||||
const u64 local_offset = relative % window_size;
|
||||
chunk = (std::min)(chunk, window_size - local_offset);
|
||||
auto& group = group_for(window);
|
||||
if (!group.empty()) {
|
||||
BufferCopy& last = group.back();
|
||||
if (last.src_offset + last.size == buffer_offset + downloaded &&
|
||||
last.dst_offset + last.size == local_offset) {
|
||||
last.size += chunk;
|
||||
downloaded += chunk;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
group.push_back(BufferCopy{
|
||||
.src_offset = buffer_offset + downloaded,
|
||||
.dst_offset = local_offset,
|
||||
.size = chunk,
|
||||
});
|
||||
downloaded += chunk;
|
||||
}
|
||||
return true;
|
||||
} else {
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
bool BufferCache<P>::TryUnifiedDownloadMemory([[maybe_unused]] Buffer& buffer,
|
||||
[[maybe_unused]] std::span<BufferCopy> copies) {
|
||||
if constexpr (USE_UNIFIED_MEMORY) {
|
||||
boost::container::small_vector<u64, 4> window_ids;
|
||||
UnifiedWindowGroups groups;
|
||||
for (const BufferCopy& copy : copies) {
|
||||
if (!ResolveUnifiedWindows(buffer.CpuAddr() + copy.src_offset, copy.src_offset,
|
||||
copy.size, window_ids, groups)) {
|
||||
return false;
|
||||
const DAddr device_addr = buffer.CpuAddr() + copy.src_offset;
|
||||
u64 downloaded = 0;
|
||||
while (downloaded < copy.size) {
|
||||
const DAddr page_addr = device_addr + downloaded;
|
||||
const u8* const ptr = device_memory.GetPointer<u8>(page_addr);
|
||||
if (ptr == nullptr) {
|
||||
return false;
|
||||
}
|
||||
const u64 page_offset = page_addr & Core::DEVICE_PAGEMASK;
|
||||
u64 chunk = (std::min)(copy.size - downloaded,
|
||||
static_cast<u64>(Core::DEVICE_PAGESIZE) - page_offset);
|
||||
const u64 phys_offset = static_cast<u64>(ptr - physical_base);
|
||||
if (phys_offset + chunk > unified_size) {
|
||||
return false;
|
||||
}
|
||||
const u64 window = phys_offset / window_size;
|
||||
const u64 local_offset = phys_offset % window_size;
|
||||
chunk = (std::min)(chunk, window_size - local_offset);
|
||||
auto& group = group_for(window);
|
||||
if (!group.empty()) {
|
||||
BufferCopy& last = group.back();
|
||||
if (last.src_offset + last.size == copy.src_offset + downloaded &&
|
||||
last.dst_offset + last.size == local_offset) {
|
||||
last.size += chunk;
|
||||
downloaded += chunk;
|
||||
continue;
|
||||
}
|
||||
}
|
||||
group.push_back(BufferCopy{
|
||||
.src_offset = copy.src_offset + downloaded,
|
||||
.dst_offset = local_offset,
|
||||
.size = chunk,
|
||||
});
|
||||
downloaded += chunk;
|
||||
}
|
||||
}
|
||||
for (const BufferCopy& copy : copies) {
|
||||
buffer.MarkUsage(copy.src_offset, copy.size);
|
||||
}
|
||||
runtime.PreCopyBarrier();
|
||||
for (size_t i = 0; i < window_ids.size(); ++i) {
|
||||
const std::span<const BufferCopy> group_span(groups[i].data(), groups[i].size());
|
||||
runtime.CopyToUnifiedMemory(window_ids[i], buffer, group_span);
|
||||
const std::span<BufferCopy> group_span(groups[i].data(), groups[i].size());
|
||||
runtime.CopyBuffer(runtime.UnifiedMemoryWindowBuffer(window_ids[i]), buffer,
|
||||
group_span, true);
|
||||
}
|
||||
runtime.UnifiedMemoryHostBarrier();
|
||||
runtime.Finish();
|
||||
return true;
|
||||
} else {
|
||||
|
||||
@@ -221,8 +221,6 @@ public:
|
||||
|
||||
u64 ReclaimMemory(u64 target_bytes, bool allow_download);
|
||||
|
||||
void ReclaimDeferredResources(u64 completed_sync_point);
|
||||
|
||||
void WriteMemory(DAddr device_addr, u64 size);
|
||||
|
||||
void CachedWriteMemory(DAddr device_addr, u64 size);
|
||||
@@ -455,13 +453,6 @@ private:
|
||||
|
||||
bool TryUnifiedDownloadMemory(Buffer& buffer, std::span<BufferCopy> copies);
|
||||
|
||||
using UnifiedWindowGroups =
|
||||
boost::container::small_vector<boost::container::small_vector<BufferCopy, 16>, 4>;
|
||||
|
||||
bool ResolveUnifiedWindows(DAddr device_addr, u64 buffer_offset, u64 size,
|
||||
boost::container::small_vector<u64, 4>& window_ids,
|
||||
UnifiedWindowGroups& groups);
|
||||
|
||||
void DownloadBufferMemory(Buffer& buffer_id);
|
||||
|
||||
void DownloadBufferMemory(Buffer& buffer_id, DAddr device_addr, u64 size);
|
||||
@@ -512,14 +503,9 @@ private:
|
||||
std::deque<Common::RangeSet<DAddr>> committed_gpu_modified_ranges;
|
||||
|
||||
// Async Buffers
|
||||
struct AsyncDownloadBatch {
|
||||
boost::container::small_vector<BufferCopy, 4> staging_copies;
|
||||
boost::container::small_vector<BufferCopy, 4> unified_copies;
|
||||
};
|
||||
|
||||
Common::OverlapRangeSet<DAddr> async_downloads;
|
||||
std::deque<std::optional<Async_Buffer>> async_buffers;
|
||||
std::deque<AsyncDownloadBatch> pending_downloads;
|
||||
std::deque<boost::container::small_vector<BufferCopy, 4>> pending_downloads;
|
||||
std::optional<Async_Buffer> current_buffer;
|
||||
|
||||
std::deque<Async_Buffer> async_buffers_death_ring;
|
||||
|
||||
@@ -366,91 +366,14 @@ BufferCacheRuntime::BufferCacheRuntime(const Device& device_, MemoryAllocator& m
|
||||
|
||||
void BufferCacheRuntime::TryEnableUnifiedMemory(void* base, size_t size,
|
||||
std::span<AHardwareBuffer* const> hardware_buffers,
|
||||
size_t hardware_buffer_window,
|
||||
size_t hardware_buffer_base) {
|
||||
unified_memory = std::make_unique<HostMemoryImport>(
|
||||
device, base, size, hardware_buffers, hardware_buffer_window, hardware_buffer_base);
|
||||
size_t hardware_buffer_window) {
|
||||
unified_memory = std::make_unique<HostMemoryImport>(device, base, size, hardware_buffers,
|
||||
hardware_buffer_window);
|
||||
if (!unified_memory->IsValid()) {
|
||||
unified_memory.reset();
|
||||
}
|
||||
}
|
||||
|
||||
void BufferCacheRuntime::CopyToUnifiedMemory(
|
||||
size_t window_index, VkBuffer src_buffer,
|
||||
std::span<const VideoCommon::BufferCopy> copies) {
|
||||
if (!unified_memory || src_buffer == VK_NULL_HANDLE || copies.empty() ||
|
||||
window_index >= unified_memory->GetWindowCount()) {
|
||||
return;
|
||||
}
|
||||
const VkBuffer dst_buffer = unified_memory->GetWindowBuffer(window_index);
|
||||
if (dst_buffer == VK_NULL_HANDLE) {
|
||||
return;
|
||||
}
|
||||
VkDeviceSize covered_begin = std::numeric_limits<VkDeviceSize>::max();
|
||||
VkDeviceSize covered_end = 0;
|
||||
for (const VideoCommon::BufferCopy& copy : copies) {
|
||||
covered_begin = (std::min)(covered_begin, static_cast<VkDeviceSize>(copy.dst_offset));
|
||||
covered_end = (std::max)(covered_end,
|
||||
static_cast<VkDeviceSize>(copy.dst_offset + copy.size));
|
||||
}
|
||||
|
||||
boost::container::small_vector<VkBufferCopy, 8> vk_copies(copies.size());
|
||||
std::ranges::transform(copies, vk_copies.begin(), MakeBufferCopy);
|
||||
|
||||
const bool foreign = unified_memory->NeedsForeignOwnershipTransfer();
|
||||
const u32 queue_family = device.GetGraphicsFamily();
|
||||
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
scheduler.Record([src_buffer, dst_buffer, vk_copies, foreign, queue_family, covered_begin,
|
||||
covered_end](vk::CommandBuffer cmdbuf) {
|
||||
if (foreign) {
|
||||
const VkBufferMemoryBarrier acquire{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER,
|
||||
.pNext = nullptr,
|
||||
.srcAccessMask = 0,
|
||||
.dstAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT,
|
||||
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_FOREIGN_EXT,
|
||||
.dstQueueFamilyIndex = queue_family,
|
||||
.buffer = dst_buffer,
|
||||
.offset = covered_begin,
|
||||
.size = covered_end - covered_begin,
|
||||
};
|
||||
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT,
|
||||
VK_PIPELINE_STAGE_TRANSFER_BIT, 0, acquire);
|
||||
}
|
||||
cmdbuf.CopyBuffer(src_buffer, dst_buffer, VideoCommon::FixSmallVectorADL(vk_copies));
|
||||
if (foreign) {
|
||||
const VkBufferMemoryBarrier release{
|
||||
.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER,
|
||||
.pNext = nullptr,
|
||||
.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT,
|
||||
.dstAccessMask = 0,
|
||||
.srcQueueFamilyIndex = queue_family,
|
||||
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_FOREIGN_EXT,
|
||||
.buffer = dst_buffer,
|
||||
.offset = covered_begin,
|
||||
.size = covered_end - covered_begin,
|
||||
};
|
||||
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_TRANSFER_BIT,
|
||||
VK_PIPELINE_STAGE_BOTTOM_OF_PIPE_BIT, 0, release);
|
||||
}
|
||||
});
|
||||
}
|
||||
|
||||
void BufferCacheRuntime::UnifiedMemoryHostBarrier() {
|
||||
static constexpr VkMemoryBarrier HOST_BARRIER{
|
||||
.sType = VK_STRUCTURE_TYPE_MEMORY_BARRIER,
|
||||
.pNext = nullptr,
|
||||
.srcAccessMask = VK_ACCESS_TRANSFER_WRITE_BIT,
|
||||
.dstAccessMask = VK_ACCESS_HOST_READ_BIT,
|
||||
};
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
scheduler.Record([](vk::CommandBuffer cmdbuf) {
|
||||
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_TRANSFER_BIT, VK_PIPELINE_STAGE_HOST_BIT, 0,
|
||||
HOST_BARRIER);
|
||||
});
|
||||
}
|
||||
|
||||
StagingBufferRef BufferCacheRuntime::UploadStagingBuffer(size_t size) {
|
||||
return staging_pool.Request(size, MemoryUsage::Upload);
|
||||
}
|
||||
|
||||
@@ -101,7 +101,7 @@ public:
|
||||
|
||||
void TryEnableUnifiedMemory(void* base, size_t size,
|
||||
std::span<AHardwareBuffer* const> hardware_buffers,
|
||||
size_t hardware_buffer_window, size_t hardware_buffer_base);
|
||||
size_t hardware_buffer_window);
|
||||
|
||||
[[nodiscard]] bool HasUnifiedMemory() const noexcept {
|
||||
return unified_memory != nullptr && unified_memory->IsValid();
|
||||
@@ -111,18 +111,13 @@ public:
|
||||
return unified_memory ? unified_memory->GetSize() : 0;
|
||||
}
|
||||
|
||||
[[nodiscard]] u64 UnifiedMemoryBase() const noexcept {
|
||||
return unified_memory ? unified_memory->GetBaseOffset() : 0;
|
||||
}
|
||||
|
||||
[[nodiscard]] u64 UnifiedMemoryWindowSize() const noexcept {
|
||||
return unified_memory ? unified_memory->GetWindowSize() : 0;
|
||||
}
|
||||
|
||||
void CopyToUnifiedMemory(size_t window_index, VkBuffer src_buffer,
|
||||
std::span<const VideoCommon::BufferCopy> copies);
|
||||
|
||||
void UnifiedMemoryHostBarrier();
|
||||
[[nodiscard]] VkBuffer UnifiedMemoryWindowBuffer(size_t index) const noexcept {
|
||||
return unified_memory ? unified_memory->GetWindowBuffer(index) : VK_NULL_HANDLE;
|
||||
}
|
||||
|
||||
u64 CurrentTick();
|
||||
|
||||
|
||||
@@ -959,7 +959,7 @@ bool BlockLinearUnswizzle2DPass::IsSupported(const VideoCommon::ImageInfo& info)
|
||||
if (info.type != VideoCommon::ImageType::e2D) {
|
||||
return false;
|
||||
}
|
||||
if (info.resources.levels != 1 || info.resources.layers != 1) {
|
||||
if (info.resources.levels != 1) {
|
||||
return false;
|
||||
}
|
||||
if (info.num_samples > 1) {
|
||||
|
||||
@@ -346,7 +346,7 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
use_asynchronous_shaders{Settings::values.use_asynchronous_shaders.GetValue()},
|
||||
use_vulkan_pipeline_cache{Settings::values.use_vulkan_driver_pipeline_cache.GetValue()},
|
||||
workers(device.HasBrokenParallelShaderCompiling() ? 1ULL : GetTotalPipelineWorkers(),
|
||||
"VkPipelineBuilder", {}, Common::ThreadPlacement::Background),
|
||||
"VkPipelineBuilder"),
|
||||
serialization_thread(1, "VkPipelineSerialization", {},
|
||||
Common::ThreadPlacement::Background) {
|
||||
const auto& float_control{device.FloatControlProperties()};
|
||||
|
||||
@@ -266,7 +266,6 @@ void PresentManager::WaitPresent() {
|
||||
|
||||
void PresentManager::PresentThread(std::stop_token token) {
|
||||
Common::SetCurrentThreadName("VulkanPresent");
|
||||
Common::SetCurrentThreadPriority(Common::ThreadPriority::High);
|
||||
while (!token.stop_requested()) {
|
||||
std::unique_lock lock{queue_mutex};
|
||||
// Wait for presentation frames
|
||||
|
||||
@@ -239,40 +239,24 @@ RasterizerVulkan::RasterizerVulkan(Core::Frontend::EmuWindow& emu_window_, Tegra
|
||||
fence_manager(*this, gpu, texture_cache, buffer_cache, query_cache, device, scheduler),
|
||||
wfi_event(device.GetLogical().CreateEvent()) {
|
||||
scheduler.SetQueryCache(query_cache);
|
||||
if (Settings::values.use_unified_memory.GetValue() && device_memory.IsBackingShared()) {
|
||||
if (Settings::values.use_unified_memory.GetValue()) {
|
||||
buffer_cache_runtime.TryEnableUnifiedMemory(
|
||||
device_memory.GetPhysicalBase(), device_memory.GetPhysicalSize(),
|
||||
device_memory.GetBackingHardwareBuffers(),
|
||||
device_memory.GetBackingHardwareBufferWindowSize(),
|
||||
device_memory.GetBackingHardwareBufferBase());
|
||||
device_memory.GetBackingHardwareBufferWindowSize());
|
||||
}
|
||||
memory_allocator.SetReclaimCallback([this](u64 bytes) -> u64 {
|
||||
u64 freed = staging_pool.ReclaimMemory(bytes);
|
||||
if (freed < bytes) {
|
||||
freed += texture_cache.ReclaimMemory(bytes - freed, false);
|
||||
}
|
||||
if (freed < bytes) {
|
||||
freed += buffer_cache.ReclaimMemory(bytes - freed, false);
|
||||
}
|
||||
auto& master_semaphore = scheduler.GetMasterSemaphore();
|
||||
const u64 usage_before = device.GetMemoryBudgetInfo().allocation_bytes;
|
||||
master_semaphore.Refresh();
|
||||
const u64 completed = master_semaphore.KnownGpuTick();
|
||||
texture_cache.ReclaimDeferredResources(completed);
|
||||
buffer_cache.ReclaimDeferredResources(completed);
|
||||
vk::TickDeletionQueue(completed);
|
||||
const u64 usage_after = device.GetMemoryBudgetInfo().allocation_bytes;
|
||||
const u64 drained = usage_before > usage_after ? usage_before - usage_after : 0;
|
||||
if (drained >= bytes) {
|
||||
return drained;
|
||||
}
|
||||
const u64 remaining = bytes - drained;
|
||||
u64 evicted = staging_pool.ReclaimMemory(remaining);
|
||||
if (evicted < remaining) {
|
||||
evicted += texture_cache.ReclaimMemory(remaining - evicted, false);
|
||||
}
|
||||
if (evicted < remaining) {
|
||||
evicted += buffer_cache.ReclaimMemory(remaining - evicted, false);
|
||||
}
|
||||
master_semaphore.Refresh();
|
||||
const u64 completed_after = master_semaphore.KnownGpuTick();
|
||||
texture_cache.ReclaimDeferredResources(completed_after);
|
||||
buffer_cache.ReclaimDeferredResources(completed_after);
|
||||
vk::TickDeletionQueue(completed_after);
|
||||
return drained + evicted;
|
||||
vk::TickDeletionQueue(master_semaphore.KnownGpuTick());
|
||||
return freed;
|
||||
});
|
||||
}
|
||||
|
||||
|
||||
@@ -346,7 +346,6 @@ bool Scheduler::UpdateRescaling(bool is_rescaling) {
|
||||
|
||||
void Scheduler::WorkerThread(std::stop_token stop_token) {
|
||||
Common::SetCurrentThreadName("VulkanWorker");
|
||||
Common::SetCurrentThreadPriority(Common::ThreadPriority::VeryHigh);
|
||||
|
||||
const auto TryPopQueue{[this](auto& work) -> bool {
|
||||
if (work_queue.empty()) {
|
||||
|
||||
@@ -1880,8 +1880,6 @@ Image::Image(TextureCacheRuntime& runtime_, const ImageInfo& info_, GPUVAddr gpu
|
||||
default:
|
||||
break;
|
||||
}
|
||||
} else if (runtime->bl2d_unswizzle_pass &&
|
||||
BlockLinearUnswizzle2DPass::IsSupported(info)) {
|
||||
flags |= VideoCommon::ImageFlagBits::AcceleratedUpload;
|
||||
flags |= VideoCommon::ImageFlagBits::CostlyLoad;
|
||||
} else if (runtime->bl3db_unswizzle_pass &&
|
||||
|
||||
@@ -182,9 +182,10 @@ u64 TextureCache<P>::ReclaimMemory(u64 target_bytes, bool allow_download) {
|
||||
}
|
||||
const bool must_download = image.IsSafeDownload();
|
||||
if (must_download && True(image.flags & ImageFlagBits::BadOverlap)) {
|
||||
LOG_DEBUG(HW_GPU, "Recovering bad overlap on eviction: gpu_addr=0x{:x} fmt={} {}x{}x{}",
|
||||
image.gpu_addr, static_cast<u32>(image.info.format), image.info.size.width,
|
||||
image.info.size.height, image.info.size.depth);
|
||||
LOG_WARNING(HW_GPU,
|
||||
"Recovering bad overlap on eviction: gpu_addr=0x{:x} fmt={} {}x{}x{}",
|
||||
image.gpu_addr, static_cast<u32>(image.info.format), image.info.size.width,
|
||||
image.info.size.height, image.info.size.depth);
|
||||
}
|
||||
bool queued_download = false;
|
||||
if (must_download) {
|
||||
@@ -225,13 +226,6 @@ u64 TextureCache<P>::ReclaimMemory(u64 target_bytes, bool allow_download) {
|
||||
return freed;
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void TextureCache<P>::ReclaimDeferredResources(u64 completed_sync_point) {
|
||||
sentenced_images.Reclaim(completed_sync_point);
|
||||
sentenced_framebuffers.Reclaim(completed_sync_point);
|
||||
sentenced_image_view.Reclaim(completed_sync_point);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void TextureCache<P>::EnsureHeadroom(bool allow_download) {
|
||||
if (reclaim_stalled) {
|
||||
@@ -256,10 +250,12 @@ template <class P>
|
||||
void TextureCache<P>::TickFrame() {
|
||||
usage_refresh_countdown = 0;
|
||||
reclaim_stalled = false;
|
||||
EnsureHeadroom(true);
|
||||
const u64 completed_sync_point = runtime.CompletedSyncPoint();
|
||||
TickEvictionDownloads(completed_sync_point);
|
||||
ReclaimDeferredResources(completed_sync_point);
|
||||
EnsureHeadroom(true);
|
||||
sentenced_images.Reclaim(completed_sync_point);
|
||||
sentenced_framebuffers.Reclaim(completed_sync_point);
|
||||
sentenced_image_view.Reclaim(completed_sync_point);
|
||||
TickAsyncDecode();
|
||||
TickAsyncUnswizzle();
|
||||
|
||||
@@ -1245,6 +1241,8 @@ void TextureCache<P>::UploadImageContents(Image& image, StagingBuffer& staging)
|
||||
return;
|
||||
}
|
||||
|
||||
gpu_memory->FlushRegion(gpu_addr, image.guest_size_bytes,
|
||||
VideoCommon::CacheType::NoTextureCache);
|
||||
Tegra::Memory::GpuGuestMemory<u8, Tegra::Memory::GuestMemoryFlags::UnsafeRead> swizzle_data(
|
||||
*gpu_memory, gpu_addr, image.guest_size_bytes, &swizzle_data_buffer);
|
||||
if (True(image.flags & ImageFlagBits::Converted)) {
|
||||
@@ -1610,7 +1608,7 @@ ImageId TextureCache<P>::InsertImage(const ImageInfo& info, GPUVAddr gpu_addr,
|
||||
}
|
||||
}
|
||||
ASSERT_MSG(cpu_addr, "Tried to insert an image to an invalid gpu_addr=0x{:x}", gpu_addr);
|
||||
const ImageId image_id = JoinImages(info, gpu_addr, *cpu_addr, options);
|
||||
const ImageId image_id = JoinImages(info, gpu_addr, *cpu_addr);
|
||||
const Image& image = slot_images[image_id];
|
||||
// Using "image.gpu_addr" instead of "gpu_addr" is important because it might be different
|
||||
const auto [it, is_new] = image_allocs_table.try_emplace(image.gpu_addr);
|
||||
@@ -1622,8 +1620,7 @@ ImageId TextureCache<P>::InsertImage(const ImageInfo& info, GPUVAddr gpu_addr,
|
||||
}
|
||||
|
||||
template <class P>
|
||||
ImageId TextureCache<P>::JoinImages(const ImageInfo& info, GPUVAddr gpu_addr, DAddr cpu_addr,
|
||||
RelaxedOptions options) {
|
||||
ImageId TextureCache<P>::JoinImages(const ImageInfo& info, GPUVAddr gpu_addr, DAddr cpu_addr) {
|
||||
EnsureHeadroom(false);
|
||||
ImageInfo new_info = info;
|
||||
const size_t size_bytes = CalculateGuestSizeInBytes(new_info);
|
||||
@@ -1635,11 +1632,9 @@ ImageId TextureCache<P>::JoinImages(const ImageInfo& info, GPUVAddr gpu_addr, DA
|
||||
join_right_aliased_ids.clear();
|
||||
join_ignore_textures.clear();
|
||||
join_bad_overlap_ids.clear();
|
||||
join_stale_ids.clear();
|
||||
join_copies_to_do.clear();
|
||||
join_alias_indices.clear();
|
||||
const bool this_is_linear = info.type == ImageType::Linear;
|
||||
const bool is_render_target = True(options & RelaxedOptions::RenderTarget);
|
||||
const auto region_check = [&](ImageId overlap_id, ImageBase& overlap) {
|
||||
if (True(overlap.flags & ImageFlagBits::Remapped)) {
|
||||
join_ignore_textures.insert(overlap_id);
|
||||
@@ -1679,9 +1674,6 @@ ImageId TextureCache<P>::JoinImages(const ImageInfo& info, GPUVAddr gpu_addr, DA
|
||||
join_right_aliased_ids.push_back(overlap_id);
|
||||
overlap.flags |= ImageFlagBits::Alias;
|
||||
join_copies_to_do.emplace_back(JoinCopy{true, overlap_id});
|
||||
} else if (is_render_target && IsStaleReallocation(new_info, overlap, gpu_addr) &&
|
||||
slot_images[overlap_id].allocation_tick != frame_tick) {
|
||||
join_stale_ids.push_back(overlap_id);
|
||||
} else {
|
||||
join_bad_overlap_ids.push_back(overlap_id);
|
||||
}
|
||||
@@ -1775,19 +1767,6 @@ ImageId TextureCache<P>::JoinImages(const ImageInfo& info, GPUVAddr gpu_addr, DA
|
||||
DeleteImage(overlap_id);
|
||||
}
|
||||
|
||||
for (const ImageId stale_id : join_stale_ids) {
|
||||
Image& stale = slot_images[stale_id];
|
||||
LOG_DEBUG(HW_GPU,
|
||||
"Retiring resized image: gpu_addr=0x{:x} fmt={} {}x{} -> {}x{}", stale.gpu_addr,
|
||||
static_cast<u32>(stale.info.format), stale.info.size.width,
|
||||
stale.info.size.height, new_image.info.size.width, new_image.info.size.height);
|
||||
if (True(stale.flags & ImageFlagBits::Tracked)) {
|
||||
UntrackImage(stale, stale_id);
|
||||
}
|
||||
UnregisterImage(stale_id);
|
||||
DeleteImage(stale_id);
|
||||
}
|
||||
|
||||
// TODO: Only upload what we need
|
||||
RefreshContents(new_image, new_image_id);
|
||||
|
||||
@@ -1839,17 +1818,17 @@ ImageId TextureCache<P>::JoinImages(const ImageInfo& info, GPUVAddr gpu_addr, DA
|
||||
const bool aliased_is_bad = True(aliased.flags & ImageFlagBits::BadOverlap);
|
||||
const bool new_is_bad = True(new_image.flags & ImageFlagBits::BadOverlap);
|
||||
if ((!aliased_was_bad && aliased_is_bad) || (!new_was_bad && new_is_bad)) {
|
||||
LOG_DEBUG(HW_GPU,
|
||||
"Bad overlap: existing gpu_addr={:#x} {}x{}x{} fmt={} type={} rt={} | "
|
||||
"incoming gpu_addr={:#x} {}x{}x{} fmt={} type={} rt={}",
|
||||
aliased.gpu_addr, aliased.info.size.width, aliased.info.size.height,
|
||||
aliased.info.size.depth, static_cast<u32>(aliased.info.format),
|
||||
static_cast<u32>(aliased.info.type),
|
||||
True(aliased.flags & ImageFlagBits::GpuModified),
|
||||
new_image.gpu_addr, new_image.info.size.width, new_image.info.size.height,
|
||||
new_image.info.size.depth, static_cast<u32>(new_image.info.format),
|
||||
static_cast<u32>(new_image.info.type),
|
||||
True(new_image.flags & ImageFlagBits::GpuModified));
|
||||
LOG_WARNING(HW_GPU,
|
||||
"Bad overlap: existing gpu_addr={:#x} {}x{}x{} fmt={} type={} rt={} | "
|
||||
"incoming gpu_addr={:#x} {}x{}x{} fmt={} type={} rt={}",
|
||||
aliased.gpu_addr, aliased.info.size.width, aliased.info.size.height,
|
||||
aliased.info.size.depth, static_cast<u32>(aliased.info.format),
|
||||
static_cast<u32>(aliased.info.type),
|
||||
True(aliased.flags & ImageFlagBits::GpuModified),
|
||||
new_image.gpu_addr, new_image.info.size.width, new_image.info.size.height,
|
||||
new_image.info.size.depth, static_cast<u32>(new_image.info.format),
|
||||
static_cast<u32>(new_image.info.type),
|
||||
True(new_image.flags & ImageFlagBits::GpuModified));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2141,7 +2120,7 @@ ImageViewId TextureCache<P>::FindRenderTargetView(const ImageInfo& info, GPUVAdd
|
||||
bool delete_state = has_deleted_images;
|
||||
do {
|
||||
has_deleted_images = false;
|
||||
image_id = FindOrInsertImage(info, gpu_addr, RelaxedOptions::RenderTarget);
|
||||
image_id = FindOrInsertImage(info, gpu_addr);
|
||||
delete_state |= has_deleted_images;
|
||||
} while (has_deleted_images);
|
||||
has_deleted_images = delete_state;
|
||||
|
||||
@@ -159,8 +159,6 @@ public:
|
||||
|
||||
u64 ReclaimMemory(u64 target_bytes, bool allow_download);
|
||||
|
||||
void ReclaimDeferredResources(u64 completed_sync_point);
|
||||
|
||||
/// Return a constant reference to the given image view id
|
||||
[[nodiscard]] const ImageView& GetImageView(ImageViewId id) const noexcept;
|
||||
|
||||
@@ -342,8 +340,7 @@ private:
|
||||
|
||||
/// Create a new image and join perfectly matching existing images
|
||||
/// Remove joined images from the cache
|
||||
[[nodiscard]] ImageId JoinImages(const ImageInfo& info, GPUVAddr gpu_addr, DAddr cpu_addr,
|
||||
RelaxedOptions options);
|
||||
[[nodiscard]] ImageId JoinImages(const ImageInfo& info, GPUVAddr gpu_addr, DAddr cpu_addr);
|
||||
|
||||
[[nodiscard]] ImageId FindDMAImage(const ImageInfo& info, GPUVAddr gpu_addr);
|
||||
|
||||
@@ -535,7 +532,7 @@ private:
|
||||
u64 last_sampler_gc_frame = (std::numeric_limits<u64>::max)();
|
||||
|
||||
Common::ThreadWorker texture_decode_worker{1, "TextureDecoder", {},
|
||||
Common::ThreadPlacement::Efficiency};
|
||||
Common::ThreadPlacement::Background};
|
||||
std::vector<std::unique_ptr<AsyncDecodeContext>> async_decodes;
|
||||
|
||||
std::deque<PendingUnswizzle> unswizzle_queue;
|
||||
@@ -548,7 +545,6 @@ private:
|
||||
boost::container::small_vector<ImageId, 4> join_right_aliased_ids;
|
||||
ankerl::unordered_dense::set<ImageId> join_ignore_textures;
|
||||
boost::container::small_vector<ImageId, 4> join_bad_overlap_ids;
|
||||
boost::container::small_vector<ImageId, 4> join_stale_ids;
|
||||
struct JoinCopy {
|
||||
bool is_alias;
|
||||
ImageId id;
|
||||
|
||||
@@ -57,7 +57,6 @@ enum class RelaxedOptions : u32 {
|
||||
Format = 1 << 1,
|
||||
Samples = 1 << 2,
|
||||
ForceBrokenViews = 1 << 3,
|
||||
RenderTarget = 1 << 4,
|
||||
};
|
||||
DECLARE_ENUM_FLAG_OPERATORS(RelaxedOptions)
|
||||
|
||||
|
||||
@@ -1194,35 +1194,6 @@ bool IsLayerStrideCompatible(const ImageInfo& lhs, const ImageInfo& rhs) {
|
||||
return false;
|
||||
}
|
||||
|
||||
bool IsStaleReallocation(const ImageInfo& new_info, const ImageBase& overlap,
|
||||
GPUVAddr gpu_addr) noexcept {
|
||||
if (overlap.gpu_addr != gpu_addr) {
|
||||
return false;
|
||||
}
|
||||
const ImageInfo& info = overlap.info;
|
||||
if (new_info.type != ImageType::e2D || info.type != ImageType::e2D) {
|
||||
return false;
|
||||
}
|
||||
if (new_info.resources.levels != 1 || info.resources.levels != 1) {
|
||||
return false;
|
||||
}
|
||||
if (new_info.resources.layers != info.resources.layers) {
|
||||
return false;
|
||||
}
|
||||
if (new_info.block != info.block || new_info.num_samples != info.num_samples) {
|
||||
return false;
|
||||
}
|
||||
if (new_info.tile_width_spacing != info.tile_width_spacing) {
|
||||
return false;
|
||||
}
|
||||
if (BytesPerBlock(new_info.format) != BytesPerBlock(info.format) ||
|
||||
DefaultBlockWidth(new_info.format) != DefaultBlockWidth(info.format) ||
|
||||
DefaultBlockHeight(new_info.format) != DefaultBlockHeight(info.format)) {
|
||||
return false;
|
||||
}
|
||||
return new_info.size.width != info.size.width || new_info.size.height != info.size.height;
|
||||
}
|
||||
|
||||
std::optional<SubresourceBase> FindSubresource(const ImageInfo& candidate, const ImageBase& image,
|
||||
GPUVAddr candidate_addr, RelaxedOptions options,
|
||||
bool broken_views, bool native_bgr) {
|
||||
|
||||
@@ -104,9 +104,6 @@ void SwizzleImage(Tegra::MemoryManager& gpu_memory, GPUVAddr gpu_addr, const Ima
|
||||
|
||||
[[nodiscard]] bool IsLayerStrideCompatible(const ImageInfo& lhs, const ImageInfo& rhs);
|
||||
|
||||
[[nodiscard]] bool IsStaleReallocation(const ImageInfo& new_info, const ImageBase& overlap,
|
||||
GPUVAddr gpu_addr) noexcept;
|
||||
|
||||
[[nodiscard]] std::optional<SubresourceBase> FindSubresource(const ImageInfo& candidate,
|
||||
const ImageBase& image,
|
||||
GPUVAddr candidate_addr,
|
||||
|
||||
@@ -11,7 +11,7 @@ namespace Tegra::Texture {
|
||||
Common::ThreadWorker& GetThreadWorkers() {
|
||||
static Common::ThreadWorker workers{(std::max)(std::thread::hardware_concurrency(), 2U) / 2,
|
||||
"ImageTranscode", {},
|
||||
Common::ThreadPlacement::Efficiency};
|
||||
Common::ThreadPlacement::Background};
|
||||
|
||||
return workers;
|
||||
}
|
||||
|
||||
@@ -1256,11 +1256,6 @@ bool Device::GetSuitability(bool requires_swapchain) {
|
||||
VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_EXTERNAL_MEMORY_HOST_PROPERTIES_EXT;
|
||||
SetNext(next, properties.external_memory_host);
|
||||
}
|
||||
if (extensions.maintenance3 || instance_version >= VK_API_VERSION_1_1) {
|
||||
properties.maintenance3.sType =
|
||||
VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MAINTENANCE_3_PROPERTIES;
|
||||
SetNext(next, properties.maintenance3);
|
||||
}
|
||||
if (extensions.maintenance4 || features.maintenance4.maintenance4) {
|
||||
properties.maintenance4.sType =
|
||||
VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_MAINTENANCE_4_PROPERTIES;
|
||||
|
||||
@@ -416,8 +416,7 @@ FN_MAX_LIMIT_LIST
|
||||
|
||||
/// Returns true if descriptor aliasing is natively supported.
|
||||
bool IsDescriptorAliasingSupported() const {
|
||||
return features.descriptor_indexing.descriptorBindingPartiallyBound &&
|
||||
features.descriptor_indexing.descriptorBindingVariableDescriptorCount;
|
||||
return GetDriverID() != VK_DRIVER_ID_QUALCOMM_PROPRIETARY;
|
||||
}
|
||||
|
||||
bool IsSampledImageArrayNonUniformIndexingSupported() const {
|
||||
@@ -948,10 +947,6 @@ FN_MAX_LIMIT_LIST
|
||||
return properties.maintenance4.maxBufferSize;
|
||||
}
|
||||
|
||||
u64 GetMaxMemoryAllocationSize() const {
|
||||
return properties.maintenance3.maxMemoryAllocationSize;
|
||||
}
|
||||
|
||||
bool HasTimelineSemaphore() const;
|
||||
|
||||
/// Returns true if the device supports VK_KHR_synchronization2.
|
||||
@@ -1300,7 +1295,6 @@ private:
|
||||
VkPhysicalDeviceDescriptorBufferPropertiesEXT descriptor_buffer{};
|
||||
VkPhysicalDeviceSubgroupSizeControlProperties subgroup_size_control{};
|
||||
VkPhysicalDeviceTransformFeedbackPropertiesEXT transform_feedback{};
|
||||
VkPhysicalDeviceMaintenance3Properties maintenance3{};
|
||||
VkPhysicalDeviceMaintenance4Properties maintenance4{};
|
||||
VkPhysicalDeviceMaintenance5PropertiesKHR maintenance5{};
|
||||
VkPhysicalDeviceDepthStencilResolveProperties depth_stencil_resolve{};
|
||||
|
||||
@@ -95,22 +95,15 @@ namespace Vulkan {
|
||||
|
||||
HostMemoryImport::HostMemoryImport(const Device &device_, void *base, size_t size,
|
||||
std::span<AHardwareBuffer *const> hardware_buffers,
|
||||
size_t hardware_buffer_window, size_t hardware_buffer_base)
|
||||
size_t hardware_buffer_window)
|
||||
: device{device_} {
|
||||
if (ImportHardwareBuffers(hardware_buffers, hardware_buffer_window, hardware_buffer_base,
|
||||
size)) {
|
||||
return;
|
||||
}
|
||||
if (!hardware_buffers.empty()) {
|
||||
LOG_INFO(Render_Vulkan,
|
||||
"Unified memory disabled, guest memory is backed by hardware buffers that "
|
||||
"could not be imported");
|
||||
return;
|
||||
}
|
||||
if (ImportHostPointer(base, size)) {
|
||||
return;
|
||||
}
|
||||
LOG_INFO(Render_Vulkan, "Unified memory disabled, no host memory import path");
|
||||
ImportHardwareBuffers(hardware_buffers, hardware_buffer_window, size);
|
||||
if (windows.empty()) {
|
||||
LOG_INFO(Render_Vulkan, "Unified memory disabled, no host memory import path");
|
||||
}
|
||||
}
|
||||
|
||||
bool HostMemoryImport::ImportHostPointer(void *base, size_t size) {
|
||||
@@ -131,13 +124,8 @@ namespace Vulkan {
|
||||
VkDeviceSize candidate_window = 1_GiB;
|
||||
const u64 max_buffer_size = device.GetMaxBufferSize();
|
||||
if (max_buffer_size != 0 && max_buffer_size < candidate_window) {
|
||||
candidate_window = max_buffer_size;
|
||||
candidate_window = Common::AlignDown(max_buffer_size, alignment);
|
||||
}
|
||||
const u64 max_allocation_size = device.GetMaxMemoryAllocationSize();
|
||||
if (max_allocation_size != 0 && max_allocation_size < candidate_window) {
|
||||
candidate_window = max_allocation_size;
|
||||
}
|
||||
candidate_window = Common::AlignDown(candidate_window, alignment);
|
||||
if (candidate_window == 0) {
|
||||
return false;
|
||||
}
|
||||
@@ -238,31 +226,19 @@ namespace Vulkan {
|
||||
return true;
|
||||
}
|
||||
|
||||
bool HostMemoryImport::ImportHardwareBuffers(
|
||||
void HostMemoryImport::ImportHardwareBuffers(
|
||||
[[maybe_unused]] std::span<AHardwareBuffer *const> hardware_buffers,
|
||||
[[maybe_unused]] size_t hardware_buffer_window,
|
||||
[[maybe_unused]] size_t hardware_buffer_base, [[maybe_unused]] size_t size) {
|
||||
[[maybe_unused]] size_t hardware_buffer_window, [[maybe_unused]] size_t size) {
|
||||
#ifdef __ANDROID__
|
||||
if (hardware_buffers.empty() || hardware_buffer_window == 0 ||
|
||||
!device.IsExtExternalMemoryAhbSupported()) {
|
||||
return false;
|
||||
}
|
||||
const u64 max_allocation_size = device.GetMaxMemoryAllocationSize();
|
||||
if (max_allocation_size != 0 && hardware_buffer_window > max_allocation_size) {
|
||||
LOG_WARNING(Render_Vulkan,
|
||||
"Hardware buffer windows of {} MiB exceed the {} MiB allocation limit",
|
||||
hardware_buffer_window >> 20, max_allocation_size >> 20);
|
||||
return false;
|
||||
}
|
||||
if (hardware_buffer_base >= size) {
|
||||
return false;
|
||||
return;
|
||||
}
|
||||
const auto &logical = device.GetLogical();
|
||||
const auto memory_props = device.GetPhysical().GetMemoryProperties().memoryProperties;
|
||||
window_size = hardware_buffer_window;
|
||||
base_offset = hardware_buffer_base;
|
||||
for (size_t i = 0; i < hardware_buffers.size(); ++i) {
|
||||
const size_t offset = hardware_buffer_base + i * hardware_buffer_window;
|
||||
const size_t offset = i * hardware_buffer_window;
|
||||
if (offset >= size) {
|
||||
break;
|
||||
}
|
||||
@@ -344,19 +320,11 @@ namespace Vulkan {
|
||||
});
|
||||
imported_size += static_cast<size_t>(window_len);
|
||||
}
|
||||
if (windows.empty()) {
|
||||
LOG_INFO(Render_Vulkan, "Hardware buffer import failed");
|
||||
window_size = 0;
|
||||
base_offset = 0;
|
||||
return false;
|
||||
if (!windows.empty()) {
|
||||
LOG_INFO(Render_Vulkan,
|
||||
"Imported {} MiB of guest memory via hardware buffers in {} windows",
|
||||
imported_size >> 20, windows.size());
|
||||
}
|
||||
foreign_ownership = true;
|
||||
LOG_INFO(Render_Vulkan,
|
||||
"Imported {} MiB of guest memory at {:#x} via hardware buffers in {} windows",
|
||||
imported_size >> 20, base_offset, windows.size());
|
||||
return true;
|
||||
#else
|
||||
return false;
|
||||
#endif
|
||||
}
|
||||
|
||||
|
||||
@@ -46,7 +46,7 @@ namespace Vulkan {
|
||||
public:
|
||||
explicit HostMemoryImport(const Device &device_, void *base, size_t size,
|
||||
std::span<AHardwareBuffer *const> hardware_buffers,
|
||||
size_t hardware_buffer_window, size_t hardware_buffer_base);
|
||||
size_t hardware_buffer_window);
|
||||
|
||||
~HostMemoryImport();
|
||||
|
||||
@@ -62,14 +62,6 @@ namespace Vulkan {
|
||||
return imported_size;
|
||||
}
|
||||
|
||||
[[nodiscard]] size_t GetBaseOffset() const noexcept {
|
||||
return base_offset;
|
||||
}
|
||||
|
||||
[[nodiscard]] bool NeedsForeignOwnershipTransfer() const noexcept {
|
||||
return foreign_ownership;
|
||||
}
|
||||
|
||||
[[nodiscard]] VkDeviceSize GetWindowSize() const noexcept {
|
||||
return window_size;
|
||||
}
|
||||
@@ -90,16 +82,13 @@ namespace Vulkan {
|
||||
|
||||
bool ImportHostPointer(void *base, size_t size);
|
||||
|
||||
bool ImportHardwareBuffers(std::span<AHardwareBuffer *const> hardware_buffers,
|
||||
size_t hardware_buffer_window, size_t hardware_buffer_base,
|
||||
size_t size);
|
||||
void ImportHardwareBuffers(std::span<AHardwareBuffer *const> hardware_buffers,
|
||||
size_t hardware_buffer_window, size_t size);
|
||||
|
||||
const Device &device;
|
||||
std::vector<Window> windows;
|
||||
VkDeviceSize window_size{};
|
||||
size_t imported_size{};
|
||||
size_t base_offset{};
|
||||
bool foreign_ownership{};
|
||||
};
|
||||
|
||||
/// Memory allocator container.
|
||||
|
||||
Reference in New Issue
Block a user