Compare commits

..

26 Commits

Author SHA1 Message Date
CamilleLaVey 29825a3290 AAaaa 2026-07-19 16:16:56 -04:00
CamilleLaVey 10e924c8cb Fix build 2026-07-19 01:07:06 -04:00
CamilleLaVey f67a300a69 [TEST] ASTC to fragment 2026-07-19 01:00:09 -04:00
CamilleLaVey 1330abd2b6 [TEST] Changes on the ASTC enconding path 2026-07-18 23:45:01 -04:00
CamilleLaVey db708d2ea0 [TEST] Remove certain annoying finish on compute pass KEEEEEEEEEEEEEEELLLLLLLLLLEEEEEEEEEEEEEEEEEEEEEEEEEER 2026-07-18 21:54:18 -04:00
CamilleLaVey cf76977977 Revert "[TEST] Extend coalescing to more gpu to cpu regions" 2026-07-18 21:28:17 -04:00
CamilleLaVey 20e6c163a6 [TEST] Extend coalescing to more gpu to cpu regions 2026-07-18 17:49:22 -04:00
CamilleLaVey dfb2fa717e I despite you clang 2026-07-18 16:51:03 -04:00
CamilleLaVey 787f80e05c [TEST] Extend coalescing to more cpu regions 2026-07-18 14:49:00 -04:00
CamilleLaVey 142eeb0b8a fix build 2026-07-18 02:20:32 -04:00
CamilleLaVey c394eef32e [TEST] Depth clip bugs + tessellation 2026-07-18 02:11:33 -04:00
CamilleLaVey a2d24b3eb7 [TEST] Updated list of QCOM driver removed features 2026-07-18 01:46:03 -04:00
CamilleLaVey 59ae72c97d [TEST] Review rasters on Maxwell values 2026-07-18 01:16:28 -04:00
CamilleLaVey 723afa7d46 [TEST] Hunting down recursive mutex 8 2026-07-18 00:00:32 -04:00
CamilleLaVey 8d96b2e894 [TEST] Caching for texture + pages on NCE
(cherry picked from commit 9c313fb787)
2026-07-17 23:19:46 -04:00
CamilleLaVey 1b01f5c93a [TEST] Coalesce NCE fault write.
(cherry picked from commit eec29b83f3)
2026-07-17 23:19:46 -04:00
CamilleLaVey 43ed5cf9b5 [TEST] Hunting down recursive mutex 7
(cherry picked from commit 4956bc86c3)
2026-07-17 23:19:46 -04:00
CamilleLaVey 3c2f298101 [TEST] Adjustments on CommandPools + ResetQueryPool
(cherry picked from commit e8b1dc7c0b)
2026-07-17 23:19:45 -04:00
CamilleLaVey cece80d688 [TEST] Remove unnecessary memory upload
(cherry picked from commit e81d170458)
2026-07-17 23:19:45 -04:00
CamilleLaVey 0b1c69bc39 [TEST] Remove unnecessary submit
(cherry picked from commit 7ed7e5e31d)
2026-07-17 23:19:45 -04:00
CamilleLaVey f3bd7aa402 [TEST] Hunting down recursive mutex 6
(cherry picked from commit eb32b8766a)
2026-07-17 23:19:45 -04:00
CamilleLaVey e0a742277a [TEST] Hunting down recursive mutex 5
(cherry picked from commit a484e6c34b)
2026-07-17 23:19:45 -04:00
CamilleLaVey aec1658697 [TEST] Hunting down recursive mutex 4
(cherry picked from commit 84490a7d6f)
2026-07-17 23:19:45 -04:00
CamilleLaVey 65e45f0a9a [TEST] Hunting down recursive mutex 3
(cherry picked from commit 933f79af95)
2026-07-17 23:19:44 -04:00
CamilleLaVey a8f4fdd20f [TEST] Hunting down recursive mutex 2
(cherry picked from commit f532357793)
2026-07-17 23:19:44 -04:00
CamilleLaVey b24d8b3912 [TEST] Hunting down recursive mutex 1
(cherry picked from commit ab92e5fa52)
2026-07-17 23:19:43 -04:00
68 changed files with 2520 additions and 782 deletions
-5
View File
@@ -1,5 +0,0 @@
- [ ] I have read and followed the [Contribution Guidelines](https://git.eden-emu.dev/eden-emu/eden/src/branch/master/CONTRIBUTING.md#code-contributions).
- [ ] I have read and followed the [AI Policy](https://git.eden-emu.dev/eden-emu/eden/src/branch/master/docs/policies/AI.md)
- [ ] I have read and followed the [Coding Guidelines](https://git.eden-emu.dev/eden-emu/eden/src/branch/master/docs/policies/Coding.md) to the best of my ability.
-------------------
-2
View File
@@ -2,8 +2,6 @@
# SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project # SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
# SPDX-License-Identifier: GPL-3.0-or-later # SPDX-License-Identifier: GPL-3.0-or-later
# SPDX-FileCopyrightText: 2018 yuzu Emulator Project # SPDX-FileCopyrightText: 2018 yuzu Emulator Project
# SPDX-License-Identifier: GPL-2.0-or-later # SPDX-License-Identifier: GPL-2.0-or-later
--> -->
@@ -594,7 +594,7 @@ abstract class SettingsItem(
IntSetting.ANDROID_PIPELINE_WORKERS, IntSetting.ANDROID_PIPELINE_WORKERS,
titleId = R.string.pipeline_worker_cores, titleId = R.string.pipeline_worker_cores,
descriptionId = R.string.pipeline_worker_cores_description, descriptionId = R.string.pipeline_worker_cores_description,
min = 4, min = 1,
max = 8, max = 8,
units = "cores" units = "cores"
) )
@@ -147,7 +147,7 @@ namespace AndroidSettings {
&show_performance_overlay}; &show_performance_overlay};
Settings::Setting<s32> pipeline_worker_count{linkage, 4, "pipeline_worker_count", Settings::Setting<s32> pipeline_worker_count{linkage, 2, "pipeline_worker_count",
Settings::Category::Android, Settings::Category::Android,
Settings::Specialization::Default, Settings::Specialization::Default,
true, true,
@@ -1000,6 +1000,7 @@
<!-- Renderer Accuracy --> <!-- Renderer Accuracy -->
<string name="renderer_accuracy_low">سريع</string> <string name="renderer_accuracy_low">سريع</string>
<string name="renderer_accuracy_medium">متوازن</string>
<string name="renderer_accuracy_high">دقيق</string> <string name="renderer_accuracy_high">دقيق</string>
<!-- DMA Accuracy --> <!-- DMA Accuracy -->
@@ -707,6 +707,7 @@
<!-- Renderer Accuracy --> <!-- Renderer Accuracy -->
<string name="renderer_accuracy_low">Rychlý</string> <string name="renderer_accuracy_low">Rychlý</string>
<string name="renderer_accuracy_medium">Vyvážený</string>
<string name="renderer_accuracy_high">Přesný</string> <string name="renderer_accuracy_high">Přesný</string>
<!-- DMA Accuracy --> <!-- DMA Accuracy -->
@@ -927,6 +927,7 @@ Wirklich fortfahren?</string>
<!-- Renderer Accuracy --> <!-- Renderer Accuracy -->
<string name="renderer_accuracy_low">Schnell</string> <string name="renderer_accuracy_low">Schnell</string>
<string name="renderer_accuracy_medium">Ausgeglichen</string>
<string name="renderer_accuracy_high">Genau</string> <string name="renderer_accuracy_high">Genau</string>
<!-- DMA Accuracy --> <!-- DMA Accuracy -->
@@ -993,6 +993,7 @@
<!-- Renderer Accuracy --> <!-- Renderer Accuracy -->
<string name="renderer_accuracy_low">Rápido</string> <string name="renderer_accuracy_low">Rápido</string>
<string name="renderer_accuracy_medium">Equilibrado</string>
<string name="renderer_accuracy_high">Preciso</string> <string name="renderer_accuracy_high">Preciso</string>
<!-- DMA Accuracy --> <!-- DMA Accuracy -->
@@ -939,6 +939,7 @@
<!-- Renderer Accuracy --> <!-- Renderer Accuracy -->
<string name="renderer_accuracy_low">Rapide</string> <string name="renderer_accuracy_low">Rapide</string>
<string name="renderer_accuracy_medium">Moyen</string>
<string name="renderer_accuracy_high">Précis</string> <string name="renderer_accuracy_high">Précis</string>
<!-- DMA Accuracy --> <!-- DMA Accuracy -->
@@ -893,6 +893,7 @@
<!-- Renderer Accuracy --> <!-- Renderer Accuracy -->
<string name="renderer_accuracy_low">Szybkie</string> <string name="renderer_accuracy_low">Szybkie</string>
<string name="renderer_accuracy_medium">Zrównoważony</string>
<string name="renderer_accuracy_high">Dokładny</string> <string name="renderer_accuracy_high">Dokładny</string>
<!-- DMA Accuracy --> <!-- DMA Accuracy -->
@@ -840,6 +840,7 @@
<string name="renderer_none">Nenhum</string> <string name="renderer_none">Nenhum</string>
<string name="renderer_accuracy_medium">Média</string>
<string name="renderer_accuracy_high">Alta</string> <string name="renderer_accuracy_high">Alta</string>
<!-- DMA Accuracy --> <!-- DMA Accuracy -->
@@ -983,6 +983,7 @@
<!-- Renderer Accuracy --> <!-- Renderer Accuracy -->
<string name="renderer_accuracy_low">Быстрый</string> <string name="renderer_accuracy_low">Быстрый</string>
<string name="renderer_accuracy_medium">Сбалансированный</string>
<string name="renderer_accuracy_high">Точный</string> <string name="renderer_accuracy_high">Точный</string>
<!-- DMA Accuracy --> <!-- DMA Accuracy -->
@@ -986,6 +986,7 @@
<!-- Renderer Accuracy --> <!-- Renderer Accuracy -->
<string name="renderer_accuracy_low">Швидко</string> <string name="renderer_accuracy_low">Швидко</string>
<string name="renderer_accuracy_medium">Збалансовано</string>
<string name="renderer_accuracy_high">Точно</string> <string name="renderer_accuracy_high">Точно</string>
<!-- DMA Accuracy --> <!-- DMA Accuracy -->
@@ -990,6 +990,7 @@
<!-- Renderer Accuracy --> <!-- Renderer Accuracy -->
<string name="renderer_accuracy_low">快速</string> <string name="renderer_accuracy_low">快速</string>
<string name="renderer_accuracy_medium">均衡</string>
<string name="renderer_accuracy_high">精确</string> <string name="renderer_accuracy_high">精确</string>
<!-- DMA Accuracy --> <!-- DMA Accuracy -->
@@ -843,6 +843,7 @@
<string name="renderer_none">無</string> <string name="renderer_none">無</string>
<string name="renderer_accuracy_medium">平衡</string>
<string name="renderer_accuracy_high">準確</string> <string name="renderer_accuracy_high">準確</string>
<!-- DMA Accuracy --> <!-- DMA Accuracy -->
@@ -103,12 +103,14 @@
<string-array name="rendererAccuracyNames"> <string-array name="rendererAccuracyNames">
<item>@string/renderer_accuracy_low</item> <item>@string/renderer_accuracy_low</item>
<item>@string/renderer_accuracy_medium</item>
<item>@string/renderer_accuracy_high</item> <item>@string/renderer_accuracy_high</item>
</string-array> </string-array>
<integer-array name="rendererAccuracyValues"> <integer-array name="rendererAccuracyValues">
<item>0</item> <item>0</item>
<item>1</item> <item>1</item>
<item>2</item>
</integer-array> </integer-array>
<!-- VRAM USAGE MODE CHOICES --> <!-- VRAM USAGE MODE CHOICES -->
@@ -479,7 +479,7 @@
<string name="advanced">Advanced</string> <string name="advanced">Advanced</string>
<string name="renderer_accuracy">GPU Mode</string> <string name="renderer_accuracy">GPU Mode</string>
<string name="renderer_accuracy_description">Controls the GPU emulation mode. Most games render fine with Fast, but Accurate is still required for some. Particles tend to only render correctly with Accurate mode.</string> <string name="renderer_accuracy_description">Controls the GPU emulation mode. Most games render fine with Fast or Balanced modes, but Accurate is still required for some. Particles tend to only render correctly with Accurate mode.</string>
<string name="dma_accuracy">DMA Accuracy</string> <string name="dma_accuracy">DMA Accuracy</string>
<string name="dma_accuracy_description">Controls the DMA precision accuracy. Safe precision can fix issues in some games, but it can also impact performance in some cases. If unsure, leave this on Default.</string> <string name="dma_accuracy_description">Controls the DMA precision accuracy. Safe precision can fix issues in some games, but it can also impact performance in some cases. If unsure, leave this on Default.</string>
<string name="gpu_fence_behavior">GPU Fence Behavior</string> <string name="gpu_fence_behavior">GPU Fence Behavior</string>
@@ -1039,6 +1039,7 @@
<!-- Renderer Accuracy --> <!-- Renderer Accuracy -->
<string name="renderer_accuracy_low">Fast</string> <string name="renderer_accuracy_low">Fast</string>
<string name="renderer_accuracy_medium">Balanced</string>
<string name="renderer_accuracy_high">Accurate</string> <string name="renderer_accuracy_high">Accurate</string>
<!-- DMA Accuracy --> <!-- DMA Accuracy -->
-23
View File
@@ -476,29 +476,6 @@ std::string SanitizePath(std::string_view path_, DirectorySeparator directory_se
path.erase(std::unique(start, path.end(), path.erase(std::unique(start, path.end(),
[type2](char c1, char c2) { return c1 == type2 && c2 == type2; }), [type2](char c1, char c2) { return c1 == type2 && c2 == type2; }),
path.end()); path.end());
const bool absolute = !path.empty() && path[0] == type2;
std::vector<std::string_view> parts;
for (const auto part : SplitPathComponents(path))
{
if (part.empty() || part == ".")
continue;
if (part == ".." && !parts.empty() && parts.back() != "..")
parts.pop_back();
else if (part != "..") parts.push_back(part);
}
std::string resolved = absolute ? std::string(1, type2) : std::string{};
for (std::size_t i = 0; i < parts.size(); ++i)
{
if (i != 0)
resolved += type2;
resolved.append(parts[i].data(), parts[i].size());
}
path = std::move(resolved);
return std::string(RemoveTrailingSlash(path)); return std::string(RemoveTrailingSlash(path));
} }
+3 -1
View File
@@ -157,6 +157,8 @@ bool ArmNce::HandleGuestAlignmentFault(GuestContext* guest_ctx, void* raw_info,
return HandleFailedGuestFault(guest_ctx, raw_info, raw_context); return HandleFailedGuestFault(guest_ctx, raw_info, raw_context);
} }
constexpr size_t NCE_WRITE_FAULT_CLUSTER_PAGES = 4;
bool ArmNce::HandleGuestAccessFault(GuestContext* guest_ctx, void* raw_info, void* raw_context) { bool ArmNce::HandleGuestAccessFault(GuestContext* guest_ctx, void* raw_info, void* raw_context) {
auto* info = static_cast<siginfo_t*>(raw_info); auto* info = static_cast<siginfo_t*>(raw_info);
@@ -165,7 +167,7 @@ bool ArmNce::HandleGuestAccessFault(GuestContext* guest_ctx, void* raw_info, voi
const Common::ProcessAddress addr = const Common::ProcessAddress addr =
(reinterpret_cast<u64>(info->si_addr) & ~Memory::YUZU_PAGEMASK); (reinterpret_cast<u64>(info->si_addr) & ~Memory::YUZU_PAGEMASK);
auto& memory = guest_ctx->parent->m_running_thread->GetOwnerProcess()->GetMemory(); auto& memory = guest_ctx->parent->m_running_thread->GetOwnerProcess()->GetMemory();
if (memory.InvalidateNCE(addr, Memory::YUZU_PAGESIZE)) { if (memory.InvalidateNCE(addr, Memory::YUZU_PAGESIZE * NCE_WRITE_FAULT_CLUSTER_PAGES)) {
// We handled the access successfully and are returning to guest code. // We handled the access successfully and are returning to guest code.
return true; return true;
} }
+1 -18
View File
@@ -145,14 +145,7 @@ bool Patcher::PatchText(std::span<const u8> program_image, const Kernel::CodeSet
// MRS Xn, CNTFRQ_EL0 // MRS Xn, CNTFRQ_EL0
if (auto mrs = MRS{inst}; mrs.Verify() && mrs.GetSystemReg() == CntfrqEl0) { if (auto mrs = MRS{inst}; mrs.Verify() && mrs.GetSystemReg() == CntfrqEl0) {
bool pre_buffer = false; UNREACHABLE();
auto ret = AddRelocations(pre_buffer);
if (pre_buffer) {
WriteCntfrqHandler(ret, oaknut::XReg{static_cast<int>(mrs.GetRt())}, c_pre);
} else {
WriteCntfrqHandler(ret, oaknut::XReg{static_cast<int>(mrs.GetRt())}, c);
}
continue;
} }
// MSR TPIDR_EL0, Xn // MSR TPIDR_EL0, Xn
@@ -584,16 +577,6 @@ void Patcher::WriteMsrHandler(ModuleDestLabel module_dest, oaknut::XReg src_reg,
this->BranchToModule(module_dest); this->BranchToModule(module_dest);
} }
void Patcher::WriteCntfrqHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg, oaknut::VectorCodeGenerator& cg) {
cg.MOV(dest_reg, Common::WallClock::CNTFRQ);
// Jump back to the instruction after the emulated MRS.
if (&cg == &c_pre)
this->BranchToModulePre(module_dest);
else
this->BranchToModule(module_dest);
}
void Patcher::WriteCntpctHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg, oaknut::VectorCodeGenerator& cg) { void Patcher::WriteCntpctHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg, oaknut::VectorCodeGenerator& cg) {
#if defined(HAS_NCE) #if defined(HAS_NCE)
static Common::WallClock clock(false, 1); static Common::WallClock clock(false, 1);
-2
View File
@@ -80,7 +80,6 @@ private:
void WriteSvcTrampoline(ModuleDestLabel module_dest, u32 svc_id, oaknut::VectorCodeGenerator& code, oaknut::Label& save_ctx, oaknut::Label& load_ctx); void WriteSvcTrampoline(ModuleDestLabel module_dest, u32 svc_id, oaknut::VectorCodeGenerator& code, oaknut::Label& save_ctx, oaknut::Label& load_ctx);
void WriteMrsHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg, oaknut::SystemReg src_reg, oaknut::VectorCodeGenerator& code); void WriteMrsHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg, oaknut::SystemReg src_reg, oaknut::VectorCodeGenerator& code);
void WriteMsrHandler(ModuleDestLabel module_dest, oaknut::XReg src_reg, oaknut::VectorCodeGenerator& code); void WriteMsrHandler(ModuleDestLabel module_dest, oaknut::XReg src_reg, oaknut::VectorCodeGenerator& code);
void WriteCntfrqHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg, oaknut::VectorCodeGenerator& code);
void WriteCntpctHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg, oaknut::VectorCodeGenerator& code); void WriteCntpctHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg, oaknut::VectorCodeGenerator& code);
// Convenience wrappers using default code generator // Convenience wrappers using default code generator
@@ -91,7 +90,6 @@ private:
void WriteSvcTrampoline(ModuleDestLabel module_dest, u32 svc_id) { WriteSvcTrampoline(module_dest, svc_id, c, m_save_context, m_load_context); } void WriteSvcTrampoline(ModuleDestLabel module_dest, u32 svc_id) { WriteSvcTrampoline(module_dest, svc_id, c, m_save_context, m_load_context); }
void WriteMrsHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg, oaknut::SystemReg src_reg) { WriteMrsHandler(module_dest, dest_reg, src_reg, c); } void WriteMrsHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg, oaknut::SystemReg src_reg) { WriteMrsHandler(module_dest, dest_reg, src_reg, c); }
void WriteMsrHandler(ModuleDestLabel module_dest, oaknut::XReg src_reg) { WriteMsrHandler(module_dest, src_reg, c); } void WriteMsrHandler(ModuleDestLabel module_dest, oaknut::XReg src_reg) { WriteMsrHandler(module_dest, src_reg, c); }
void WriteCntfrqHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg) { WriteCntfrqHandler(module_dest, dest_reg, c); }
void WriteCntpctHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg) { WriteCntpctHandler(module_dest, dest_reg, c); } void WriteCntpctHandler(ModuleDestLabel module_dest, oaknut::XReg dest_reg) { WriteCntpctHandler(module_dest, dest_reg, c); }
private: private:
+5
View File
@@ -126,6 +126,10 @@ public:
// New batch API to update multiple ranges with a single lock acquisition. // New batch API to update multiple ranges with a single lock acquisition.
void UpdatePagesCachedBatch(std::span<const std::pair<DAddr, size_t>> ranges, s32 delta); void UpdatePagesCachedBatch(std::span<const std::pair<DAddr, size_t>> ranges, s32 delta);
void UpdateTexturePagesCount(DAddr addr, size_t size, s32 delta);
[[nodiscard]] bool IsRegionTextureCached(DAddr addr, size_t size) const noexcept;
private: private:
struct TranslationEntry { struct TranslationEntry {
DAddr guest_page{}; DAddr guest_page{};
@@ -234,6 +238,7 @@ private:
(1ULL << (device_virtual_bits - page_bits)) / subentries; (1ULL << (device_virtual_bits - page_bits)) / subentries;
using CachedPages = std::array<CounterEntry, num_counter_entries>; using CachedPages = std::array<CounterEntry, num_counter_entries>;
std::unique_ptr<CachedPages> cached_pages; std::unique_ptr<CachedPages> cached_pages;
std::unique_ptr<CachedPages> texture_cached_pages;
Common::RangeMutex counter_guard; Common::RangeMutex counter_guard;
std::mutex mapping_guard; std::mutex mapping_guard;
+23
View File
@@ -177,6 +177,7 @@ DeviceMemoryManager<Traits>::DeviceMemoryManager(const DeviceMemory& device_memo
{ {
impl = std::make_unique<DeviceMemoryManagerAllocator<Traits>>(); impl = std::make_unique<DeviceMemoryManagerAllocator<Traits>>();
cached_pages = std::make_unique<CachedPages>(); cached_pages = std::make_unique<CachedPages>();
texture_cached_pages = std::make_unique<CachedPages>();
const size_t total_virtual = device_as_size >> Memory::YUZU_PAGEBITS; const size_t total_virtual = device_as_size >> Memory::YUZU_PAGEBITS;
for (size_t i = 0; i < total_virtual; i++) { for (size_t i = 0; i < total_virtual; i++) {
@@ -625,6 +626,28 @@ void DeviceMemoryManager<Traits>::UpdatePagesCachedCount(DAddr addr, size_t size
UpdatePagesCachedCountNoLock(addr, size, delta); UpdatePagesCachedCountNoLock(addr, size, delta);
} }
template <typename Traits>
void DeviceMemoryManager<Traits>::UpdateTexturePagesCount(DAddr addr, size_t size, s32 delta) {
Common::ScopedRangeLock lk(counter_guard, addr, size);
const size_t page_end = Common::DivCeil(addr + size, Memory::YUZU_PAGESIZE);
for (size_t page = addr >> Memory::YUZU_PAGEBITS; page != page_end; ++page) {
CounterAtomicType& count = texture_cached_pages->at(page >> subentries_shift).Count(page);
count.fetch_add(static_cast<CounterType>(delta), std::memory_order_release);
}
}
template <typename Traits>
bool DeviceMemoryManager<Traits>::IsRegionTextureCached(DAddr addr, size_t size) const noexcept {
const size_t page_end = Common::DivCeil(addr + size, Memory::YUZU_PAGESIZE);
for (size_t page = addr >> Memory::YUZU_PAGEBITS; page != page_end; ++page) {
if (texture_cached_pages->at(page >> subentries_shift).Count(page).load(
std::memory_order_acquire) != 0) {
return true;
}
}
return false;
}
template <typename Traits> template <typename Traits>
void DeviceMemoryManager<Traits>::UpdatePagesCachedBatch(std::span<const std::pair<DAddr, size_t>> ranges, s32 delta) { void DeviceMemoryManager<Traits>::UpdatePagesCachedBatch(std::span<const std::pair<DAddr, size_t>> ranges, s32 delta) {
if (ranges.empty()) { if (ranges.empty()) {
-16
View File
@@ -984,22 +984,6 @@ bool RegisteredCache::RemoveExistingEntry(u64 title_id) const {
return removed_data; return removed_data;
} }
bool RegisteredCache::Delete(const NcaID& id) const {
const auto path = GetRelativePathFromNcaID(id, false, true, false);
const bool is_file = dir->GetFileRelative(path) != nullptr;
const bool is_dir = dir->GetDirectoryRelative(path) != nullptr;
if (is_file) {
return dir->DeleteFile(path);
}
if (is_dir) {
return dir->DeleteSubdirectoryRecursive(path);
}
return true;
}
InstallResult RegisteredCache::RawInstallNCA(const NCA& nca, const VfsCopyFunction& copy, InstallResult RegisteredCache::RawInstallNCA(const NCA& nca, const VfsCopyFunction& copy,
bool overwrite_if_exists, bool overwrite_if_exists,
std::optional<NcaID> override_id) { std::optional<NcaID> override_id) {
-1
View File
@@ -188,7 +188,6 @@ public:
// Removes an existing entry based on title id // Removes an existing entry based on title id
bool RemoveExistingEntry(u64 title_id) const; bool RemoveExistingEntry(u64 title_id) const;
bool Delete(const NcaID& id) const;
private: private:
template <typename T> template <typename T>
+5 -17
View File
@@ -35,16 +35,6 @@ namespace {
constexpr size_t MaxOpenFiles = 8192; constexpr size_t MaxOpenFiles = 8192;
bool IsWithinRoot(std::string_view root, std::string_view full_path) {
if (root.empty())
return true;
if (full_path.size() < root.size() || full_path.substr(0, root.size()) != root)
return false;
return full_path.size() == root.size() || full_path[root.size()] == '/' || full_path[root.size()] == '\\';
}
constexpr FS::FileAccessMode ModeFlagsToFileAccessMode(OpenMode mode) { constexpr FS::FileAccessMode ModeFlagsToFileAccessMode(OpenMode mode) {
switch (mode) { switch (mode) {
case OpenMode::Read: case OpenMode::Read:
@@ -413,8 +403,7 @@ RealVfsDirectory::~RealVfsDirectory() = default;
VirtualFile RealVfsDirectory::GetFileRelative(std::string_view relative_path) const { VirtualFile RealVfsDirectory::GetFileRelative(std::string_view relative_path) const {
const auto full_path = FS::SanitizePath(path + '/' + std::string(relative_path)); const auto full_path = FS::SanitizePath(path + '/' + std::string(relative_path));
if (!FS::Exists(full_path) || FS::IsDir(full_path) if (!FS::Exists(full_path) || FS::IsDir(full_path)) {
|| !IsWithinRoot(FS::SanitizePath(path), full_path)) {
return nullptr; return nullptr;
} }
return base.OpenFile(full_path, perms); return base.OpenFile(full_path, perms);
@@ -422,8 +411,7 @@ VirtualFile RealVfsDirectory::GetFileRelative(std::string_view relative_path) co
VirtualDir RealVfsDirectory::GetDirectoryRelative(std::string_view relative_path) const { VirtualDir RealVfsDirectory::GetDirectoryRelative(std::string_view relative_path) const {
const auto full_path = FS::SanitizePath(path + '/' + std::string(relative_path)); const auto full_path = FS::SanitizePath(path + '/' + std::string(relative_path));
if (!FS::Exists(full_path) || !FS::IsDir(full_path) if (!FS::Exists(full_path) || !FS::IsDir(full_path)) {
|| !IsWithinRoot(FS::SanitizePath(path), full_path)) {
return nullptr; return nullptr;
} }
return base.OpenDirectory(full_path, perms); return base.OpenDirectory(full_path, perms);
@@ -439,7 +427,7 @@ VirtualDir RealVfsDirectory::GetSubdirectory(std::string_view name) const {
VirtualFile RealVfsDirectory::CreateFileRelative(std::string_view relative_path) { VirtualFile RealVfsDirectory::CreateFileRelative(std::string_view relative_path) {
const auto full_path = FS::SanitizePath(path + '/' + std::string(relative_path)); const auto full_path = FS::SanitizePath(path + '/' + std::string(relative_path));
if (!FS::CreateParentDirs(full_path) || !IsWithinRoot(FS::SanitizePath(path), full_path)) { if (!FS::CreateParentDirs(full_path)) {
return nullptr; return nullptr;
} }
return base.CreateFile(full_path, perms); return base.CreateFile(full_path, perms);
@@ -452,7 +440,7 @@ VirtualDir RealVfsDirectory::CreateDirectoryRelative(std::string_view relative_p
bool RealVfsDirectory::DeleteSubdirectoryRecursive(std::string_view name) { bool RealVfsDirectory::DeleteSubdirectoryRecursive(std::string_view name) {
const auto full_path = FS::SanitizePath(this->path + '/' + std::string(name)); const auto full_path = FS::SanitizePath(this->path + '/' + std::string(name));
return FS::RemoveDirRecursively(full_path); return base.DeleteDirectory(full_path);
} }
std::vector<VirtualFile> RealVfsDirectory::GetFiles() const { std::vector<VirtualFile> RealVfsDirectory::GetFiles() const {
@@ -518,7 +506,7 @@ VirtualFile RealVfsDirectory::CreateFile(std::string_view name) {
bool RealVfsDirectory::DeleteSubdirectory(std::string_view name) { bool RealVfsDirectory::DeleteSubdirectory(std::string_view name) {
const std::string subdir_path = (path + '/').append(name); const std::string subdir_path = (path + '/').append(name);
return FS::RemoveDir(subdir_path); return base.DeleteDirectory(subdir_path);
} }
bool RealVfsDirectory::DeleteFile(std::string_view name) { bool RealVfsDirectory::DeleteFile(std::string_view name) {
@@ -57,7 +57,7 @@ ISelfController::ISelfController(Core::System& system_, std::shared_ptr<Applet>
{64, nullptr, "SetInputDetectionSourceSet"}, {64, nullptr, "SetInputDetectionSourceSet"},
{65, D<&ISelfController::ReportUserIsActive>, "ReportUserIsActive"}, {65, D<&ISelfController::ReportUserIsActive>, "ReportUserIsActive"},
{66, nullptr, "GetCurrentIlluminance"}, {66, nullptr, "GetCurrentIlluminance"},
{67, D<&ISelfController::IsIlluminanceAvailable>, "IsIlluminanceAvailable"}, {67, nullptr, "IsIlluminanceAvailable"},
{68, D<&ISelfController::SetAutoSleepDisabled>, "SetAutoSleepDisabled"}, {68, D<&ISelfController::SetAutoSleepDisabled>, "SetAutoSleepDisabled"},
{69, D<&ISelfController::IsAutoSleepDisabled>, "IsAutoSleepDisabled"}, {69, D<&ISelfController::IsAutoSleepDisabled>, "IsAutoSleepDisabled"},
{70, nullptr, "ReportMultimediaError"}, {70, nullptr, "ReportMultimediaError"},
@@ -347,12 +347,6 @@ Result ISelfController::IsAutoSleepDisabled(Out<bool> out_is_auto_sleep_disabled
R_SUCCEED(); R_SUCCEED();
} }
Result ISelfController::IsIlluminanceAvailable(Out<bool> out_is_illuminance_available) {
LOG_WARNING(Service_AM, "(stubbed)");
*out_is_illuminance_available = false;
R_SUCCEED();
}
Result ISelfController::SetInputDetectionPolicy(InputDetectionPolicy input_detection_policy) { Result ISelfController::SetInputDetectionPolicy(InputDetectionPolicy input_detection_policy) {
LOG_WARNING(Service_AM, "(STUBBED) called"); LOG_WARNING(Service_AM, "(STUBBED) called");
R_SUCCEED(); R_SUCCEED();
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project // SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later // SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2024 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2024 yuzu Emulator Project
@@ -60,7 +60,6 @@ private:
Result ReportUserIsActive(); Result ReportUserIsActive();
Result SetAutoSleepDisabled(bool is_auto_sleep_disabled); Result SetAutoSleepDisabled(bool is_auto_sleep_disabled);
Result IsAutoSleepDisabled(Out<bool> out_is_auto_sleep_disabled); Result IsAutoSleepDisabled(Out<bool> out_is_auto_sleep_disabled);
Result IsIlluminanceAvailable(Out<bool> out_is_illuminance_available);
Result SetInputDetectionPolicy(InputDetectionPolicy input_detection_policy); Result SetInputDetectionPolicy(InputDetectionPolicy input_detection_policy);
Result GetAccumulatedSuspendedTickValue(Out<u64> out_accumulated_suspended_tick_value); Result GetAccumulatedSuspendedTickValue(Out<u64> out_accumulated_suspended_tick_value);
Result GetAccumulatedSuspendedTickChangedEvent(OutCopyHandle<Kernel::KReadableEvent> out_event); Result GetAccumulatedSuspendedTickChangedEvent(OutCopyHandle<Kernel::KReadableEvent> out_event);
+10 -42
View File
@@ -43,16 +43,6 @@ static FileSys::VirtualDir GetDirectoryRelativeWrapped(FileSys::VirtualDir base,
return base->GetDirectoryRelative(dir_name); return base->GetDirectoryRelative(dir_name);
} }
static std::string_view GetGuestParentPath(std::string_view path) {
const auto name_index = path.find_last_of("\\/");
return name_index == std::string_view::npos ? std::string_view{} : path.substr(0, name_index);
}
static std::string_view GetGuestFilename(std::string_view path) {
const auto name_index = path.find_last_of("\\/");
return name_index == std::string_view::npos ? path : path.substr(name_index + 1);
}
VfsDirectoryServiceWrapper::VfsDirectoryServiceWrapper(FileSys::VirtualDir backing_) VfsDirectoryServiceWrapper::VfsDirectoryServiceWrapper(FileSys::VirtualDir backing_)
: backing(std::move(backing_)) {} : backing(std::move(backing_)) {}
@@ -93,12 +83,11 @@ Result VfsDirectoryServiceWrapper::DeleteFile(const std::string& path_) const {
return ResultSuccess; return ResultSuccess;
} }
const auto filename = GetGuestFilename(path); auto dir = GetDirectoryRelativeWrapped(backing, Common::FS::GetParentPath(path));
auto dir = GetDirectoryRelativeWrapped(backing, GetGuestParentPath(path)); if (dir == nullptr || dir->GetFile(Common::FS::GetFilename(path)) == nullptr) {
if (filename.empty() || dir == nullptr || dir->GetFile(filename) == nullptr) {
return FileSys::ResultPathNotFound; return FileSys::ResultPathNotFound;
} }
if (!dir->DeleteFile(filename)) { if (!dir->DeleteFile(Common::FS::GetFilename(path))) {
// TODO(DarkLordZach): Find a better error code for this // TODO(DarkLordZach): Find a better error code for this
return ResultUnknown; return ResultUnknown;
} }
@@ -108,9 +97,7 @@ Result VfsDirectoryServiceWrapper::DeleteFile(const std::string& path_) const {
Result VfsDirectoryServiceWrapper::CreateDirectory(const std::string& path_) const { Result VfsDirectoryServiceWrapper::CreateDirectory(const std::string& path_) const {
std::string path(Common::FS::SanitizePath(path_)); std::string path(Common::FS::SanitizePath(path_));
if (GetDirectoryRelativeWrapped(backing, path) != nullptr) {
return FileSys::ResultPathAlreadyExists;
}
// NOTE: This is inaccurate behavior. CreateDirectory is not recursive. // NOTE: This is inaccurate behavior. CreateDirectory is not recursive.
// CreateDirectory should return PathNotFound if the parent directory does not exist. // CreateDirectory should return PathNotFound if the parent directory does not exist.
// This is here temporarily in order to have UMM "work" in the meantime. // This is here temporarily in order to have UMM "work" in the meantime.
@@ -130,19 +117,8 @@ Result VfsDirectoryServiceWrapper::CreateDirectory(const std::string& path_) con
Result VfsDirectoryServiceWrapper::DeleteDirectory(const std::string& path_) const { Result VfsDirectoryServiceWrapper::DeleteDirectory(const std::string& path_) const {
std::string path(Common::FS::SanitizePath(path_)); std::string path(Common::FS::SanitizePath(path_));
const auto dirname = GetGuestFilename(path); auto dir = GetDirectoryRelativeWrapped(backing, Common::FS::GetParentPath(path));
auto dir = GetDirectoryRelativeWrapped(backing, GetGuestParentPath(path)); if (!dir->DeleteSubdirectory(Common::FS::GetFilename(path))) {
FileSys::VirtualDir target{};
if (!dirname.empty() && dir != nullptr) {
target = dir->GetSubdirectory(dirname);
}
if (target == nullptr) {
return FileSys::ResultPathNotFound;
}
if (!target->GetFiles().empty() || !target->GetSubdirectories().empty()) {
return ResultUnknown;
}
if (!dir->DeleteSubdirectory(dirname)) {
// TODO(DarkLordZach): Find a better error code for this // TODO(DarkLordZach): Find a better error code for this
return ResultUnknown; return ResultUnknown;
} }
@@ -151,12 +127,8 @@ Result VfsDirectoryServiceWrapper::DeleteDirectory(const std::string& path_) con
Result VfsDirectoryServiceWrapper::DeleteDirectoryRecursively(const std::string& path_) const { Result VfsDirectoryServiceWrapper::DeleteDirectoryRecursively(const std::string& path_) const {
std::string path(Common::FS::SanitizePath(path_)); std::string path(Common::FS::SanitizePath(path_));
const auto dirname = GetGuestFilename(path); auto dir = GetDirectoryRelativeWrapped(backing, Common::FS::GetParentPath(path));
auto dir = GetDirectoryRelativeWrapped(backing, GetGuestParentPath(path)); if (!dir->DeleteSubdirectoryRecursive(Common::FS::GetFilename(path))) {
if (dirname.empty() || dir == nullptr || dir->GetSubdirectory(dirname) == nullptr) {
return FileSys::ResultPathNotFound;
}
if (!dir->DeleteSubdirectoryRecursive(dirname)) {
// TODO(DarkLordZach): Find a better error code for this // TODO(DarkLordZach): Find a better error code for this
return ResultUnknown; return ResultUnknown;
} }
@@ -165,13 +137,9 @@ Result VfsDirectoryServiceWrapper::DeleteDirectoryRecursively(const std::string&
Result VfsDirectoryServiceWrapper::CleanDirectoryRecursively(const std::string& path) const { Result VfsDirectoryServiceWrapper::CleanDirectoryRecursively(const std::string& path) const {
const std::string sanitized_path(Common::FS::SanitizePath(path)); const std::string sanitized_path(Common::FS::SanitizePath(path));
const auto dirname = GetGuestFilename(sanitized_path); auto dir = GetDirectoryRelativeWrapped(backing, Common::FS::GetParentPath(sanitized_path));
auto dir = GetDirectoryRelativeWrapped(backing, GetGuestParentPath(sanitized_path));
if (dirname.empty() || dir == nullptr || dir->GetSubdirectory(dirname) == nullptr) { if (!dir->CleanSubdirectoryRecursive(Common::FS::GetFilename(sanitized_path))) {
return FileSys::ResultPathNotFound;
}
if (!dir->CleanSubdirectoryRecursive(dirname)) {
// TODO(DarkLordZach): Find a better error code for this // TODO(DarkLordZach): Find a better error code for this
return ResultUnknown; return ResultUnknown;
} }
+1 -7
View File
@@ -237,7 +237,7 @@ IHidServer::IHidServer(Core::System& system_, std::shared_ptr<ResourceManager> r
{3013, nullptr, "SetDebugPadGenericPadMap"}, //21.0.0+ {3013, nullptr, "SetDebugPadGenericPadMap"}, //21.0.0+
{3014, nullptr, "GetDebugPadKeyboardMap"}, //21.0.0+ {3014, nullptr, "GetDebugPadKeyboardMap"}, //21.0.0+
{3015, nullptr, "SetDebugPadKeyboardMap"}, //21.0.0+ {3015, nullptr, "SetDebugPadKeyboardMap"}, //21.0.0+
{3150, C<&IHidServer::SetMouseLibraryVersion>, "SetMouseLibraryVersion"}, //21.0.0+ {3150, nullptr, "SetMouseLibraryVersion"}, //21.0.0+
// What? -- {12010, nullptr, "SetButtonConfigLeft"}, // What? -- {12010, nullptr, "SetButtonConfigLeft"},
}; };
// clang-format on // clang-format on
@@ -1471,12 +1471,6 @@ Result IHidServer::SetTouchScreenResolution(u32 width, u32 height,
R_SUCCEED(); R_SUCCEED();
} }
Result IHidServer::SetMouseLibraryVersion(ClientAppletResourceUserId aruid) {
LOG_INFO(Service_HID, "(STUBBED) called, applet_resource_user_id={}", aruid.pid);
R_SUCCEED();
}
std::shared_ptr<ResourceManager> IHidServer::GetResourceManager() { std::shared_ptr<ResourceManager> IHidServer::GetResourceManager() {
resource_manager->Initialize(); resource_manager->Initialize();
return resource_manager; return resource_manager;
+1 -2
View File
@@ -1,4 +1,4 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project // SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later // SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
@@ -263,7 +263,6 @@ private:
Result IsFirmwareUpdateNeededForNotification(Out<bool> out_is_firmware_update_needed, Result IsFirmwareUpdateNeededForNotification(Out<bool> out_is_firmware_update_needed,
s32 unknown, ClientAppletResourceUserId aruid); s32 unknown, ClientAppletResourceUserId aruid);
Result SetTouchScreenResolution(u32 width, u32 height, ClientAppletResourceUserId aruid); Result SetTouchScreenResolution(u32 width, u32 height, ClientAppletResourceUserId aruid);
Result SetMouseLibraryVersion(ClientAppletResourceUserId aruid);
std::shared_ptr<ResourceManager> resource_manager; std::shared_ptr<ResourceManager> resource_manager;
std::shared_ptr<HidFirmwareSettings> firmware_settings; std::shared_ptr<HidFirmwareSettings> firmware_settings;
+2 -297
View File
@@ -1,19 +1,9 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
#include <algorithm>
#include <array>
#include <memory> #include <memory>
#include <vector>
#include "core/core.h"
#include "core/file_sys/registered_cache.h"
#include "core/file_sys/romfs_factory.h" #include "core/file_sys/romfs_factory.h"
#include "core/hle/api_version.h"
#include "core/hle/service/filesystem/filesystem.h"
#include "core/hle/service/ipc_helpers.h" #include "core/hle/service/ipc_helpers.h"
#include "core/hle/service/ncm/ncm.h" #include "core/hle/service/ncm/ncm.h"
#include "core/hle/service/server_manager.h" #include "core/hle/service/server_manager.h"
@@ -98,268 +88,6 @@ public:
} }
}; };
class IContentStorage final : public ServiceFramework<IContentStorage> {
public:
explicit IContentStorage(Core::System& system_, FileSys::StorageId id)
: ServiceFramework{system_, "IContentStorage"}, storage{id} {
// clang-format off
static const FunctionInfo functions[] = {
{0, &IContentStorage::GeneratePlaceHolderId, "GeneratePlaceHolderId"},
{1, &IContentStorage::CreatePlaceHolder, "CreatePlaceHolder"},
{2, &IContentStorage::DeletePlaceHolder, "DeletePlaceHolder"},
{4, &IContentStorage::WritePlaceHolder, "WritePlaceHolder"},
{5, &IContentStorage::Register, "Register"},
{6, &IContentStorage::Delete, "Delete"},
};
// clang-format on
RegisterHandlers(functions);
}
private:
void GeneratePlaceHolderId(HLERequestContext& ctx) {
LOG_DEBUG(Service_NCM, "called");
IPC::ResponseBuilder rb{ctx, 6};
rb.Push(ResultSuccess);
rb.PushRaw(FileSys::PlaceholderCache::Generate());
}
void CreatePlaceHolder(HLERequestContext& ctx) {
IPC::RequestParser rp{ctx};
[[maybe_unused]] FileSys::NcaID content_id{};
FileSys::NcaID placeholder_id{};
if constexpr (HLE::ApiVersion::HOS_VERSION_MAJOR >= 16) {
placeholder_id = rp.PopRaw<FileSys::NcaID>();
content_id = rp.PopRaw<FileSys::NcaID>();
} else {
content_id = rp.PopRaw<FileSys::NcaID>();
placeholder_id = rp.PopRaw<FileSys::NcaID>();
}
const auto size = rp.Pop<s64>();
auto* const placeholder_cache =
system.GetFileSystemController().GetPlaceholderCacheForStorage(storage);
const bool succeeded =
placeholder_cache != nullptr && size >= 0 &&
(placeholder_cache->Exists(placeholder_id) ||
placeholder_cache->Create(placeholder_id, static_cast<u64>(size)));
if (succeeded) {
LOG_DEBUG(Service_NCM, "called, storage_id={}, size={}", static_cast<u32>(storage),
size);
} else {
LOG_WARNING(Service_NCM, "failed, storage_id={}, size={}", static_cast<u32>(storage),
size);
}
IPC::ResponseBuilder rb{ctx, 2};
rb.Push(succeeded ? ResultSuccess : ResultUnknown);
}
void DeletePlaceHolder(HLERequestContext& ctx) {
IPC::RequestParser rp{ctx};
const auto placeholder_id = rp.PopRaw<FileSys::NcaID>();
auto* const placeholder_cache =
system.GetFileSystemController().GetPlaceholderCacheForStorage(storage);
const bool succeeded =
placeholder_cache != nullptr &&
(!placeholder_cache->Exists(placeholder_id) || placeholder_cache->Delete(placeholder_id));
if (succeeded) {
LOG_DEBUG(Service_NCM, "called, storage_id={}", static_cast<u32>(storage));
} else {
LOG_WARNING(Service_NCM, "failed, storage_id={}", static_cast<u32>(storage));
}
IPC::ResponseBuilder rb{ctx, 2};
rb.Push(succeeded ? ResultSuccess : ResultUnknown);
}
void WritePlaceHolder(HLERequestContext& ctx) {
IPC::RequestParser rp{ctx};
const auto placeholder_id = rp.PopRaw<FileSys::NcaID>();
const auto offset = rp.Pop<u64>();
const auto data = ctx.ReadBuffer();
auto* const placeholder_cache =
system.GetFileSystemController().GetPlaceholderCacheForStorage(storage);
const std::vector<u8> write_data{data.begin(), data.end()};
const bool succeeded =
placeholder_cache != nullptr &&
placeholder_cache->Write(placeholder_id, offset, write_data);
if (succeeded) {
LOG_DEBUG(Service_NCM, "called, storage_id={}, offset={}, size={}",
static_cast<u32>(storage), offset, data.size());
} else {
LOG_WARNING(Service_NCM, "failed, storage_id={}, offset={}, size={}",
static_cast<u32>(storage), offset, data.size());
}
IPC::ResponseBuilder rb{ctx, 2};
rb.Push(succeeded ? ResultSuccess : ResultUnknown);
}
void Register(HLERequestContext& ctx) {
IPC::RequestParser rp{ctx};
FileSys::NcaID content_id{};
FileSys::NcaID placeholder_id{};
if constexpr (HLE::ApiVersion::HOS_VERSION_MAJOR >= 16) {
placeholder_id = rp.PopRaw<FileSys::NcaID>();
content_id = rp.PopRaw<FileSys::NcaID>();
} else {
content_id = rp.PopRaw<FileSys::NcaID>();
placeholder_id = rp.PopRaw<FileSys::NcaID>();
}
auto& fsc = system.GetFileSystemController();
auto* const placeholder_cache = fsc.GetPlaceholderCacheForStorage(storage);
auto* const registered_cache = fsc.GetRegisteredCacheForStorage(storage);
const bool succeeded =
placeholder_cache != nullptr && registered_cache != nullptr &&
placeholder_cache->Register(registered_cache, placeholder_id, content_id);
if (succeeded) {
LOG_DEBUG(Service_NCM, "called, storage_id={}", static_cast<u32>(storage));
} else {
LOG_WARNING(Service_NCM, "failed, storage_id={}", static_cast<u32>(storage));
}
IPC::ResponseBuilder rb{ctx, 2};
rb.Push(succeeded ? ResultSuccess : ResultUnknown);
}
void Delete(HLERequestContext& ctx) {
IPC::RequestParser rp{ctx};
const auto content_id = rp.PopRaw<FileSys::NcaID>();
auto* const registered_cache =
system.GetFileSystemController().GetRegisteredCacheForStorage(storage);
const bool succeeded = registered_cache != nullptr && registered_cache->Delete(content_id);
if (succeeded) {
registered_cache->Refresh();
}
if (succeeded) {
LOG_DEBUG(Service_NCM, "called, storage_id={}", static_cast<u32>(storage));
} else {
LOG_WARNING(Service_NCM, "failed, storage_id={}", static_cast<u32>(storage));
}
IPC::ResponseBuilder rb{ctx, 2};
rb.Push(succeeded ? ResultSuccess : ResultUnknown);
}
FileSys::StorageId storage;
};
class IContentMetaDatabase final : public ServiceFramework<IContentMetaDatabase> {
public:
explicit IContentMetaDatabase(Core::System& system_, FileSys::StorageId id)
: ServiceFramework{system_, "IContentMetaDatabase"}, storage{id} {
// clang-format off
static const FunctionInfo functions[] = {
{0, &IContentMetaDatabase::Set, "Set"},
{2, &IContentMetaDatabase::Remove, "Remove"},
{8, &IContentMetaDatabase::Has, "Has"},
{15, &IContentMetaDatabase::Commit, "Commit"},
};
// clang-format on
RegisterHandlers(functions);
}
private:
struct ContentMetaKey {
u64 id;
u32 version;
FileSys::TitleType type;
u8 install_type;
std::array<u8, 2> padding;
};
static_assert(sizeof(ContentMetaKey) == 0x10);
void Set(HLERequestContext& ctx) {
IPC::RequestParser rp{ctx};
const auto key = rp.PopRaw<ContentMetaKey>();
const auto entry_matches = [&key](const ContentMetaKey& entry) {
return entry.id == key.id && entry.version == key.version && entry.type == key.type &&
entry.install_type == key.install_type;
};
if (std::find_if(entries.begin(), entries.end(), entry_matches) == entries.end()) {
entries.push_back(key);
}
LOG_DEBUG(Service_NCM,
"called, storage_id={}, title_id={:016X}, version={}, type={}, size={}",
static_cast<u32>(storage), key.id, key.version, static_cast<u8>(key.type),
ctx.GetReadBufferSize());
IPC::ResponseBuilder rb{ctx, 2};
rb.Push(ResultSuccess);
}
void Remove(HLERequestContext& ctx) {
IPC::RequestParser rp{ctx};
const auto key = rp.PopRaw<ContentMetaKey>();
std::erase_if(entries, [&key](const ContentMetaKey& entry) {
return entry.id == key.id && entry.version == key.version && entry.type == key.type &&
entry.install_type == key.install_type;
});
LOG_DEBUG(Service_NCM, "called, storage_id={}, title_id={:016X}, version={}, type={}",
static_cast<u32>(storage), key.id, key.version, static_cast<u8>(key.type));
IPC::ResponseBuilder rb{ctx, 2};
rb.Push(ResultSuccess);
}
void Has(HLERequestContext& ctx) {
IPC::RequestParser rp{ctx};
const auto key = rp.PopRaw<ContentMetaKey>();
const bool has_pending =
std::find_if(entries.begin(), entries.end(), [&key](const ContentMetaKey& entry) {
return entry.id == key.id && entry.version == key.version &&
entry.type == key.type && entry.install_type == key.install_type;
}) != entries.end();
auto* const registered_cache =
system.GetFileSystemController().GetRegisteredCacheForStorage(storage);
const bool has_registered =
registered_cache != nullptr &&
registered_cache->HasEntry(key.id, FileSys::ContentRecordType::Meta);
LOG_DEBUG(Service_NCM, "called, storage_id={}, title_id={:016X}, version={}, type={}, has={}",
static_cast<u32>(storage), key.id, key.version, static_cast<u8>(key.type),
has_pending || has_registered);
IPC::ResponseBuilder rb{ctx, 3};
rb.Push(ResultSuccess);
rb.Push(has_pending || has_registered);
}
void Commit(HLERequestContext& ctx) {
auto* const registered_cache =
system.GetFileSystemController().GetRegisteredCacheForStorage(storage);
if (registered_cache != nullptr) {
registered_cache->Refresh();
}
LOG_DEBUG(Service_NCM, "called, storage_id={}", static_cast<u32>(storage));
IPC::ResponseBuilder rb{ctx, 2};
rb.Push(ResultSuccess);
}
FileSys::StorageId storage;
std::vector<ContentMetaKey> entries;
};
class LR final : public ServiceFramework<LR> { class LR final : public ServiceFramework<LR> {
public: public:
explicit LR(Core::System& system_) : ServiceFramework{system_, "lr"} { explicit LR(Core::System& system_) : ServiceFramework{system_, "lr"} {
@@ -385,8 +113,8 @@ public:
{1, nullptr, "CreateContentMetaDatabase"}, {1, nullptr, "CreateContentMetaDatabase"},
{2, nullptr, "VerifyContentStorage"}, {2, nullptr, "VerifyContentStorage"},
{3, nullptr, "VerifyContentMetaDatabase"}, {3, nullptr, "VerifyContentMetaDatabase"},
{4, &NCM::OpenContentStorage, "OpenContentStorage"}, {4, nullptr, "OpenContentStorage"},
{5, &NCM::OpenContentMetaDatabase, "OpenContentMetaDatabase"}, {5, nullptr, "OpenContentMetaDatabase"},
{6, nullptr, "CloseContentStorageForcibly"}, {6, nullptr, "CloseContentStorageForcibly"},
{7, nullptr, "CloseContentMetaDatabaseForcibly"}, {7, nullptr, "CloseContentMetaDatabaseForcibly"},
{8, nullptr, "CleanupContentMetaDatabase"}, {8, nullptr, "CleanupContentMetaDatabase"},
@@ -402,29 +130,6 @@ public:
RegisterHandlers(functions); RegisterHandlers(functions);
} }
private:
void OpenContentStorage(HLERequestContext& ctx) {
IPC::RequestParser rp{ctx};
const auto storage_id = rp.PopEnum<FileSys::StorageId>();
LOG_DEBUG(Service_NCM, "called, storage_id={}", static_cast<u32>(storage_id));
IPC::ResponseBuilder rb{ctx, 2, 0, 1};
rb.Push(ResultSuccess);
rb.PushIpcInterface<IContentStorage>(ctx, system, storage_id);
}
void OpenContentMetaDatabase(HLERequestContext& ctx) {
IPC::RequestParser rp{ctx};
const auto storage_id = rp.PopEnum<FileSys::StorageId>();
LOG_DEBUG(Service_NCM, "called, storage_id={}", static_cast<u32>(storage_id));
IPC::ResponseBuilder rb{ctx, 2, 0, 1};
rb.Push(ResultSuccess);
rb.PushIpcInterface<IContentMetaDatabase>(ctx, system, storage_id);
}
}; };
void LoopProcess(Core::System& system) { void LoopProcess(Core::System& system) {
@@ -9,7 +9,6 @@
#include "core/file_sys/registered_cache.h" #include "core/file_sys/registered_cache.h"
#include "core/hle/service/cmif_serialization.h" #include "core/hle/service/cmif_serialization.h"
#include "core/hle/service/filesystem/filesystem.h" #include "core/hle/service/filesystem/filesystem.h"
#include "core/hle/service/ipc_helpers.h"
#include "core/hle/service/ns/application_manager_interface.h" #include "core/hle/service/ns/application_manager_interface.h"
#include "core/file_sys/content_archive.h" #include "core/file_sys/content_archive.h"
@@ -20,7 +19,6 @@
#include "core/launch_timestamp_cache.h" #include "core/launch_timestamp_cache.h"
#include <algorithm> #include <algorithm>
#include <cstring>
#include <vector> #include <vector>
namespace Service::NS { namespace Service::NS {
@@ -38,14 +36,14 @@ IApplicationManagerInterface::IApplicationManagerInterface(Core::System& system_
{1, nullptr, "GenerateApplicationRecordCount"}, {1, nullptr, "GenerateApplicationRecordCount"},
{2, D<&IApplicationManagerInterface::GetApplicationRecordUpdateSystemEvent>, "GetApplicationRecordUpdateSystemEvent"}, {2, D<&IApplicationManagerInterface::GetApplicationRecordUpdateSystemEvent>, "GetApplicationRecordUpdateSystemEvent"},
{3, nullptr, "GetApplicationViewDeprecated"}, {3, nullptr, "GetApplicationViewDeprecated"},
{4, D<&IApplicationManagerInterface::DeleteApplicationEntity>, "DeleteApplicationEntity"}, {4, nullptr, "DeleteApplicationEntity"},
{5, D<&IApplicationManagerInterface::DeleteApplicationCompletely>, "DeleteApplicationCompletely"}, {5, nullptr, "DeleteApplicationCompletely"},
{6, nullptr, "IsAnyApplicationEntityRedundant"}, {6, nullptr, "IsAnyApplicationEntityRedundant"},
{7, nullptr, "DeleteRedundantApplicationEntity"}, {7, nullptr, "DeleteRedundantApplicationEntity"},
{8, nullptr, "IsApplicationEntityMovable"}, {8, nullptr, "IsApplicationEntityMovable"},
{9, nullptr, "MoveApplicationEntity"}, {9, nullptr, "MoveApplicationEntity"},
{11, nullptr, "CalculateApplicationOccupiedSize"}, {11, nullptr, "CalculateApplicationOccupiedSize"},
{16, &IApplicationManagerInterface::PushApplicationRecord, "PushApplicationRecord"}, {16, nullptr, "PushApplicationRecord"},
{17, nullptr, "ListApplicationRecordContentMeta"}, {17, nullptr, "ListApplicationRecordContentMeta"},
{19, nullptr, "LaunchApplicationOld"}, {19, nullptr, "LaunchApplicationOld"},
{21, nullptr, "GetApplicationContentPath"}, {21, nullptr, "GetApplicationContentPath"},
@@ -645,27 +643,6 @@ Result IApplicationManagerInterface::IsAnyApplicationEntityInstalled(
R_SUCCEED(); R_SUCCEED();
} }
Result IApplicationManagerInterface::DeleteApplicationEntity(u64 application_id) {
LOG_DEBUG(Service_NS, "called, application_id={:016X}", application_id);
auto& fsc = system.GetFileSystemController();
if (auto* const user_cache = fsc.GetUserNANDContents(); user_cache != nullptr) {
user_cache->RemoveExistingEntry(application_id);
user_cache->Refresh();
}
if (auto* const sdmc_cache = fsc.GetSDMCContents(); sdmc_cache != nullptr) {
sdmc_cache->RemoveExistingEntry(application_id);
sdmc_cache->Refresh();
}
record_update_system_event.Signal(system.Kernel());
R_SUCCEED();
}
Result IApplicationManagerInterface::DeleteApplicationCompletely(u64 application_id) {
R_RETURN(DeleteApplicationEntity(application_id));
}
Result IApplicationManagerInterface::GetApplicationViewDeprecated( Result IApplicationManagerInterface::GetApplicationViewDeprecated(
OutArray<ApplicationViewV19, BufferAttr_HipcMapAlias> out_application_views, OutArray<ApplicationViewV19, BufferAttr_HipcMapAlias> out_application_views,
InArray<u64, BufferAttr_HipcMapAlias> application_ids) { InArray<u64, BufferAttr_HipcMapAlias> application_ids) {
@@ -866,29 +843,6 @@ Result IApplicationManagerInterface::Unknown4053() {
R_SUCCEED(); R_SUCCEED();
} }
void IApplicationManagerInterface::PushApplicationRecord(HLERequestContext& ctx) {
const auto record = ctx.ReadBuffer();
u64 application_id{};
if (record.size() >= sizeof(application_id)) {
std::memcpy(&application_id, record.data(), sizeof(application_id));
}
LOG_DEBUG(Service_NS, "called, application_id={:016X}, size={}", application_id, record.size());
auto& fsc = system.GetFileSystemController();
if (auto* const user_cache = fsc.GetUserNANDContents(); user_cache != nullptr) {
user_cache->Refresh();
}
if (auto* const sdmc_cache = fsc.GetSDMCContents(); sdmc_cache != nullptr) {
sdmc_cache->Refresh();
}
record_update_system_event.Signal(system.Kernel());
IPC::ResponseBuilder rb{ctx, 2};
rb.Push(ResultSuccess);
}
void IApplicationManagerInterface::ListApplicationTitle(HLERequestContext& ctx) { void IApplicationManagerInterface::ListApplicationTitle(HLERequestContext& ctx) {
LOG_DEBUG(Service_NS, "called"); LOG_DEBUG(Service_NS, "called");
IReadOnlyApplicationControlDataInterface(system).ListApplicationTitle(ctx); IReadOnlyApplicationControlDataInterface(system).ListApplicationTitle(ctx);
@@ -56,8 +56,6 @@ public:
Result ResumeAll(); Result ResumeAll();
Result IsQualificationTransitionSupportedByProcessId(Out<bool> out_is_supported, Result IsQualificationTransitionSupportedByProcessId(Out<bool> out_is_supported,
u64 process_id); u64 process_id);
Result DeleteApplicationEntity(u64 application_id);
Result DeleteApplicationCompletely(u64 application_id);
Result GetStorageSize(Out<s64> out_total_space_size, Out<s64> out_free_space_size, Result GetStorageSize(Out<s64> out_total_space_size, Out<s64> out_free_space_size,
FileSys::StorageId storage_id); FileSys::StorageId storage_id);
Result TouchApplication(u64 application_id); Result TouchApplication(u64 application_id);
@@ -76,7 +74,6 @@ public:
Result RequestDownloadApplicationControlDataInBackground(u64 control_source, Result RequestDownloadApplicationControlDataInBackground(u64 control_source,
u64 application_id); u64 application_id);
void PushApplicationRecord(HLERequestContext& ctx);
void ListApplicationTitle(HLERequestContext& ctx); void ListApplicationTitle(HLERequestContext& ctx);
private: private:
+2 -5
View File
@@ -1,6 +1,3 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
@@ -31,9 +28,9 @@ SPL_MIG::SPL_MIG(Core::System& system_, std::shared_ptr<Module> module_)
static const FunctionInfo functions[] = { static const FunctionInfo functions[] = {
{0, &SPL::GetConfig, "GetConfig"}, {0, &SPL::GetConfig, "GetConfig"},
{1, &SPL::ModularExponentiate, "ModularExponentiate"}, {1, &SPL::ModularExponentiate, "ModularExponentiate"},
{2, &SPL::GenerateAesKek, "GenerateAesKek"}, {2, nullptr, "GenerateAesKek"},
{3, nullptr, "LoadAesKey"}, {3, nullptr, "LoadAesKey"},
{4, &SPL::GenerateAesKey, "GenerateAesKey"}, {4, nullptr, "GenerateAesKey"},
{5, &SPL::SetConfig, "SetConfig"}, {5, &SPL::SetConfig, "SetConfig"},
{7, &SPL::GenerateRandomBytes, "GenerateRandomBytes"}, {7, &SPL::GenerateRandomBytes, "GenerateRandomBytes"},
{11, &SPL::IsDevelopment, "IsDevelopment"}, {11, &SPL::IsDevelopment, "IsDevelopment"},
-30
View File
@@ -59,36 +59,6 @@ void Module::Interface::ModularExponentiate(HLERequestContext& ctx) {
rb.Push(ResultSecureMonitorNotImplemented); rb.Push(ResultSecureMonitorNotImplemented);
} }
void Module::Interface::GenerateAesKek(HLERequestContext& ctx) {
IPC::RequestParser rp{ctx};
[[maybe_unused]] const auto key_source = rp.PopRaw<KeySource>();
const auto generation = rp.Pop<u32>();
const auto option = rp.Pop<u32>();
LOG_WARNING(Service_SPL, "(STUBBED) called, generation={:#x}, option={:#x}", generation,
option);
AccessKey access_key{};
IPC::ResponseBuilder rb{ctx, 6};
rb.Push(ResultSuccess);
rb.PushRaw(access_key);
}
void Module::Interface::GenerateAesKey(HLERequestContext& ctx) {
IPC::RequestParser rp{ctx};
[[maybe_unused]] const auto access_key = rp.PopRaw<AccessKey>();
[[maybe_unused]] const auto key_source = rp.PopRaw<KeySource>();
LOG_WARNING(Service_SPL, "(STUBBED) called");
AesKey aes_key{};
IPC::ResponseBuilder rb{ctx, 6};
rb.Push(ResultSuccess);
rb.PushRaw(aes_key);
}
void Module::Interface::SetConfig(HLERequestContext& ctx) { void Module::Interface::SetConfig(HLERequestContext& ctx) {
UNIMPLEMENTED_MSG("SetConfig is not implemented!"); UNIMPLEMENTED_MSG("SetConfig is not implemented!");
-5
View File
@@ -1,6 +1,3 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
@@ -28,8 +25,6 @@ public:
// General // General
void GetConfig(HLERequestContext& ctx); void GetConfig(HLERequestContext& ctx);
void ModularExponentiate(HLERequestContext& ctx); void ModularExponentiate(HLERequestContext& ctx);
void GenerateAesKek(HLERequestContext& ctx);
void GenerateAesKey(HLERequestContext& ctx);
void SetConfig(HLERequestContext& ctx); void SetConfig(HLERequestContext& ctx);
void GenerateRandomBytes(HLERequestContext& ctx); void GenerateRandomBytes(HLERequestContext& ctx);
void IsDevelopment(HLERequestContext& ctx); void IsDevelopment(HLERequestContext& ctx);
+112 -26
View File
@@ -121,6 +121,15 @@ void BufferCache<P>::WriteMemory(DAddr device_addr, u64 size) {
memory_tracker.MarkRegionAsCpuModified(device_addr, size); memory_tracker.MarkRegionAsCpuModified(device_addr, size);
} }
template <class P>
void BufferCache<P>::CpuWriteInvalidate(DAddr device_addr, u64 size) {
if (!memory_tracker.CpuMarkIfNotGpuModified(device_addr, size)) {
return;
}
std::scoped_lock lock{mutex};
WriteMemory(device_addr, size);
}
template <class P> template <class P>
void BufferCache<P>::CachedWriteMemory(DAddr device_addr, u64 size) { void BufferCache<P>::CachedWriteMemory(DAddr device_addr, u64 size) {
const bool is_dirty = IsRegionRegistered(device_addr, size); const bool is_dirty = IsRegionRegistered(device_addr, size);
@@ -175,9 +184,71 @@ std::optional<VideoCore::RasterizerDownloadArea> BufferCache<P>::GetFlushArea(DA
template <class P> template <class P>
void BufferCache<P>::DownloadMemory(DAddr device_addr, u64 size) { void BufferCache<P>::DownloadMemory(DAddr device_addr, u64 size) {
ForEachBufferInRange(device_addr, size, [&](BufferId, Buffer& buffer) { if constexpr (!USE_MEMORY_MAPS) {
DownloadBufferMemory(buffer, device_addr, size); std::scoped_lock lock{mutex};
ForEachBufferInRange(device_addr, size, [&](BufferId, Buffer& buffer) {
DownloadBufferMemory(buffer, device_addr, size);
});
return;
}
boost::container::small_vector<std::pair<BufferCopy, BufferId>, 8> downloads;
u64 total_size_bytes = 0;
u64 largest_copy = 0;
std::unique_lock lock{mutex};
ForEachBufferInRange(device_addr, size, [&](BufferId buffer_id, Buffer& buffer) {
memory_tracker.ForEachDownloadRangeAndClear(
device_addr, size, [&](u64 device_addr_out, u64 range_size) {
const DAddr buffer_addr = buffer.CpuAddr();
const auto add_download = [&](DAddr start, DAddr end) {
const u64 new_offset = start - buffer_addr;
const u64 new_size = end - start;
downloads.push_back({
BufferCopy{
.src_offset = new_offset,
.dst_offset = total_size_bytes,
.size = new_size,
},
buffer_id,
});
constexpr u64 align = 64ULL;
constexpr u64 mask = ~(align - 1ULL);
total_size_bytes += (new_size + align - 1) & mask;
largest_copy = (std::max)(largest_copy, new_size);
};
gpu_modified_ranges.ForEachInRange(device_addr_out, range_size, add_download);
ClearDownload(device_addr_out, range_size);
gpu_modified_ranges.Subtract(device_addr_out, range_size);
});
}); });
if (total_size_bytes == 0) {
return;
}
auto download_staging = runtime.DownloadStagingBuffer(total_size_bytes);
boost::container::small_vector<BufferCopy, 8> writebacks;
runtime.PreCopyBarrier();
for (auto& [copy, buffer_id] : downloads) {
copy.dst_offset += download_staging.offset;
Buffer& buffer = slot_buffers[buffer_id];
buffer.MarkUsage(copy.src_offset, copy.size);
const std::array copies{copy};
runtime.CopyBuffer(download_staging.buffer, buffer, copies, false);
BufferCopy writeback{copy};
writeback.src_offset = static_cast<u64>(buffer.CpuAddr()) + copy.src_offset;
writebacks.push_back(writeback);
}
runtime.PostCopyBarrier();
lock.unlock();
runtime.Finish();
const u8* const base = download_staging.mapped_span.data();
for (const BufferCopy& writeback : writebacks) {
const u64 staging_offset = writeback.dst_offset - download_staging.offset;
device_memory.WriteBlockUnsafe(static_cast<DAddr>(writeback.src_offset),
base + staging_offset, writeback.size);
}
} }
template <class P> template <class P>
@@ -214,7 +285,7 @@ bool BufferCache<P>::DMACopy(GPUVAddr src_address, GPUVAddr dest_address, u64 am
auto& src_buffer = slot_buffers[buffer_a]; auto& src_buffer = slot_buffers[buffer_a];
auto& dest_buffer = slot_buffers[buffer_b]; auto& dest_buffer = slot_buffers[buffer_b];
SynchronizeBuffer(src_buffer, *cpu_src_address, static_cast<u32>(amount)); SynchronizeBuffer(src_buffer, *cpu_src_address, static_cast<u32>(amount));
SynchronizeBuffer(dest_buffer, *cpu_dest_address, static_cast<u32>(amount)); memory_tracker.UnmarkRegionAsCpuModified(*cpu_dest_address, static_cast<u32>(amount));
std::array copies{BufferCopy{ std::array copies{BufferCopy{
.src_offset = src_buffer.Offset(*cpu_src_address), .src_offset = src_buffer.Offset(*cpu_src_address),
.dst_offset = dest_buffer.Offset(*cpu_dest_address), .dst_offset = dest_buffer.Offset(*cpu_dest_address),
@@ -673,32 +744,44 @@ void BufferCache<P>::PopAsyncFlushes() {
template <class P> template <class P>
void BufferCache<P>::PopAsyncBuffers() { void BufferCache<P>::PopAsyncBuffers() {
if (async_buffers.empty()) { struct Writeback {
return; DAddr addr;
} const u8* src;
if (!async_buffers.front().has_value()) { u64 size;
};
boost::container::small_vector<Writeback, 8> writebacks;
{
std::scoped_lock lock{mutex};
if (async_buffers.empty()) {
return;
}
if (!async_buffers.front().has_value()) {
async_buffers.pop_front();
return;
}
auto& downloads = pending_downloads.front();
auto& async_buffer = async_buffers.front();
const u8* base = async_buffer->mapped_span.data();
const size_t base_offset = async_buffer->offset;
for (const auto& copy : downloads) {
const DAddr device_addr = static_cast<DAddr>(copy.src_offset);
const u64 dst_offset = copy.dst_offset - base_offset;
const u8* read_mapped_memory = base + dst_offset;
async_downloads.ForEachInRange(device_addr, copy.size, [&](DAddr start, DAddr end, s32) {
writebacks.push_back(
{start, &read_mapped_memory[start - device_addr], end - start});
});
async_downloads.Subtract(device_addr, copy.size, [&](DAddr start, DAddr end) {
gpu_modified_ranges.Subtract(start, end - start);
});
}
async_buffers_death_ring.emplace_back(*async_buffer);
async_buffers.pop_front(); async_buffers.pop_front();
return; pending_downloads.pop_front();
} }
auto& downloads = pending_downloads.front(); for (const auto& wb : writebacks) {
auto& async_buffer = async_buffers.front(); device_memory.WriteBlockUnsafe(wb.addr, wb.src, wb.size);
u8* base = async_buffer->mapped_span.data();
const size_t base_offset = async_buffer->offset;
for (const auto& copy : downloads) {
const DAddr device_addr = static_cast<DAddr>(copy.src_offset);
const u64 dst_offset = copy.dst_offset - base_offset;
const u8* read_mapped_memory = base + dst_offset;
async_downloads.ForEachInRange(device_addr, copy.size, [&](DAddr start, DAddr end, s32) {
device_memory.WriteBlockUnsafe(start, &read_mapped_memory[start - device_addr],
end - start);
});
async_downloads.Subtract(device_addr, copy.size, [&](DAddr start, DAddr end) {
gpu_modified_ranges.Subtract(start, end - start);
});
} }
async_buffers_death_ring.emplace_back(*async_buffer);
async_buffers.pop_front();
pending_downloads.pop_front();
} }
template <class P> template <class P>
@@ -1638,6 +1721,9 @@ void BufferCache<P>::TouchBuffer(Buffer& buffer, BufferId buffer_id) noexcept {
template <class P> template <class P>
bool BufferCache<P>::SynchronizeBuffer(Buffer& buffer, DAddr device_addr, u32 size) { bool BufferCache<P>::SynchronizeBuffer(Buffer& buffer, DAddr device_addr, u32 size) {
if (!memory_tracker.HasCpuModifiedCheap(device_addr, size)) {
return true;
}
upload_copies.clear(); upload_copies.clear();
u64 total_size_bytes = 0; u64 total_size_bytes = 0;
u64 largest_copy = 0; u64 largest_copy = 0;
@@ -217,6 +217,8 @@ public:
void WriteMemory(DAddr device_addr, u64 size); void WriteMemory(DAddr device_addr, u64 size);
void CpuWriteInvalidate(DAddr device_addr, u64 size);
void CachedWriteMemory(DAddr device_addr, u64 size); void CachedWriteMemory(DAddr device_addr, u64 size);
bool OnCPUWrite(DAddr device_addr, u64 size); bool OnCPUWrite(DAddr device_addr, u64 size);
@@ -7,9 +7,11 @@
#pragma once #pragma once
#include <algorithm> #include <algorithm>
#include <atomic>
#include <bit> #include <bit>
#include <deque> #include <deque>
#include <limits> #include <limits>
#include <mutex>
#include <type_traits> #include <type_traits>
#include <ankerl/unordered_dense.h> #include <ankerl/unordered_dense.h>
#include <utility> #include <utility>
@@ -49,6 +51,24 @@ public:
}); });
} }
[[nodiscard]] bool HasCpuModifiedCheap(VAddr query_cpu_addr, u64 query_size) noexcept {
std::size_t remaining_size{query_size};
std::size_t page_index{query_cpu_addr >> HIGHER_PAGE_BITS};
u64 page_offset{query_cpu_addr & HIGHER_PAGE_MASK};
while (remaining_size > 0) {
const std::size_t copy_amount{
std::min<std::size_t>(HIGHER_PAGE_SIZE - page_offset, remaining_size)};
const Manager* manager = top_tier[page_index].load(std::memory_order_acquire);
if (manager == nullptr || manager->CpuModifiedPageCount() != 0) {
return true;
}
page_index++;
page_offset = 0;
remaining_size -= copy_amount;
}
return false;
}
/// Returns true if a region has been modified from the CPU /// Returns true if a region has been modified from the CPU
[[nodiscard]] bool IsRegionCpuModified(VAddr query_cpu_addr, u64 query_size) noexcept { [[nodiscard]] bool IsRegionCpuModified(VAddr query_cpu_addr, u64 query_size) noexcept {
return IteratePages<true>(query_cpu_addr, query_size, [](Manager* manager, u64 offset, size_t size) { return IteratePages<true>(query_cpu_addr, query_size, [](Manager* manager, u64 offset, size_t size) {
@@ -130,12 +150,28 @@ public:
} }
void FlushCachedWrites() noexcept { void FlushCachedWrites() noexcept {
std::scoped_lock lk{tracker_mutex};
for (auto id : cached_pages) { for (auto id : cached_pages) {
top_tier[id]->FlushCachedWrites(); top_tier[id].load(std::memory_order_relaxed)->FlushCachedWrites();
} }
cached_pages.clear(); cached_pages.clear();
} }
[[nodiscard]] bool CpuMarkIfNotGpuModified(VAddr addr, u64 size) {
std::scoped_lock lk{tracker_mutex};
const bool gpu = IteratePagesNoLock<false>(
addr, size, [](Manager* manager, u64 offset, size_t sz) {
return manager->IsRegionModified(Type::GPU, offset, sz);
});
if (gpu) {
return true;
}
IteratePagesNoLock<true>(addr, size, [](Manager* manager, u64 offset, size_t sz) {
manager->ChangeRegionState(Type::CPU, true, manager->cpu_addr + offset, sz);
});
return false;
}
/// Call 'func' for each CPU modified range and unmark those pages as CPU modified /// Call 'func' for each CPU modified range and unmark those pages as CPU modified
template <typename Func> template <typename Func>
void ForEachUploadRange(VAddr query_cpu_range, u64 query_size, Func&& func) { void ForEachUploadRange(VAddr query_cpu_range, u64 query_size, Func&& func) {
@@ -162,6 +198,12 @@ public:
private: private:
template <bool create_region_on_fail, typename Func> template <bool create_region_on_fail, typename Func>
bool IteratePages(VAddr cpu_address, size_t size, Func&& func) { bool IteratePages(VAddr cpu_address, size_t size, Func&& func) {
std::scoped_lock lk{tracker_mutex};
return IteratePagesNoLock<create_region_on_fail>(cpu_address, size, std::forward<Func>(func));
}
template <bool create_region_on_fail, typename Func>
bool IteratePagesNoLock(VAddr cpu_address, size_t size, Func&& func) {
using FuncReturn = typename std::invoke_result<Func, Manager*, u64, size_t>::type; using FuncReturn = typename std::invoke_result<Func, Manager*, u64, size_t>::type;
static constexpr bool BOOL_BREAK = std::is_same_v<FuncReturn, bool>; static constexpr bool BOOL_BREAK = std::is_same_v<FuncReturn, bool>;
std::size_t remaining_size{size}; std::size_t remaining_size{size};
@@ -170,7 +212,7 @@ private:
while (remaining_size > 0) { while (remaining_size > 0) {
const std::size_t copy_amount{ const std::size_t copy_amount{
std::min<std::size_t>(HIGHER_PAGE_SIZE - page_offset, remaining_size)}; std::min<std::size_t>(HIGHER_PAGE_SIZE - page_offset, remaining_size)};
auto* manager{top_tier[page_index]}; auto* manager{top_tier[page_index].load(std::memory_order_relaxed)};
if (manager) { if (manager) {
if constexpr (BOOL_BREAK) { if constexpr (BOOL_BREAK) {
if (func(manager, page_offset, copy_amount)) { if (func(manager, page_offset, copy_amount)) {
@@ -181,7 +223,7 @@ private:
} }
} else if constexpr (create_region_on_fail) { } else if constexpr (create_region_on_fail) {
CreateRegion(page_index); CreateRegion(page_index);
manager = top_tier[page_index]; manager = top_tier[page_index].load(std::memory_order_relaxed);
if constexpr (BOOL_BREAK) { if constexpr (BOOL_BREAK) {
if (func(manager, page_offset, copy_amount)) { if (func(manager, page_offset, copy_amount)) {
return true; return true;
@@ -199,6 +241,7 @@ private:
template <bool create_region_on_fail, typename Func> template <bool create_region_on_fail, typename Func>
std::pair<u64, u64> IteratePairs(VAddr cpu_address, size_t size, Func&& func) { std::pair<u64, u64> IteratePairs(VAddr cpu_address, size_t size, Func&& func) {
std::scoped_lock lk{tracker_mutex};
std::size_t remaining_size{size}; std::size_t remaining_size{size};
std::size_t page_index{cpu_address >> HIGHER_PAGE_BITS}; std::size_t page_index{cpu_address >> HIGHER_PAGE_BITS};
u64 page_offset{cpu_address & HIGHER_PAGE_MASK}; u64 page_offset{cpu_address & HIGHER_PAGE_MASK};
@@ -207,7 +250,7 @@ private:
while (remaining_size > 0) { while (remaining_size > 0) {
const std::size_t copy_amount{ const std::size_t copy_amount{
std::min<std::size_t>(HIGHER_PAGE_SIZE - page_offset, remaining_size)}; std::min<std::size_t>(HIGHER_PAGE_SIZE - page_offset, remaining_size)};
auto* manager{top_tier[page_index]}; auto* manager{top_tier[page_index].load(std::memory_order_relaxed)};
const auto execute = [&] { const auto execute = [&] {
auto [new_begin, new_end] = func(manager, page_offset, copy_amount); auto [new_begin, new_end] = func(manager, page_offset, copy_amount);
if (new_begin != 0 || new_end != 0) { if (new_begin != 0 || new_end != 0) {
@@ -220,7 +263,7 @@ private:
execute(); execute();
} else if constexpr (create_region_on_fail) { } else if constexpr (create_region_on_fail) {
CreateRegion(page_index); CreateRegion(page_index);
manager = top_tier[page_index]; manager = top_tier[page_index].load(std::memory_order_relaxed);
execute(); execute();
} }
page_index++; page_index++;
@@ -236,7 +279,7 @@ private:
void CreateRegion(std::size_t page_index) { void CreateRegion(std::size_t page_index) {
const VAddr base_cpu_addr = page_index << HIGHER_PAGE_BITS; const VAddr base_cpu_addr = page_index << HIGHER_PAGE_BITS;
top_tier[page_index] = GetNewManager(base_cpu_addr); top_tier[page_index].store(GetNewManager(base_cpu_addr), std::memory_order_release);
} }
Manager* GetNewManager(VAddr base_cpu_address) { Manager* GetNewManager(VAddr base_cpu_address) {
@@ -254,11 +297,12 @@ private:
return new_manager; return new_manager;
} }
std::array<Manager*, NUM_HIGH_PAGES> top_tier{}; std::array<std::atomic<Manager*>, NUM_HIGH_PAGES> top_tier{};
std::deque<std::array<Manager, MANAGER_POOL_SIZE>> manager_pool; std::deque<std::array<Manager, MANAGER_POOL_SIZE>> manager_pool;
std::deque<Manager*> free_managers; std::deque<Manager*> free_managers;
ankerl::unordered_dense::set<u32> cached_pages; ankerl::unordered_dense::set<u32> cached_pages;
DeviceTracker* device_tracker = nullptr; DeviceTracker* device_tracker = nullptr;
std::mutex tracker_mutex;
}; };
} // namespace VideoCommon } // namespace VideoCommon
@@ -7,6 +7,7 @@
#pragma once #pragma once
#include <algorithm> #include <algorithm>
#include <atomic>
#include <bit> #include <bit>
#include <limits> #include <limits>
#include <span> #include <span>
@@ -48,6 +49,11 @@ struct WordManager {
u64 const last_word = (~u64{0} << shift) >> shift; u64 const last_word = (~u64{0} << shift) >> shift;
heap[num_words * size_t(Type::CPU) + num_words - 1] = last_word; heap[num_words * size_t(Type::CPU) + num_words - 1] = last_word;
heap[num_words * size_t(Type::Untracked) + num_words - 1] = last_word; heap[num_words * size_t(Type::Untracked) + num_words - 1] = last_word;
u32 cpu_pages = 0;
for (size_t i = 0; i < num_words; ++i) {
cpu_pages += static_cast<u32>(std::popcount(heap[num_words * size_t(Type::CPU) + i]));
}
cpu_modified_pages.store(cpu_pages, std::memory_order_relaxed);
} }
explicit WordManager() = default; explicit WordManager() = default;
@@ -120,10 +126,15 @@ struct WordManager {
[[maybe_unused]] std::span<u64> untracked_words = Span(Type::Untracked); [[maybe_unused]] std::span<u64> untracked_words = Span(Type::Untracked);
[[maybe_unused]] std::span<u64> cached_words = Span(Type::CachedCPU); [[maybe_unused]] std::span<u64> cached_words = Span(Type::CachedCPU);
std::vector<std::pair<VAddr, u64>> ranges; std::vector<std::pair<VAddr, u64>> ranges;
s64 cpu_delta = 0;
IterateWords(dirty_addr - cpu_addr, size, [&](size_t index, u64 mask) { IterateWords(dirty_addr - cpu_addr, size, [&](size_t index, u64 mask) {
if (type == Type::CPU || type == Type::CachedCPU) { if (type == Type::CPU || type == Type::CachedCPU) {
CollectChangedRanges(!enable, index, untracked_words[index], mask, ranges); CollectChangedRanges(!enable, index, untracked_words[index], mask, ranges);
} }
if (type == Type::CPU) {
const u64 old = state_words[index];
cpu_delta += enable ? std::popcount(~old & mask) : -std::popcount(old & mask);
}
if (enable) { if (enable) {
state_words[index] |= mask; state_words[index] |= mask;
if (type == Type::CPU || type == Type::CachedCPU) if (type == Type::CPU || type == Type::CachedCPU)
@@ -138,6 +149,9 @@ struct WordManager {
untracked_words[index] &= ~mask; untracked_words[index] &= ~mask;
} }
}); });
if (cpu_delta != 0) {
cpu_modified_pages.fetch_add(static_cast<u32>(cpu_delta), std::memory_order_release);
}
if (!ranges.empty()) { if (!ranges.empty()) {
ApplyCollectedRanges(ranges, (!enable) ? 1 : -1); ApplyCollectedRanges(ranges, (!enable) ? 1 : -1);
} }
@@ -165,6 +179,7 @@ struct WordManager {
(pending_pointer - pending_offset) * BYTES_PER_PAGE); (pending_pointer - pending_offset) * BYTES_PER_PAGE);
}; };
std::vector<std::pair<VAddr, u64>> ranges; std::vector<std::pair<VAddr, u64>> ranges;
s64 cpu_delta = 0;
IterateWords(offset, size, [&](size_t index, u64 mask) { IterateWords(offset, size, [&](size_t index, u64 mask) {
if (type == Type::GPU) if (type == Type::GPU)
mask &= ~untracked_words[index]; mask &= ~untracked_words[index];
@@ -173,6 +188,8 @@ struct WordManager {
if (type == Type::CPU || type == Type::CachedCPU) { if (type == Type::CPU || type == Type::CachedCPU) {
CollectChangedRanges(true, index, untracked_words[index], mask, ranges); CollectChangedRanges(true, index, untracked_words[index], mask, ranges);
} }
if (type == Type::CPU)
cpu_delta -= std::popcount(word);
state_words[index] &= ~mask; state_words[index] &= ~mask;
if (type == Type::CPU || type == Type::CachedCPU) if (type == Type::CPU || type == Type::CachedCPU)
untracked_words[index] &= ~mask; untracked_words[index] &= ~mask;
@@ -194,6 +211,9 @@ struct WordManager {
} }
}); });
}); });
if (cpu_delta != 0) {
cpu_modified_pages.fetch_add(static_cast<u32>(cpu_delta), std::memory_order_release);
}
if (pending) { if (pending) {
release(); release();
} }
@@ -248,13 +268,18 @@ struct WordManager {
auto const untracked_words = Span(Type::Untracked); auto const untracked_words = Span(Type::Untracked);
auto const cpu_words = Span(Type::CPU); auto const cpu_words = Span(Type::CPU);
std::vector<std::pair<VAddr, u64>> ranges; std::vector<std::pair<VAddr, u64>> ranges;
s64 cpu_delta = 0;
for (u64 word_index = 0; word_index < num_words; ++word_index) { for (u64 word_index = 0; word_index < num_words; ++word_index) {
const u64 cached_bits = cached_words[word_index]; const u64 cached_bits = cached_words[word_index];
CollectChangedRanges(false, word_index, untracked_words[word_index], cached_bits, ranges); CollectChangedRanges(false, word_index, untracked_words[word_index], cached_bits, ranges);
cpu_delta += std::popcount(~cpu_words[word_index] & cached_bits);
untracked_words[word_index] |= cached_bits; untracked_words[word_index] |= cached_bits;
cpu_words[word_index] |= cached_bits; cpu_words[word_index] |= cached_bits;
cached_words[word_index] = 0; cached_words[word_index] = 0;
} }
if (cpu_delta != 0) {
cpu_modified_pages.fetch_add(static_cast<u32>(cpu_delta), std::memory_order_release);
}
if (!ranges.empty()) { if (!ranges.empty()) {
ApplyCollectedRanges(ranges, -1); ApplyCollectedRanges(ranges, -1);
} }
@@ -319,9 +344,14 @@ struct WordManager {
return std::span<const u64>(heap.data() + num_words * size_t(type), num_words); return std::span<const u64>(heap.data() + num_words * size_t(type), num_words);
} }
[[nodiscard]] u32 CpuModifiedPageCount() const noexcept {
return cpu_modified_pages.load(std::memory_order_acquire);
}
std::array<u64, size_t(Type::Max) * num_words> heap = {}; std::array<u64, size_t(Type::Max) * num_words> heap = {};
DeviceTracker* tracker = nullptr; DeviceTracker* tracker = nullptr;
VAddr cpu_addr = 0; VAddr cpu_addr = 0;
std::atomic<u32> cpu_modified_pages{0};
}; };
} // namespace VideoCommon } // namespace VideoCommon
-21
View File
@@ -16,26 +16,6 @@
#include "video_core/memory_manager.h" #include "video_core/memory_manager.h"
namespace Tegra::Control { namespace Tegra::Control {
namespace {
// Match NVK/Nouveau's initial pushbuffer subchannel layout.
constexpr u32 Nvk3DSubchannel = 0;
constexpr u32 NvkComputeSubchannel = 1;
constexpr u32 Nvk2DSubchannel = 3;
constexpr u32 NvkCopySubchannel = 4;
void BindNvkDefaultSubchannels(ChannelState::Payload& payload) {
auto& dma_pusher = payload.dma_pusher;
dma_pusher.BindSubchannel(&payload.maxwell_3d, Nvk3DSubchannel, Engines::EngineTypes::Maxwell3D);
dma_pusher.BindSubchannel(&payload.kepler_compute, NvkComputeSubchannel,
Engines::EngineTypes::KeplerCompute);
// Subchannel 2 is M2MF there; Eden does not expose a 0x9039 engine yet.
dma_pusher.BindSubchannel(&payload.fermi_2d, Nvk2DSubchannel, Engines::EngineTypes::Fermi2D);
dma_pusher.BindSubchannel(&payload.maxwell_dma, NvkCopySubchannel,
Engines::EngineTypes::MaxwellDMA);
}
} // Anonymous namespace
ChannelState::Payload::Payload(Core::System& system, MemoryManager& memory_manager, ChannelState& channel_state) ChannelState::Payload::Payload(Core::System& system, MemoryManager& memory_manager, ChannelState& channel_state)
: maxwell_3d(memory_manager) : maxwell_3d(memory_manager)
@@ -55,7 +35,6 @@ void ChannelState::Init(Core::System& system, u64 program_id_) {
ASSERT(memory_manager); ASSERT(memory_manager);
program_id = program_id_; program_id = program_id_;
payload.emplace(system, *memory_manager, *this); payload.emplace(system, *memory_manager, *this);
BindNvkDefaultSubchannels(*payload);
initialized = true; initialized = true;
} }
+3 -14
View File
@@ -22,21 +22,10 @@
namespace Tegra::Engines { namespace Tegra::Engines {
namespace {
constexpr u32 Gf100BindClassMask = 0xffff;
constexpr u32 Gf100BindValidMask = 0x1f0000 | Gf100BindClassMask;
} // Anonymous namespace
void Puller::ProcessBindMethod(DmaPusher& dma_pusher, const MethodCall& method_call) { void Puller::ProcessBindMethod(DmaPusher& dma_pusher, const MethodCall& method_call) {
LOG_DEBUG(HW_GPU, "Binding subchannel {} to engine {:#x}", method_call.subchannel, method_call.argument); // Bind the current subchannel to the desired engine id.
u32 engine = method_call.argument; LOG_DEBUG(HW_GPU, "Binding subchannel {} to engine {}", method_call.subchannel, method_call.argument);
if ((engine & ~Gf100BindClassMask) != 0 && (engine & ~Gf100BindValidMask) == 0) { const auto engine_id = static_cast<EngineID>(method_call.argument);
engine &= Gf100BindClassMask;
}
const auto engine_id = static_cast<EngineID>(engine);
bound_engines[method_call.subchannel] = engine_id; bound_engines[method_call.subchannel] = engine_id;
switch (engine_id) { switch (engine_id) {
case EngineID::FERMI_TWOD_A: case EngineID::FERMI_TWOD_A:
+2 -5
View File
@@ -91,9 +91,6 @@ public:
func(); func();
} }
fences.push(std::move(new_fence)); fences.push(std::move(new_fence));
if (should_flush) {
rasterizer.FlushCommands();
}
if constexpr (can_async_check) { if constexpr (can_async_check) {
guard.unlock(); guard.unlock();
cv.notify_all(); cv.notify_all();
@@ -238,10 +235,10 @@ private:
void PopAsyncFlushes() { void PopAsyncFlushes() {
{ {
std::scoped_lock lock{buffer_cache.mutex, texture_cache.mutex}; std::scoped_lock lock{texture_cache.mutex};
texture_cache.PopAsyncFlushes(); texture_cache.PopAsyncFlushes();
buffer_cache.PopAsyncFlushes();
} }
buffer_cache.PopAsyncFlushes();
query_cache.PopAsyncFlushes(); query_cache.PopAsyncFlushes();
} }
@@ -15,6 +15,8 @@ set(GLSL_INCLUDES
set(SHADER_FILES set(SHADER_FILES
${CMAKE_CURRENT_SOURCE_DIR}/astc_decoder.comp ${CMAKE_CURRENT_SOURCE_DIR}/astc_decoder.comp
${CMAKE_CURRENT_SOURCE_DIR}/astc_decoder.frag
${CMAKE_CURRENT_SOURCE_DIR}/astc_decoder.vert
${CMAKE_CURRENT_SOURCE_DIR}/blit_color_float.frag ${CMAKE_CURRENT_SOURCE_DIR}/blit_color_float.frag
${CMAKE_CURRENT_SOURCE_DIR}/block_linear_unswizzle_2d.comp ${CMAKE_CURRENT_SOURCE_DIR}/block_linear_unswizzle_2d.comp
${CMAKE_CURRENT_SOURCE_DIR}/blit_color_msaa.frag ${CMAKE_CURRENT_SOURCE_DIR}/blit_color_msaa.frag
+65 -31
View File
@@ -964,35 +964,70 @@ void ComputeEndpoints(out uvec4 ep1, out uvec4 ep2, uint color_endpoint_mode, ui
} }
uint UnquantizeTexelWeight(EncodingData val) { uint UnquantizeTexelWeight(EncodingData val) {
uint encoding = Encoding(val), bitlen = NumBits(val), bitval = BitValue(val); const uint encoding = Encoding(val);
if (encoding == JUST_BITS) { const uint bitlen = NumBits(val);
return (bitlen >= 1 && bitlen <= 5) const uint bitval = BitValue(val);
? uint(floor(0.5f + float(bitval) * 64.0f / float((1 << bitlen) - 1))) const uint A = ReplicateBitTo7((bitval & 1));
: FastReplicateTo6(bitval, bitlen); uint B = 0, C = 0, D = 0;
} else if (encoding == TRIT || encoding == QUINT) { uint result = 0;
uint B = 0, C = 0, D = 0; const uint bitlen_0_results[5] = {0, 16, 32, 48, 64};
uint b_mask = (0x3100 >> (bitlen * 4)) & 0xf; switch (encoding) {
uint b = (bitval >> 1) & b_mask; case JUST_BITS:
return FastReplicateTo6(bitval, bitlen);
case TRIT: {
D = QuintTritValue(val); D = QuintTritValue(val);
if (encoding == TRIT) { switch (bitlen) {
switch (bitlen) { case 0:
case 0: return D * 32; //0,32,64 return bitlen_0_results[D * 2];
case 1: C = 50; break; case 1: {
case 2: C = 23; B = (b << 6) | (b << 2) | b; break; C = 50;
case 3: C = 11; B = (b << 5) | b; break; break;
}
} else if (encoding == QUINT) {
switch (bitlen) {
case 0: return D * 16; //0, 16, 32, 48, 64
case 1: C = 28; break;
case 2: C = 13; B = (b << 6) | (b << 1); break;
}
} }
uint A = ReplicateBitTo7(bitval & 1); case 2: {
uint res = (A & 0x20) | (((D * C + B) ^ A) >> 2); C = 23;
return res + (res > 32 ? 1 : 0); const uint b = (bitval >> 1) & 1;
B = (b << 6) | (b << 2) | b;
break;
}
case 3: {
C = 11;
const uint cb = (bitval >> 1) & 3;
B = (cb << 5) | cb;
break;
}
default:
break;
}
break;
} }
return 0; case QUINT: {
D = QuintTritValue(val);
switch (bitlen) {
case 0:
return bitlen_0_results[D];
case 1: {
C = 28;
break;
}
case 2: {
C = 13;
const uint b = (bitval >> 1) & 1;
B = (b << 6) | (b << 1);
break;
}
}
break;
}
}
if (encoding != JUST_BITS && bitlen > 0) {
result = D * C + B;
result ^= A;
result = (A & 0x20) | (result >> 2);
}
if (result > 32) {
result += 1;
}
return result;
} }
void UnquantizeTexelWeights(uvec2 size, bool is_dual_plane) { void UnquantizeTexelWeights(uvec2 size, bool is_dual_plane) {
@@ -1394,11 +1429,10 @@ void DecompressBlock(ivec3 coord) {
} }
uint SwizzleOffset(uvec2 pos) { uint SwizzleOffset(uvec2 pos) {
return ((pos.x & 32u) << 3u) | const uint x = pos.x;
((pos.y & 6u) << 5u) | const uint y = pos.y;
((pos.x & 16u) << 1u) | return ((x % 64) / 32) * 256 + ((y % 8) / 2) * 64 +
((pos.y & 1u) << 4u) | ((x % 32) / 16) * 32 + (y % 2) * 16 + (x % 16);
(pos.x & 15u);
} }
void main() { void main() {
File diff suppressed because it is too large Load Diff
@@ -0,0 +1,19 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
#version 450
#ifdef VULKAN
#define VERTEX_ID gl_VertexIndex
#else
#define VERTEX_ID gl_VertexID
out gl_PerVertex {
vec4 gl_Position;
};
#endif
void main() {
float x = float((VERTEX_ID & 1) << 2);
float y = float((VERTEX_ID & 2) << 1);
gl_Position = vec4(x - 1.0, y - 1.0, 0.0, 1.0);
}
+127 -84
View File
@@ -58,8 +58,9 @@ MemoryManager::MemoryManager(Core::System& system_, u64 address_space_bits_, GPU
MemoryManager::~MemoryManager() = default; MemoryManager::~MemoryManager() = default;
MemoryManager::EntryType MemoryManager::GetEntry(size_t position, bool is_big_page) const { template <bool is_big_page>
if (is_big_page) { MemoryManager::EntryType MemoryManager::GetEntry(size_t position) const {
if constexpr (is_big_page) {
position = position >> big_page_bits; position = position >> big_page_bits;
const u64 entry_mask = big_entries[position / 32]; const u64 entry_mask = big_entries[position / 32];
const size_t sub_index = position % 32; const size_t sub_index = position % 32;
@@ -72,8 +73,9 @@ MemoryManager::EntryType MemoryManager::GetEntry(size_t position, bool is_big_pa
} }
} }
void MemoryManager::SetEntry(size_t position, MemoryManager::EntryType entry, bool is_big_page) { template <bool is_big_page>
if (is_big_page) { void MemoryManager::SetEntry(size_t position, MemoryManager::EntryType entry) {
if constexpr (is_big_page) {
position = position >> big_page_bits; position = position >> big_page_bits;
const u64 entry_mask = big_entries[position / 32]; const u64 entry_mask = big_entries[position / 32];
const size_t sub_index = position % 32; const size_t sub_index = position % 32;
@@ -106,21 +108,23 @@ inline void MemoryManager::SetBigPageContinuous(size_t big_page_index, bool valu
(~(1ULL << sub_index) & continuous_mask) | (value ? 1ULL << sub_index : 0); (~(1ULL << sub_index) & continuous_mask) | (value ? 1ULL << sub_index : 0);
} }
GPUVAddr MemoryManager::PageTableOp(GPUVAddr gpu_addr, [[maybe_unused]] DAddr dev_addr, size_t size, PTEKind kind, MemoryManager::EntryType entry_type) { template <MemoryManager::EntryType entry_type>
GPUVAddr MemoryManager::PageTableOp(GPUVAddr gpu_addr, [[maybe_unused]] DAddr dev_addr, size_t size,
PTEKind kind) {
[[maybe_unused]] u64 remaining_size{size}; [[maybe_unused]] u64 remaining_size{size};
if (entry_type == EntryType::Mapped) { if constexpr (entry_type == EntryType::Mapped) {
page_table.ReserveRange(gpu_addr, size); page_table.ReserveRange(gpu_addr, size);
} }
for (u64 offset{}; offset < size; offset += page_size) { for (u64 offset{}; offset < size; offset += page_size) {
const GPUVAddr current_gpu_addr = gpu_addr + offset; const GPUVAddr current_gpu_addr = gpu_addr + offset;
[[maybe_unused]] const auto current_entry_type = GetEntry(current_gpu_addr, false); [[maybe_unused]] const auto current_entry_type = GetEntry<false>(current_gpu_addr);
SetEntry(current_gpu_addr, entry_type, false); SetEntry<false>(current_gpu_addr, entry_type);
if (current_entry_type != entry_type) { if (current_entry_type != entry_type) {
rasterizer->ModifyGPUMemory(unique_identifier, current_gpu_addr, page_size); rasterizer->ModifyGPUMemory(unique_identifier, current_gpu_addr, page_size);
} }
if (entry_type == EntryType::Mapped) { if constexpr (entry_type == EntryType::Mapped) {
const DAddr current_dev_addr = dev_addr + offset; const DAddr current_dev_addr = dev_addr + offset;
const auto index = PageEntryIndex(current_gpu_addr, false); const auto index = PageEntryIndex<false>(current_gpu_addr);
const u32 sub_value = static_cast<u32>(current_dev_addr >> cpu_page_bits); const u32 sub_value = static_cast<u32>(current_dev_addr >> cpu_page_bits);
page_table[index] = sub_value; page_table[index] = sub_value;
} }
@@ -130,18 +134,20 @@ GPUVAddr MemoryManager::PageTableOp(GPUVAddr gpu_addr, [[maybe_unused]] DAddr de
return gpu_addr; return gpu_addr;
} }
GPUVAddr MemoryManager::BigPageTableOp(GPUVAddr gpu_addr, [[maybe_unused]] DAddr dev_addr, size_t size, PTEKind kind, MemoryManager::EntryType entry_type) { template <MemoryManager::EntryType entry_type>
GPUVAddr MemoryManager::BigPageTableOp(GPUVAddr gpu_addr, [[maybe_unused]] DAddr dev_addr,
size_t size, PTEKind kind) {
[[maybe_unused]] u64 remaining_size{size}; [[maybe_unused]] u64 remaining_size{size};
for (u64 offset{}; offset < size; offset += big_page_size) { for (u64 offset{}; offset < size; offset += big_page_size) {
const GPUVAddr current_gpu_addr = gpu_addr + offset; const GPUVAddr current_gpu_addr = gpu_addr + offset;
[[maybe_unused]] const auto current_entry_type = GetEntry(current_gpu_addr, true); [[maybe_unused]] const auto current_entry_type = GetEntry<true>(current_gpu_addr);
SetEntry(current_gpu_addr, entry_type, true); SetEntry<true>(current_gpu_addr, entry_type);
if (current_entry_type != entry_type) { if (current_entry_type != entry_type) {
rasterizer->ModifyGPUMemory(unique_identifier, current_gpu_addr, big_page_size); rasterizer->ModifyGPUMemory(unique_identifier, current_gpu_addr, big_page_size);
} }
if (entry_type == EntryType::Mapped) { if constexpr (entry_type == EntryType::Mapped) {
const DAddr current_dev_addr = dev_addr + offset; const DAddr current_dev_addr = dev_addr + offset;
const auto index = PageEntryIndex(current_gpu_addr, true); const auto index = PageEntryIndex<true>(current_gpu_addr);
const u32 sub_value = static_cast<u32>(current_dev_addr >> cpu_page_bits); const u32 sub_value = static_cast<u32>(current_dev_addr >> cpu_page_bits);
big_page_table_dev[index] = sub_value; big_page_table_dev[index] = sub_value;
const bool is_continuous = ([&] { const bool is_continuous = ([&] {
@@ -175,16 +181,19 @@ void MemoryManager::BindRasterizer(VideoCore::RasterizerInterface* rasterizer_)
rasterizer = rasterizer_; rasterizer = rasterizer_;
} }
GPUVAddr MemoryManager::Map(GPUVAddr gpu_addr, DAddr dev_addr, std::size_t size, PTEKind kind, bool is_big_pages) { GPUVAddr MemoryManager::Map(GPUVAddr gpu_addr, DAddr dev_addr, std::size_t size, PTEKind kind,
if (is_big_pages) bool is_big_pages) {
return BigPageTableOp(gpu_addr, dev_addr, size, kind, EntryType::Mapped); if (is_big_pages) [[likely]] {
return PageTableOp(gpu_addr, dev_addr, size, kind, EntryType::Mapped); return BigPageTableOp<EntryType::Mapped>(gpu_addr, dev_addr, size, kind);
}
return PageTableOp<EntryType::Mapped>(gpu_addr, dev_addr, size, kind);
} }
GPUVAddr MemoryManager::MapSparse(GPUVAddr gpu_addr, std::size_t size, bool is_big_pages) { GPUVAddr MemoryManager::MapSparse(GPUVAddr gpu_addr, std::size_t size, bool is_big_pages) {
if (is_big_pages) if (is_big_pages) [[likely]] {
return BigPageTableOp(gpu_addr, 0, size, PTEKind::INVALID, EntryType::Reserved); return BigPageTableOp<EntryType::Reserved>(gpu_addr, 0, size, PTEKind::INVALID);
return PageTableOp(gpu_addr, 0, size, PTEKind::INVALID, EntryType::Reserved); }
return PageTableOp<EntryType::Reserved>(gpu_addr, 0, size, PTEKind::INVALID);
} }
void MemoryManager::Unmap(GPUVAddr gpu_addr, std::size_t size) { void MemoryManager::Unmap(GPUVAddr gpu_addr, std::size_t size) {
@@ -198,21 +207,26 @@ void MemoryManager::Unmap(GPUVAddr gpu_addr, std::size_t size) {
} }
page_stash.clear(); page_stash.clear();
BigPageTableOp(gpu_addr, 0, size, PTEKind::INVALID, EntryType::Free); BigPageTableOp<EntryType::Free>(gpu_addr, 0, size, PTEKind::INVALID);
PageTableOp(gpu_addr, 0, size, PTEKind::INVALID, EntryType::Free); PageTableOp<EntryType::Free>(gpu_addr, 0, size, PTEKind::INVALID);
} }
std::optional<DAddr> MemoryManager::GpuToCpuAddress(GPUVAddr gpu_addr) const { std::optional<DAddr> MemoryManager::GpuToCpuAddress(GPUVAddr gpu_addr) const {
if (!IsWithinGPUAddressRange(gpu_addr)) [[unlikely]] { if (!IsWithinGPUAddressRange(gpu_addr)) [[unlikely]] {
return std::nullopt; return std::nullopt;
} }
if (GetEntry(gpu_addr, true) != EntryType::Mapped) [[unlikely]] { if (GetEntry<true>(gpu_addr) != EntryType::Mapped) [[unlikely]] {
if (GetEntry(gpu_addr, false) != EntryType::Mapped) if (GetEntry<false>(gpu_addr) != EntryType::Mapped) {
return std::nullopt; return std::nullopt;
const DAddr dev_addr_base = DAddr(page_table[PageEntryIndex(gpu_addr, false)]) << cpu_page_bits; }
const DAddr dev_addr_base = static_cast<DAddr>(page_table[PageEntryIndex<false>(gpu_addr)])
<< cpu_page_bits;
return dev_addr_base + (gpu_addr & page_mask); return dev_addr_base + (gpu_addr & page_mask);
} }
const DAddr dev_addr_base = DAddr(big_page_table_dev[PageEntryIndex(gpu_addr, true)]) << cpu_page_bits;
const DAddr dev_addr_base =
static_cast<DAddr>(big_page_table_dev[PageEntryIndex<true>(gpu_addr)]) << cpu_page_bits;
return dev_addr_base + (gpu_addr & big_page_mask); return dev_addr_base + (gpu_addr & big_page_mask);
} }
@@ -285,8 +299,10 @@ const u8* MemoryManager::GetPointer(GPUVAddr gpu_addr) const {
#pragma inline_recursion(on) #pragma inline_recursion(on)
#endif #endif
template <typename FuncMapped, typename FuncReserved, typename FuncUnmapped> template <bool is_big_pages, typename FuncMapped, typename FuncReserved, typename FuncUnmapped>
inline void MemoryManager::MemoryOperation(GPUVAddr gpu_src_addr, std::size_t size, bool is_big_page, FuncMapped&& func_mapped, FuncReserved&& func_reserved, FuncUnmapped&& func_unmapped) const { inline void MemoryManager::MemoryOperation(GPUVAddr gpu_src_addr, std::size_t size,
FuncMapped&& func_mapped, FuncReserved&& func_reserved,
FuncUnmapped&& func_unmapped) const {
using FuncMappedReturn = using FuncMappedReturn =
typename std::invoke_result<FuncMapped, std::size_t, std::size_t, std::size_t>::type; typename std::invoke_result<FuncMapped, std::size_t, std::size_t, std::size_t>::type;
using FuncReservedReturn = using FuncReservedReturn =
@@ -299,7 +315,7 @@ inline void MemoryManager::MemoryOperation(GPUVAddr gpu_src_addr, std::size_t si
u64 used_page_size; u64 used_page_size;
u64 used_page_mask; u64 used_page_mask;
u64 used_page_bits; u64 used_page_bits;
if (is_big_page) { if constexpr (is_big_pages) {
used_page_size = big_page_size; used_page_size = big_page_size;
used_page_mask = big_page_mask; used_page_mask = big_page_mask;
used_page_bits = big_page_bits; used_page_bits = big_page_bits;
@@ -316,7 +332,7 @@ inline void MemoryManager::MemoryOperation(GPUVAddr gpu_src_addr, std::size_t si
while (remaining_size > 0) { while (remaining_size > 0) {
const std::size_t copy_amount{ const std::size_t copy_amount{
(std::min)(static_cast<std::size_t>(used_page_size) - page_offset, remaining_size)}; (std::min)(static_cast<std::size_t>(used_page_size) - page_offset, remaining_size)};
auto entry = GetEntry(current_address, is_big_page); auto entry = GetEntry<is_big_pages>(current_address);
if (entry == EntryType::Mapped) [[likely]] { if (entry == EntryType::Mapped) [[likely]] {
if constexpr (BOOL_BREAK_MAPPED) { if constexpr (BOOL_BREAK_MAPPED) {
if (func_mapped(page_index, page_offset, copy_amount)) { if (func_mapped(page_index, page_offset, copy_amount)) {
@@ -351,14 +367,18 @@ inline void MemoryManager::MemoryOperation(GPUVAddr gpu_src_addr, std::size_t si
} }
} }
void MemoryManager::ReadBlockImpl(GPUVAddr gpu_src_addr, void* dest_buffer, std::size_t size, [[maybe_unused]] VideoCommon::CacheType which, bool unsafe) const { template <bool is_safe>
auto set_to_zero = [&]([[maybe_unused]] std::size_t page_index, [[maybe_unused]] std::size_t offset, std::size_t copy_amount) { void MemoryManager::ReadBlockImpl(GPUVAddr gpu_src_addr, void* dest_buffer, std::size_t size,
[[maybe_unused]] VideoCommon::CacheType which) const {
auto set_to_zero = [&]([[maybe_unused]] std::size_t page_index,
[[maybe_unused]] std::size_t offset, std::size_t copy_amount) {
std::memset(dest_buffer, 0, copy_amount); std::memset(dest_buffer, 0, copy_amount);
dest_buffer = static_cast<u8*>(dest_buffer) + copy_amount; dest_buffer = static_cast<u8*>(dest_buffer) + copy_amount;
}; };
auto mapped_normal = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto mapped_normal = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) {
const DAddr dev_addr_base = (DAddr(page_table[page_index]) << cpu_page_bits) + offset; const DAddr dev_addr_base =
if (!unsafe) { (static_cast<DAddr>(page_table[page_index]) << cpu_page_bits) + offset;
if constexpr (is_safe) {
rasterizer->FlushRegion(dev_addr_base, copy_amount, which); rasterizer->FlushRegion(dev_addr_base, copy_amount, which);
} }
u8* physical = memory.GetPointer<u8>(dev_addr_base); u8* physical = memory.GetPointer<u8>(dev_addr_base);
@@ -366,8 +386,9 @@ void MemoryManager::ReadBlockImpl(GPUVAddr gpu_src_addr, void* dest_buffer, std:
dest_buffer = static_cast<u8*>(dest_buffer) + copy_amount; dest_buffer = static_cast<u8*>(dest_buffer) + copy_amount;
}; };
auto mapped_big = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto mapped_big = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) {
const DAddr dev_addr_base = (DAddr(big_page_table_dev[page_index]) << cpu_page_bits) + offset; const DAddr dev_addr_base =
if (!unsafe) { (static_cast<DAddr>(big_page_table_dev[page_index]) << cpu_page_bits) + offset;
if constexpr (is_safe) {
rasterizer->FlushRegion(dev_addr_base, copy_amount, which); rasterizer->FlushRegion(dev_addr_base, copy_amount, which);
} }
if (!IsBigPageContinuous(page_index)) [[unlikely]] { if (!IsBigPageContinuous(page_index)) [[unlikely]] {
@@ -378,28 +399,35 @@ void MemoryManager::ReadBlockImpl(GPUVAddr gpu_src_addr, void* dest_buffer, std:
} }
dest_buffer = static_cast<u8*>(dest_buffer) + copy_amount; dest_buffer = static_cast<u8*>(dest_buffer) + copy_amount;
}; };
auto read_short_pages = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto read_short_pages = [&](std::size_t page_index, std::size_t offset,
std::size_t copy_amount) {
GPUVAddr base = (page_index << big_page_bits) + offset; GPUVAddr base = (page_index << big_page_bits) + offset;
MemoryOperation(base, copy_amount, false, mapped_normal, set_to_zero, set_to_zero); MemoryOperation<false>(base, copy_amount, mapped_normal, set_to_zero, set_to_zero);
}; };
MemoryOperation(gpu_src_addr, size, true, mapped_big, set_to_zero, read_short_pages); MemoryOperation<true>(gpu_src_addr, size, mapped_big, set_to_zero, read_short_pages);
} }
void MemoryManager::ReadBlock(GPUVAddr gpu_src_addr, void* dest_buffer, std::size_t size, VideoCommon::CacheType which) const { void MemoryManager::ReadBlock(GPUVAddr gpu_src_addr, void* dest_buffer, std::size_t size,
ReadBlockImpl(gpu_src_addr, dest_buffer, size, which, false); VideoCommon::CacheType which) const {
ReadBlockImpl<true>(gpu_src_addr, dest_buffer, size, which);
} }
void MemoryManager::ReadBlockUnsafe(GPUVAddr gpu_src_addr, void* dest_buffer, const std::size_t size) const { void MemoryManager::ReadBlockUnsafe(GPUVAddr gpu_src_addr, void* dest_buffer,
ReadBlockImpl(gpu_src_addr, dest_buffer, size, VideoCommon::CacheType::None, true); const std::size_t size) const {
ReadBlockImpl<false>(gpu_src_addr, dest_buffer, size, VideoCommon::CacheType::None);
} }
void MemoryManager::WriteBlockImpl(GPUVAddr gpu_dest_addr, const void* src_buffer, std::size_t size, [[maybe_unused]] VideoCommon::CacheType which, bool unsafe) { template <bool is_safe>
auto just_advance = [&]([[maybe_unused]] std::size_t page_index, [[maybe_unused]] std::size_t offset, std::size_t copy_amount) { void MemoryManager::WriteBlockImpl(GPUVAddr gpu_dest_addr, const void* src_buffer, std::size_t size,
[[maybe_unused]] VideoCommon::CacheType which) {
auto just_advance = [&]([[maybe_unused]] std::size_t page_index,
[[maybe_unused]] std::size_t offset, std::size_t copy_amount) {
src_buffer = static_cast<const u8*>(src_buffer) + copy_amount; src_buffer = static_cast<const u8*>(src_buffer) + copy_amount;
}; };
auto mapped_normal = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto mapped_normal = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) {
const DAddr dev_addr_base = (DAddr(page_table[page_index]) << cpu_page_bits) + offset; const DAddr dev_addr_base =
if (!unsafe) { (static_cast<DAddr>(page_table[page_index]) << cpu_page_bits) + offset;
if constexpr (is_safe) {
rasterizer->InvalidateRegion(dev_addr_base, copy_amount, which); rasterizer->InvalidateRegion(dev_addr_base, copy_amount, which);
} }
u8* physical = memory.GetPointer<u8>(dev_addr_base); u8* physical = memory.GetPointer<u8>(dev_addr_base);
@@ -407,8 +435,9 @@ void MemoryManager::WriteBlockImpl(GPUVAddr gpu_dest_addr, const void* src_buffe
src_buffer = static_cast<const u8*>(src_buffer) + copy_amount; src_buffer = static_cast<const u8*>(src_buffer) + copy_amount;
}; };
auto mapped_big = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto mapped_big = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) {
const DAddr dev_addr_base = (DAddr(big_page_table_dev[page_index]) << cpu_page_bits) + offset; const DAddr dev_addr_base =
if (!unsafe) { (static_cast<DAddr>(big_page_table_dev[page_index]) << cpu_page_bits) + offset;
if constexpr (is_safe) {
rasterizer->InvalidateRegion(dev_addr_base, copy_amount, which); rasterizer->InvalidateRegion(dev_addr_base, copy_amount, which);
} }
if (!IsBigPageContinuous(page_index)) [[unlikely]] { if (!IsBigPageContinuous(page_index)) [[unlikely]] {
@@ -419,23 +448,26 @@ void MemoryManager::WriteBlockImpl(GPUVAddr gpu_dest_addr, const void* src_buffe
} }
src_buffer = static_cast<const u8*>(src_buffer) + copy_amount; src_buffer = static_cast<const u8*>(src_buffer) + copy_amount;
}; };
auto write_short_pages = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto write_short_pages = [&](std::size_t page_index, std::size_t offset,
std::size_t copy_amount) {
GPUVAddr base = (page_index << big_page_bits) + offset; GPUVAddr base = (page_index << big_page_bits) + offset;
MemoryOperation(base, copy_amount, false, mapped_normal, just_advance, just_advance); MemoryOperation<false>(base, copy_amount, mapped_normal, just_advance, just_advance);
}; };
MemoryOperation(gpu_dest_addr, size, true, mapped_big, just_advance, write_short_pages); MemoryOperation<true>(gpu_dest_addr, size, mapped_big, just_advance, write_short_pages);
} }
void MemoryManager::WriteBlock(GPUVAddr gpu_dest_addr, const void* src_buffer, std::size_t size, VideoCommon::CacheType which) { void MemoryManager::WriteBlock(GPUVAddr gpu_dest_addr, const void* src_buffer, std::size_t size,
WriteBlockImpl(gpu_dest_addr, src_buffer, size, which, false); VideoCommon::CacheType which) {
WriteBlockImpl<true>(gpu_dest_addr, src_buffer, size, which);
} }
void MemoryManager::WriteBlockUnsafe(GPUVAddr gpu_dest_addr, const void* src_buffer, std::size_t size) { void MemoryManager::WriteBlockUnsafe(GPUVAddr gpu_dest_addr, const void* src_buffer,
WriteBlockImpl(gpu_dest_addr, src_buffer, size, VideoCommon::CacheType::None, true); std::size_t size) {
WriteBlockImpl<false>(gpu_dest_addr, src_buffer, size, VideoCommon::CacheType::None);
} }
void MemoryManager::WriteBlockCached(GPUVAddr gpu_dest_addr, const void* src_buffer, std::size_t size) { void MemoryManager::WriteBlockCached(GPUVAddr gpu_dest_addr, const void* src_buffer, std::size_t size) {
WriteBlockImpl(gpu_dest_addr, src_buffer, size, VideoCommon::CacheType::None, true); WriteBlockImpl<false>(gpu_dest_addr, src_buffer, size, VideoCommon::CacheType::None);
accumulator.Add(gpu_dest_addr, size); accumulator.Add(gpu_dest_addr, size);
} }
@@ -446,18 +478,21 @@ void MemoryManager::FlushRegion(GPUVAddr gpu_addr, size_t size,
[[maybe_unused]] std::size_t copy_amount) {}; [[maybe_unused]] std::size_t copy_amount) {};
auto mapped_normal = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto mapped_normal = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) {
const DAddr dev_addr_base = (DAddr(page_table[page_index]) << cpu_page_bits) + offset; const DAddr dev_addr_base =
(static_cast<DAddr>(page_table[page_index]) << cpu_page_bits) + offset;
rasterizer->FlushRegion(dev_addr_base, copy_amount, which); rasterizer->FlushRegion(dev_addr_base, copy_amount, which);
}; };
auto mapped_big = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto mapped_big = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) {
const DAddr dev_addr_base = (DAddr(big_page_table_dev[page_index]) << cpu_page_bits) + offset; const DAddr dev_addr_base =
(static_cast<DAddr>(big_page_table_dev[page_index]) << cpu_page_bits) + offset;
rasterizer->FlushRegion(dev_addr_base, copy_amount, which); rasterizer->FlushRegion(dev_addr_base, copy_amount, which);
}; };
auto flush_short_pages = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto flush_short_pages = [&](std::size_t page_index, std::size_t offset,
std::size_t copy_amount) {
GPUVAddr base = (page_index << big_page_bits) + offset; GPUVAddr base = (page_index << big_page_bits) + offset;
MemoryOperation(base, copy_amount, false, mapped_normal, do_nothing, do_nothing); MemoryOperation<false>(base, copy_amount, mapped_normal, do_nothing, do_nothing);
}; };
MemoryOperation(gpu_addr, size, true, mapped_big, do_nothing, flush_short_pages); MemoryOperation<true>(gpu_addr, size, mapped_big, do_nothing, flush_short_pages);
} }
bool MemoryManager::IsMemoryDirty(GPUVAddr gpu_addr, size_t size, bool MemoryManager::IsMemoryDirty(GPUVAddr gpu_addr, size_t size,
@@ -482,10 +517,10 @@ bool MemoryManager::IsMemoryDirty(GPUVAddr gpu_addr, size_t size,
auto check_short_pages = [&](std::size_t page_index, std::size_t offset, auto check_short_pages = [&](std::size_t page_index, std::size_t offset,
std::size_t copy_amount) { std::size_t copy_amount) {
GPUVAddr base = (page_index << big_page_bits) + offset; GPUVAddr base = (page_index << big_page_bits) + offset;
MemoryOperation(base, copy_amount, false, mapped_normal, do_nothing, do_nothing); MemoryOperation<false>(base, copy_amount, mapped_normal, do_nothing, do_nothing);
return result; return result;
}; };
MemoryOperation(gpu_addr, size, true, mapped_big, do_nothing, check_short_pages); MemoryOperation<true>(gpu_addr, size, mapped_big, do_nothing, check_short_pages);
return result; return result;
} }
@@ -522,10 +557,10 @@ size_t MemoryManager::MaxContinuousRange(GPUVAddr gpu_addr, size_t size) const {
auto check_short_pages = [&](std::size_t page_index, std::size_t offset, auto check_short_pages = [&](std::size_t page_index, std::size_t offset,
std::size_t copy_amount) { std::size_t copy_amount) {
GPUVAddr base = (page_index << big_page_bits) + offset; GPUVAddr base = (page_index << big_page_bits) + offset;
MemoryOperation(base, copy_amount, false, short_check, fail, fail); MemoryOperation<false>(base, copy_amount, short_check, fail, fail);
return result; return result;
}; };
MemoryOperation(gpu_addr, size, true, big_check, fail, check_short_pages); MemoryOperation<true>(gpu_addr, size, big_check, fail, check_short_pages);
return range_so_far; return range_so_far;
} }
@@ -541,18 +576,21 @@ void MemoryManager::InvalidateRegion(GPUVAddr gpu_addr, size_t size,
[[maybe_unused]] std::size_t copy_amount) {}; [[maybe_unused]] std::size_t copy_amount) {};
auto mapped_normal = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto mapped_normal = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) {
const DAddr dev_addr_base = (DAddr(page_table[page_index]) << cpu_page_bits) + offset; const DAddr dev_addr_base =
(static_cast<DAddr>(page_table[page_index]) << cpu_page_bits) + offset;
rasterizer->InvalidateRegion(dev_addr_base, copy_amount, which); rasterizer->InvalidateRegion(dev_addr_base, copy_amount, which);
}; };
auto mapped_big = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto mapped_big = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) {
const DAddr dev_addr_base = (DAddr(big_page_table_dev[page_index]) << cpu_page_bits) + offset; const DAddr dev_addr_base =
(static_cast<DAddr>(big_page_table_dev[page_index]) << cpu_page_bits) + offset;
rasterizer->InvalidateRegion(dev_addr_base, copy_amount, which); rasterizer->InvalidateRegion(dev_addr_base, copy_amount, which);
}; };
auto invalidate_short_pages = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto invalidate_short_pages = [&](std::size_t page_index, std::size_t offset,
std::size_t copy_amount) {
GPUVAddr base = (page_index << big_page_bits) + offset; GPUVAddr base = (page_index << big_page_bits) + offset;
MemoryOperation(base, copy_amount, false, mapped_normal, do_nothing, do_nothing); MemoryOperation<false>(base, copy_amount, mapped_normal, do_nothing, do_nothing);
}; };
MemoryOperation(gpu_addr, size, true, mapped_big, do_nothing, invalidate_short_pages); MemoryOperation<true>(gpu_addr, size, mapped_big, do_nothing, invalidate_short_pages);
} }
void MemoryManager::CopyBlock(GPUVAddr gpu_dest_addr, GPUVAddr gpu_src_addr, std::size_t size, void MemoryManager::CopyBlock(GPUVAddr gpu_dest_addr, GPUVAddr gpu_src_addr, std::size_t size,
@@ -564,7 +602,7 @@ void MemoryManager::CopyBlock(GPUVAddr gpu_dest_addr, GPUVAddr gpu_src_addr, std
} }
bool MemoryManager::IsGranularRange(GPUVAddr gpu_addr, std::size_t size) const { bool MemoryManager::IsGranularRange(GPUVAddr gpu_addr, std::size_t size) const {
if (GetEntry(gpu_addr, true) == EntryType::Mapped) [[likely]] { if (GetEntry<true>(gpu_addr) == EntryType::Mapped) [[likely]] {
size_t page_index = gpu_addr >> big_page_bits; size_t page_index = gpu_addr >> big_page_bits;
if (IsBigPageContinuous(page_index)) [[likely]] { if (IsBigPageContinuous(page_index)) [[likely]] {
const std::size_t page{(page_index & big_page_mask) + size}; const std::size_t page{(page_index & big_page_mask) + size};
@@ -573,7 +611,7 @@ bool MemoryManager::IsGranularRange(GPUVAddr gpu_addr, std::size_t size) const {
const std::size_t page{(gpu_addr & Core::DEVICE_PAGEMASK) + size}; const std::size_t page{(gpu_addr & Core::DEVICE_PAGEMASK) + size};
return page <= Core::DEVICE_PAGESIZE; return page <= Core::DEVICE_PAGESIZE;
} }
if (GetEntry(gpu_addr, false) != EntryType::Mapped) { if (GetEntry<false>(gpu_addr) != EntryType::Mapped) {
return false; return false;
} }
const std::size_t page{(gpu_addr & Core::DEVICE_PAGEMASK) + size}; const std::size_t page{(gpu_addr & Core::DEVICE_PAGEMASK) + size};
@@ -611,10 +649,10 @@ bool MemoryManager::IsContinuousRange(GPUVAddr gpu_addr, std::size_t size) const
auto check_short_pages = [&](std::size_t page_index, std::size_t offset, auto check_short_pages = [&](std::size_t page_index, std::size_t offset,
std::size_t copy_amount) { std::size_t copy_amount) {
GPUVAddr base = (page_index << big_page_bits) + offset; GPUVAddr base = (page_index << big_page_bits) + offset;
MemoryOperation(base, copy_amount, false, short_check, fail, fail); MemoryOperation<false>(base, copy_amount, short_check, fail, fail);
return !result; return !result;
}; };
MemoryOperation(gpu_addr, size, true, big_check, fail, check_short_pages); MemoryOperation<true>(gpu_addr, size, big_check, fail, check_short_pages);
return result; return result;
} }
@@ -627,12 +665,13 @@ bool MemoryManager::IsFullyMappedRange(GPUVAddr gpu_addr, std::size_t size) cons
}; };
auto pass = [&]([[maybe_unused]] std::size_t page_index, [[maybe_unused]] std::size_t offset, auto pass = [&]([[maybe_unused]] std::size_t page_index, [[maybe_unused]] std::size_t offset,
[[maybe_unused]] std::size_t copy_amount) { return false; }; [[maybe_unused]] std::size_t copy_amount) { return false; };
auto check_short_pages = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto check_short_pages = [&](std::size_t page_index, std::size_t offset,
std::size_t copy_amount) {
GPUVAddr base = (page_index << big_page_bits) + offset; GPUVAddr base = (page_index << big_page_bits) + offset;
MemoryOperation(base, copy_amount, false, pass, pass, fail); MemoryOperation<false>(base, copy_amount, pass, pass, fail);
return !result; return !result;
}; };
MemoryOperation(gpu_addr, size, true, pass, fail, check_short_pages); MemoryOperation<true>(gpu_addr, size, pass, fail, check_short_pages);
return result; return result;
} }
@@ -644,9 +683,13 @@ MemoryManager::GetSubmappedRange(GPUVAddr gpu_addr, std::size_t size) const {
} }
template <bool is_gpu_address> template <bool is_gpu_address>
void MemoryManager::GetSubmappedRangeImpl(GPUVAddr gpu_addr, std::size_t size, boost::container::small_vector<std::pair<std::conditional_t<is_gpu_address, GPUVAddr, DAddr>, std::size_t>, 32>& result) void MemoryManager::GetSubmappedRangeImpl(
GPUVAddr gpu_addr, std::size_t size,
boost::container::small_vector<
std::pair<std::conditional_t<is_gpu_address, GPUVAddr, DAddr>, std::size_t>, 32>& result)
const { const {
std::optional<std::pair<std::conditional_t<is_gpu_address, GPUVAddr, DAddr>, std::size_t>> last_segment{}; std::optional<std::pair<std::conditional_t<is_gpu_address, GPUVAddr, DAddr>, std::size_t>>
last_segment{};
std::optional<DAddr> old_page_addr{}; std::optional<DAddr> old_page_addr{};
const auto split = [&last_segment, &result]([[maybe_unused]] std::size_t page_index, const auto split = [&last_segment, &result]([[maybe_unused]] std::size_t page_index,
[[maybe_unused]] std::size_t offset, [[maybe_unused]] std::size_t offset,
@@ -702,9 +745,9 @@ void MemoryManager::GetSubmappedRangeImpl(GPUVAddr gpu_addr, std::size_t size, b
}; };
auto do_short_pages = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) { auto do_short_pages = [&](std::size_t page_index, std::size_t offset, std::size_t copy_amount) {
GPUVAddr base = (page_index << big_page_bits) + offset; GPUVAddr base = (page_index << big_page_bits) + offset;
MemoryOperation(base, copy_amount, false, extend_size_short, split, split); MemoryOperation<false>(base, copy_amount, extend_size_short, split, split);
}; };
MemoryOperation(gpu_addr, size, true, extend_size_big, split, do_short_pages); MemoryOperation<true>(gpu_addr, size, extend_size_big, split, do_short_pages);
split(0, 0, 0); split(0, 0, 0);
} }
+40 -19
View File
@@ -45,7 +45,7 @@ public:
static constexpr bool HAS_FLUSH_INVALIDATION = true; static constexpr bool HAS_FLUSH_INVALIDATION = true;
inline size_t GetID() const noexcept { size_t GetID() const {
return unique_identifier; return unique_identifier;
} }
@@ -66,15 +66,16 @@ public:
[[nodiscard]] const u8* GetPointer(GPUVAddr addr) const; [[nodiscard]] const u8* GetPointer(GPUVAddr addr) const;
template <typename T> template <typename T>
[[nodiscard]] inline T* GetPointer(GPUVAddr addr) noexcept { [[nodiscard]] T* GetPointer(GPUVAddr addr) {
const auto address = GpuToCpuAddress(addr); const auto address{GpuToCpuAddress(addr)};
if (!address) if (!address) {
return {}; return {};
}
return memory.GetPointer<T>(*address); return memory.GetPointer<T>(*address);
} }
template <typename T> template <typename T>
[[nodiscard]] inline const T* GetPointer(GPUVAddr addr) const noexcept { [[nodiscard]] const T* GetPointer(GPUVAddr addr) const {
return GetPointer<T*>(addr); return GetPointer<T*>(addr);
} }
@@ -84,9 +85,12 @@ public:
* in the Host Memory counterpart. Note: This functions cause Host GPU Memory * in the Host Memory counterpart. Note: This functions cause Host GPU Memory
* Flushes and Invalidations, respectively to each operation. * Flushes and Invalidations, respectively to each operation.
*/ */
void ReadBlock(GPUVAddr gpu_src_addr, void* dest_buffer, std::size_t size, VideoCommon::CacheType which = VideoCommon::CacheType::All) const; void ReadBlock(GPUVAddr gpu_src_addr, void* dest_buffer, std::size_t size,
void WriteBlock(GPUVAddr gpu_dest_addr, const void* src_buffer, std::size_t size, VideoCommon::CacheType which = VideoCommon::CacheType::All); VideoCommon::CacheType which = VideoCommon::CacheType::All) const;
void CopyBlock(GPUVAddr gpu_dest_addr, GPUVAddr gpu_src_addr, std::size_t size, VideoCommon::CacheType which = VideoCommon::CacheType::All); void WriteBlock(GPUVAddr gpu_dest_addr, const void* src_buffer, std::size_t size,
VideoCommon::CacheType which = VideoCommon::CacheType::All);
void CopyBlock(GPUVAddr gpu_dest_addr, GPUVAddr gpu_src_addr, std::size_t size,
VideoCommon::CacheType which = VideoCommon::CacheType::All);
/** /**
* ReadBlockUnsafe and WriteBlockUnsafe are special versions of ReadBlock and * ReadBlockUnsafe and WriteBlockUnsafe are special versions of ReadBlock and
@@ -156,14 +160,21 @@ public:
u8* GetSpan(const GPUVAddr src_addr, const std::size_t size); u8* GetSpan(const GPUVAddr src_addr, const std::size_t size);
private: private:
template <typename FuncMapped, typename FuncReserved, typename FuncUnmapped> template <bool is_big_pages, typename FuncMapped, typename FuncReserved, typename FuncUnmapped>
inline void MemoryOperation(GPUVAddr gpu_src_addr, std::size_t size, bool is_big_page, FuncMapped&& func_mapped, FuncReserved&& func_reserved, FuncUnmapped&& func_unmapped) const; inline void MemoryOperation(GPUVAddr gpu_src_addr, std::size_t size, FuncMapped&& func_mapped,
FuncReserved&& func_reserved, FuncUnmapped&& func_unmapped) const;
void ReadBlockImpl(GPUVAddr gpu_src_addr, void* dest_buffer, std::size_t size, VideoCommon::CacheType which, bool unsafe) const; template <bool is_safe>
void WriteBlockImpl(GPUVAddr gpu_dest_addr, const void* src_buffer, std::size_t size, VideoCommon::CacheType which, bool unsafe); void ReadBlockImpl(GPUVAddr gpu_src_addr, void* dest_buffer, std::size_t size,
VideoCommon::CacheType which) const;
[[nodiscard]] std::size_t PageEntryIndex(GPUVAddr gpu_addr, bool is_big_page) const { template <bool is_safe>
if (is_big_page) { void WriteBlockImpl(GPUVAddr gpu_dest_addr, const void* src_buffer, std::size_t size,
VideoCommon::CacheType which);
template <bool is_big_page>
[[nodiscard]] std::size_t PageEntryIndex(GPUVAddr gpu_addr) const {
if constexpr (is_big_page) {
return (gpu_addr >> big_page_bits) & big_page_table_mask; return (gpu_addr >> big_page_bits) & big_page_table_mask;
} else { } else {
return (gpu_addr >> page_bits) & page_table_mask; return (gpu_addr >> page_bits) & page_table_mask;
@@ -176,7 +187,9 @@ private:
template <bool is_gpu_address> template <bool is_gpu_address>
void GetSubmappedRangeImpl( void GetSubmappedRangeImpl(
GPUVAddr gpu_addr, std::size_t size, GPUVAddr gpu_addr, std::size_t size,
boost::container::small_vector<std::pair<std::conditional_t<is_gpu_address, GPUVAddr, DAddr>, std::size_t>, 32>& result) const; boost::container::small_vector<
std::pair<std::conditional_t<is_gpu_address, GPUVAddr, DAddr>, std::size_t>, 32>&
result) const;
Core::System& system; Core::System& system;
MaxwellDeviceMemoryManager& memory; MaxwellDeviceMemoryManager& memory;
@@ -206,11 +219,19 @@ private:
std::vector<u64> entries; std::vector<u64> entries;
std::vector<u64> big_entries; std::vector<u64> big_entries;
GPUVAddr PageTableOp(GPUVAddr gpu_addr, [[maybe_unused]] DAddr dev_addr, size_t size, PTEKind kind, EntryType entry_type); template <EntryType entry_type>
GPUVAddr BigPageTableOp(GPUVAddr gpu_addr, [[maybe_unused]] DAddr dev_addr, size_t size, PTEKind kind, EntryType entry_type); GPUVAddr PageTableOp(GPUVAddr gpu_addr, [[maybe_unused]] DAddr dev_addr, size_t size,
PTEKind kind);
inline EntryType GetEntry(size_t position, bool is_big_page) const; template <EntryType entry_type>
inline void SetEntry(size_t position, EntryType entry, bool is_big_page); GPUVAddr BigPageTableOp(GPUVAddr gpu_addr, [[maybe_unused]] DAddr dev_addr, size_t size,
PTEKind kind);
template <bool is_big_page>
inline EntryType GetEntry(size_t position) const;
template <bool is_big_page>
inline void SetEntry(size_t position, EntryType entry);
Common::MultiLevelPageTable<u32> page_table; Common::MultiLevelPageTable<u32> page_table;
Common::RangeMap<GPUVAddr, PTEKind> kind_map; Common::RangeMap<GPUVAddr, PTEKind> kind_map;
@@ -485,7 +485,6 @@ void RasterizerOpenGL::FlushRegion(DAddr addr, u64 size, VideoCommon::CacheType
texture_cache.DownloadMemory(addr, size); texture_cache.DownloadMemory(addr, size);
} }
if ((True(which & VideoCommon::CacheType::BufferCache))) { if ((True(which & VideoCommon::CacheType::BufferCache))) {
std::scoped_lock lock{buffer_cache.mutex};
buffer_cache.DownloadMemory(addr, size); buffer_cache.DownloadMemory(addr, size);
} }
if ((True(which & VideoCommon::CacheType::QueryCache))) { if ((True(which & VideoCommon::CacheType::QueryCache))) {
@@ -1,9 +1,13 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
#include <cstddef> #include <cstddef>
#include "video_core/renderer_vulkan/vk_command_pool.h" #include "video_core/renderer_vulkan/vk_command_pool.h"
#include "video_core/renderer_vulkan/vk_master_semaphore.h"
#include "video_core/vulkan_common/vulkan_device.h" #include "video_core/vulkan_common/vulkan_device.h"
#include "video_core/vulkan_common/vulkan_wrapper.h" #include "video_core/vulkan_common/vulkan_wrapper.h"
@@ -14,32 +18,52 @@ constexpr size_t COMMAND_BUFFER_POOL_SIZE = 4;
struct CommandPool::Pool { struct CommandPool::Pool {
vk::CommandPool handle; vk::CommandPool handle;
vk::CommandBuffers cmdbufs; vk::CommandBuffers cmdbufs;
u64 tick;
}; };
CommandPool::CommandPool(MasterSemaphore& master_semaphore_, const Device& device_) CommandPool::CommandPool(MasterSemaphore& master_semaphore_, const Device& device_)
: ResourcePool(master_semaphore_, COMMAND_BUFFER_POOL_SIZE), device{device_} {} : master_semaphore{master_semaphore_}, device{device_} {}
CommandPool::~CommandPool() = default; CommandPool::~CommandPool() = default;
void CommandPool::Allocate(size_t begin, size_t end) { void CommandPool::AllocatePool() {
// Command buffers are going to be committed, recorded, executed every single usage cycle.
// They are also going to be reset when committed.
Pool& pool = pools.emplace_back(); Pool& pool = pools.emplace_back();
pool.handle = device.GetLogical().CreateCommandPool({ pool.handle = device.GetLogical().CreateCommandPool({
.sType = VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO, .sType = VK_STRUCTURE_TYPE_COMMAND_POOL_CREATE_INFO,
.pNext = nullptr, .pNext = nullptr,
.flags = .flags = VK_COMMAND_POOL_CREATE_TRANSIENT_BIT,
VK_COMMAND_POOL_CREATE_TRANSIENT_BIT | VK_COMMAND_POOL_CREATE_RESET_COMMAND_BUFFER_BIT,
.queueFamilyIndex = device.GetGraphicsFamily(), .queueFamilyIndex = device.GetGraphicsFamily(),
}); });
pool.cmdbufs = pool.handle.Allocate(COMMAND_BUFFER_POOL_SIZE); pool.cmdbufs = pool.handle.Allocate(COMMAND_BUFFER_POOL_SIZE);
pool.tick = 0;
}
void CommandPool::AcquirePool() {
if (!pools.empty()) {
master_semaphore.Refresh();
const u64 gpu_tick = master_semaphore.KnownGpuTick();
for (size_t i = 0; i < pools.size(); ++i) {
const size_t candidate = (current_pool + 1 + i) % pools.size();
if (gpu_tick >= pools[candidate].tick) {
current_pool = candidate;
current_index = 0;
pools[current_pool].handle.Reset();
return;
}
}
}
AllocatePool();
current_pool = pools.size() - 1;
current_index = 0;
} }
VkCommandBuffer CommandPool::Commit() { VkCommandBuffer CommandPool::Commit() {
const size_t index = CommitResource(); if (pools.empty() || current_index >= COMMAND_BUFFER_POOL_SIZE) {
const auto pool_index = index / COMMAND_BUFFER_POOL_SIZE; AcquirePool();
const auto sub_index = index % COMMAND_BUFFER_POOL_SIZE; }
return pools[pool_index].cmdbufs[sub_index]; Pool& pool = pools[current_pool];
pool.tick = master_semaphore.CurrentTick();
return pool.cmdbufs[current_index++];
} }
} // namespace Vulkan } // namespace Vulkan
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
@@ -6,7 +9,7 @@
#include <cstddef> #include <cstddef>
#include <vector> #include <vector>
#include "video_core/renderer_vulkan/vk_resource_pool.h" #include "common/common_types.h"
#include "video_core/vulkan_common/vulkan_wrapper.h" #include "video_core/vulkan_common/vulkan_wrapper.h"
namespace Vulkan { namespace Vulkan {
@@ -14,20 +17,24 @@ namespace Vulkan {
class Device; class Device;
class MasterSemaphore; class MasterSemaphore;
class CommandPool final : public ResourcePool { class CommandPool final {
public: public:
explicit CommandPool(MasterSemaphore& master_semaphore_, const Device& device_); explicit CommandPool(MasterSemaphore& master_semaphore_, const Device& device_);
~CommandPool() override; ~CommandPool();
void Allocate(size_t begin, size_t end) override;
VkCommandBuffer Commit(); VkCommandBuffer Commit();
private: private:
struct Pool; struct Pool;
void AllocatePool();
void AcquirePool();
MasterSemaphore& master_semaphore;
const Device& device; const Device& device;
std::vector<Pool> pools; std::vector<Pool> pools;
size_t current_pool = 0;
size_t current_index = 0;
}; };
} // namespace Vulkan } // namespace Vulkan
@@ -4,6 +4,7 @@
// SPDX-FileCopyrightText: Copyright 2019 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2019 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
#include <algorithm>
#include <array> #include <array>
#include <memory> #include <memory>
#include <numeric> #include <numeric>
@@ -17,6 +18,8 @@
#include "common/div_ceil.h" #include "common/div_ceil.h"
#include "common/vector_math.h" #include "common/vector_math.h"
#include "video_core/host_shaders/astc_decoder_comp_spv.h" #include "video_core/host_shaders/astc_decoder_comp_spv.h"
#include "video_core/host_shaders/astc_decoder_frag_spv.h"
#include "video_core/host_shaders/astc_decoder_vert_spv.h"
#include "video_core/host_shaders/queries_prefix_scan_sum_comp_spv.h" #include "video_core/host_shaders/queries_prefix_scan_sum_comp_spv.h"
#include "video_core/host_shaders/queries_prefix_scan_sum_nosubgroups_comp_spv.h" #include "video_core/host_shaders/queries_prefix_scan_sum_nosubgroups_comp_spv.h"
#include "video_core/host_shaders/resolve_conditional_render_comp_spv.h" #include "video_core/host_shaders/resolve_conditional_render_comp_spv.h"
@@ -25,8 +28,11 @@
#include "video_core/host_shaders/block_linear_unswizzle_3d_bcn_comp_spv.h" #include "video_core/host_shaders/block_linear_unswizzle_3d_bcn_comp_spv.h"
#include "video_core/renderer_vulkan/vk_compute_pass.h" #include "video_core/renderer_vulkan/vk_compute_pass.h"
#include "video_core/surface.h" #include "video_core/surface.h"
#include "video_core/renderer_vulkan/maxwell_to_vk.h"
#include "video_core/renderer_vulkan/vk_descriptor_pool.h" #include "video_core/renderer_vulkan/vk_descriptor_pool.h"
#include "video_core/renderer_vulkan/vk_render_pass_cache.h"
#include "video_core/renderer_vulkan/vk_scheduler.h" #include "video_core/renderer_vulkan/vk_scheduler.h"
#include "video_core/renderer_vulkan/vk_shader_util.h"
#include "video_core/renderer_vulkan/vk_staging_buffer_pool.h" #include "video_core/renderer_vulkan/vk_staging_buffer_pool.h"
#include "video_core/renderer_vulkan/vk_update_descriptor.h" #include "video_core/renderer_vulkan/vk_update_descriptor.h"
#include "video_core/texture_cache/accelerated_swizzle.h" #include "video_core/texture_cache/accelerated_swizzle.h"
@@ -613,7 +619,394 @@ void ASTCDecoderPass::Assemble(Image& image, const StagingBufferRef& map,
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
vk::PIPELINE_STAGE_GRAPHICS_COMPUTE, 0, image_barrier); vk::PIPELINE_STAGE_GRAPHICS_COMPUTE, 0, image_barrier);
}); });
scheduler.Finish(); }
namespace {
struct AstcFragPushConstants {
std::array<u32, 2> blocks_dims;
u32 layer_stride;
u32 block_size;
u32 x_shift;
u32 block_height;
u32 block_height_mask;
u32 dest_layer;
};
constexpr VkDescriptorSetLayoutBinding ASTC_FRAG_BINDING{
.binding = 0,
.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER,
.descriptorCount = 1,
.stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT,
.pImmutableSamplers = nullptr,
};
constexpr VkDescriptorUpdateTemplateEntry ASTC_FRAG_TEMPLATE{
.dstBinding = 0,
.dstArrayElement = 0,
.descriptorCount = 1,
.descriptorType = VK_DESCRIPTOR_TYPE_STORAGE_BUFFER,
.offset = 0,
.stride = sizeof(DescriptorUpdateEntry),
};
constexpr DescriptorBankInfo ASTC_FRAG_BANK_INFO{
.uniform_buffers = 0,
.storage_buffers = 1,
.texture_buffers = 0,
.image_buffers = 0,
.textures = 0,
.images = 0,
.score = 1,
};
constexpr VkPushConstantRange ASTC_FRAG_PUSH_CONSTANT_RANGE{
.stageFlags = VK_SHADER_STAGE_FRAGMENT_BIT,
.offset = 0,
.size = static_cast<u32>(sizeof(AstcFragPushConstants)),
};
constexpr VkPipelineVertexInputStateCreateInfo ASTC_FRAG_VERTEX_INPUT{
.sType = VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.vertexBindingDescriptionCount = 0,
.pVertexBindingDescriptions = nullptr,
.vertexAttributeDescriptionCount = 0,
.pVertexAttributeDescriptions = nullptr,
};
constexpr VkPipelineInputAssemblyStateCreateInfo ASTC_FRAG_INPUT_ASSEMBLY{
.sType = VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.topology = VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST,
.primitiveRestartEnable = VK_FALSE,
};
constexpr VkPipelineViewportStateCreateInfo ASTC_FRAG_VIEWPORT{
.sType = VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.viewportCount = 1,
.pViewports = nullptr,
.scissorCount = 1,
.pScissors = nullptr,
};
constexpr VkPipelineRasterizationStateCreateInfo ASTC_FRAG_RASTERIZATION{
.sType = VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.depthClampEnable = VK_FALSE,
.rasterizerDiscardEnable = VK_FALSE,
.polygonMode = VK_POLYGON_MODE_FILL,
.cullMode = VK_CULL_MODE_NONE,
.frontFace = VK_FRONT_FACE_COUNTER_CLOCKWISE,
.depthBiasEnable = VK_FALSE,
.depthBiasConstantFactor = 0.0f,
.depthBiasClamp = 0.0f,
.depthBiasSlopeFactor = 0.0f,
.lineWidth = 1.0f,
};
constexpr VkPipelineMultisampleStateCreateInfo ASTC_FRAG_MULTISAMPLE{
.sType = VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.rasterizationSamples = VK_SAMPLE_COUNT_1_BIT,
.sampleShadingEnable = VK_FALSE,
.minSampleShading = 0.0f,
.pSampleMask = nullptr,
.alphaToCoverageEnable = VK_FALSE,
.alphaToOneEnable = VK_FALSE,
};
constexpr VkPipelineColorBlendAttachmentState ASTC_FRAG_BLEND_ATTACHMENT{
.blendEnable = VK_FALSE,
.srcColorBlendFactor = VK_BLEND_FACTOR_ONE,
.dstColorBlendFactor = VK_BLEND_FACTOR_ZERO,
.colorBlendOp = VK_BLEND_OP_ADD,
.srcAlphaBlendFactor = VK_BLEND_FACTOR_ONE,
.dstAlphaBlendFactor = VK_BLEND_FACTOR_ZERO,
.alphaBlendOp = VK_BLEND_OP_ADD,
.colorWriteMask = VK_COLOR_COMPONENT_R_BIT | VK_COLOR_COMPONENT_G_BIT |
VK_COLOR_COMPONENT_B_BIT | VK_COLOR_COMPONENT_A_BIT,
};
constexpr std::array ASTC_FRAG_DYNAMIC_STATES{VK_DYNAMIC_STATE_VIEWPORT,
VK_DYNAMIC_STATE_SCISSOR};
} // Anonymous namespace
ASTCDecoderFragmentPass::ASTCDecoderFragmentPass(
const Device& device_, Scheduler& scheduler_, DescriptorPool& descriptor_pool_,
ComputePassDescriptorQueue& compute_pass_descriptor_queue_,
RenderPassCache& render_pass_cache_)
: device{device_}, scheduler{scheduler_},
compute_pass_descriptor_queue{compute_pass_descriptor_queue_},
render_pass_cache{render_pass_cache_},
descriptor_set_layout{device.GetLogical().CreateDescriptorSetLayout({
.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.bindingCount = 1,
.pBindings = &ASTC_FRAG_BINDING,
})},
descriptor_template{device.GetLogical().CreateDescriptorUpdateTemplate({
.sType = VK_STRUCTURE_TYPE_DESCRIPTOR_UPDATE_TEMPLATE_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.descriptorUpdateEntryCount = 1,
.pDescriptorUpdateEntries = &ASTC_FRAG_TEMPLATE,
.templateType = VK_DESCRIPTOR_UPDATE_TEMPLATE_TYPE_DESCRIPTOR_SET,
.descriptorSetLayout = *descriptor_set_layout,
.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS,
.pipelineLayout = VK_NULL_HANDLE,
.set = 0,
})},
pipeline_layout{device.GetLogical().CreatePipelineLayout({
.sType = VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.setLayoutCount = 1,
.pSetLayouts = descriptor_set_layout.address(),
.pushConstantRangeCount = 1,
.pPushConstantRanges = &ASTC_FRAG_PUSH_CONSTANT_RANGE,
})},
descriptor_allocator{descriptor_pool_.Allocator(device_, scheduler_, *descriptor_set_layout,
ASTC_FRAG_BANK_INFO)},
vertex_shader{BuildShader(device, ASTC_DECODER_VERT_SPV)},
fragment_shader{BuildShader(device, ASTC_DECODER_FRAG_SPV)} {}
ASTCDecoderFragmentPass::~ASTCDecoderFragmentPass() = default;
VkPipeline ASTCDecoderFragmentPass::FindOrEmplacePipeline(VkRenderPass render_pass) {
const auto it = std::ranges::find(pipeline_keys, render_pass);
if (it != pipeline_keys.end()) {
return *pipelines[std::distance(pipeline_keys.begin(), it)];
}
pipeline_keys.push_back(render_pass);
const std::array stages{
VkPipelineShaderStageCreateInfo{
.sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.stage = VK_SHADER_STAGE_VERTEX_BIT,
.module = *vertex_shader,
.pName = "main",
.pSpecializationInfo = nullptr,
},
VkPipelineShaderStageCreateInfo{
.sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.stage = VK_SHADER_STAGE_FRAGMENT_BIT,
.module = *fragment_shader,
.pName = "main",
.pSpecializationInfo = nullptr,
},
};
const VkPipelineColorBlendStateCreateInfo color_blend{
.sType = VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.logicOpEnable = VK_FALSE,
.logicOp = VK_LOGIC_OP_CLEAR,
.attachmentCount = 1,
.pAttachments = &ASTC_FRAG_BLEND_ATTACHMENT,
.blendConstants = {0.0f, 0.0f, 0.0f, 0.0f},
};
const VkPipelineDynamicStateCreateInfo dynamic_state{
.sType = VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.dynamicStateCount = static_cast<u32>(ASTC_FRAG_DYNAMIC_STATES.size()),
.pDynamicStates = ASTC_FRAG_DYNAMIC_STATES.data(),
};
pipelines.push_back(device.GetLogical().CreateGraphicsPipeline({
.sType = VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.stageCount = static_cast<u32>(stages.size()),
.pStages = stages.data(),
.pVertexInputState = &ASTC_FRAG_VERTEX_INPUT,
.pInputAssemblyState = &ASTC_FRAG_INPUT_ASSEMBLY,
.pTessellationState = nullptr,
.pViewportState = &ASTC_FRAG_VIEWPORT,
.pRasterizationState = &ASTC_FRAG_RASTERIZATION,
.pMultisampleState = &ASTC_FRAG_MULTISAMPLE,
.pDepthStencilState = nullptr,
.pColorBlendState = &color_blend,
.pDynamicState = &dynamic_state,
.layout = *pipeline_layout,
.renderPass = render_pass,
.subpass = 0,
.basePipelineHandle = VK_NULL_HANDLE,
.basePipelineIndex = 0,
}));
return *pipelines.back();
}
void ASTCDecoderFragmentPass::Assemble(Image& image, const StagingBufferRef& map,
std::span<const VideoCommon::SwizzleParameters> swizzles,
VideoCore::Surface::PixelFormat decoded_format) {
using namespace VideoCommon::Accelerated;
while (!frame_resources.empty() && scheduler.IsFree(frame_resources.front().tick)) {
frame_resources.pop_front();
}
const std::array<u32, 2> block_dims{
VideoCore::Surface::DefaultBlockWidth(image.info.format),
VideoCore::Surface::DefaultBlockHeight(image.info.format),
};
RenderPassKey key{};
key.color_formats.fill(VideoCore::Surface::PixelFormat::Invalid);
key.color_formats[0] = decoded_format;
key.depth_format = VideoCore::Surface::PixelFormat::Invalid;
key.samples = VK_SAMPLE_COUNT_1_BIT;
const VkRenderPass render_pass = render_pass_cache.Get(key);
const VkPipeline pipeline = FindOrEmplacePipeline(render_pass);
const VkFormat vk_format =
MaxwellToVK::SurfaceFormat(device, FormatType::Optimal, false, decoded_format).format;
const VkImage vk_image = image.Handle();
const VkImageAspectFlags aspect_mask = image.AspectMask();
scheduler.RequestOutsideRenderPassOperationContext();
const bool is_initialized = image.ExchangeInitialization();
scheduler.Record([vk_image, aspect_mask, is_initialized](vk::CommandBuffer cmdbuf) {
const VkImageMemoryBarrier barrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
.pNext = nullptr,
.srcAccessMask = static_cast<VkAccessFlags>(
is_initialized ? VK_ACCESS_SHADER_READ_BIT : VK_ACCESS_NONE),
.dstAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT,
.oldLayout = is_initialized ? VK_IMAGE_LAYOUT_GENERAL : VK_IMAGE_LAYOUT_UNDEFINED,
.newLayout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = vk_image,
.subresourceRange{
.aspectMask = aspect_mask,
.baseMipLevel = 0,
.levelCount = VK_REMAINING_MIP_LEVELS,
.baseArrayLayer = 0,
.layerCount = VK_REMAINING_ARRAY_LAYERS,
},
};
cmdbuf.PipelineBarrier(vk::PIPELINE_STAGE_GRAPHICS_COMPUTE_TRANSFER,
VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT, 0, barrier);
});
for (const VideoCommon::SwizzleParameters& swizzle : swizzles) {
const size_t input_offset = swizzle.buffer_offset + map.offset;
const auto params = MakeBlockLinearSwizzle2DParams(swizzle, image.info);
const u32 level = swizzle.level;
const u32 width = std::max(1u, image.info.size.width >> level);
const u32 height = std::max(1u, image.info.size.height >> level);
const u32 layers = image.info.resources.layers;
for (u32 layer = 0; layer < layers; ++layer) {
vk::ImageView view = device.GetLogical().CreateImageView(VkImageViewCreateInfo{
.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.image = vk_image,
.viewType = VK_IMAGE_VIEW_TYPE_2D,
.format = vk_format,
.components{},
.subresourceRange{
.aspectMask = aspect_mask,
.baseMipLevel = level,
.levelCount = 1,
.baseArrayLayer = layer,
.layerCount = 1,
},
});
vk::Framebuffer framebuffer =
device.GetLogical().CreateFramebuffer(VkFramebufferCreateInfo{
.sType = VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO,
.pNext = nullptr,
.flags = 0,
.renderPass = render_pass,
.attachmentCount = 1,
.pAttachments = view.address(),
.width = width,
.height = height,
.layers = 1,
});
const AstcFragPushConstants pc{
.blocks_dims = block_dims,
.layer_stride = params.layer_stride,
.block_size = params.block_size,
.x_shift = params.x_shift,
.block_height = params.block_height,
.block_height_mask = params.block_height_mask,
.dest_layer = layer,
};
compute_pass_descriptor_queue.Acquire(scheduler, 1);
compute_pass_descriptor_queue.AddBuffer(map.buffer, input_offset,
image.guest_size_bytes - swizzle.buffer_offset);
const void* const descriptor_data{compute_pass_descriptor_queue.UpdateData()};
scheduler.Record([this, pipeline, render_pass, descriptor_data, pc, width, height,
fb = *framebuffer](vk::CommandBuffer cmdbuf) {
const VkDescriptorSet set = descriptor_allocator.Commit();
device.GetLogical().UpdateDescriptorSet(set, *descriptor_template, descriptor_data);
const VkRenderPassBeginInfo begin{
.sType = VK_STRUCTURE_TYPE_RENDER_PASS_BEGIN_INFO,
.pNext = nullptr,
.renderPass = render_pass,
.framebuffer = fb,
.renderArea{.offset = {0, 0}, .extent = {width, height}},
.clearValueCount = 0,
.pClearValues = nullptr,
};
const VkViewport viewport{
.x = 0.0f,
.y = 0.0f,
.width = static_cast<float>(width),
.height = static_cast<float>(height),
.minDepth = 0.0f,
.maxDepth = 1.0f,
};
const VkRect2D scissor{.offset = {0, 0}, .extent = {width, height}};
cmdbuf.BeginRenderPass(begin, VK_SUBPASS_CONTENTS_INLINE);
cmdbuf.BindPipeline(VK_PIPELINE_BIND_POINT_GRAPHICS, pipeline);
cmdbuf.SetViewport(0, viewport);
cmdbuf.SetScissor(0, scissor);
cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_GRAPHICS, *pipeline_layout, 0, set,
{});
cmdbuf.PushConstants(*pipeline_layout, VK_SHADER_STAGE_FRAGMENT_BIT, pc);
cmdbuf.Draw(3, 1, 0, 0);
cmdbuf.EndRenderPass();
});
frame_resources.push_back(FrameResources{
.tick = scheduler.CurrentTick(),
.view = std::move(view),
.framebuffer = std::move(framebuffer),
});
}
}
scheduler.Record([vk_image, aspect_mask](vk::CommandBuffer cmdbuf) {
const VkImageMemoryBarrier barrier{
.sType = VK_STRUCTURE_TYPE_IMAGE_MEMORY_BARRIER,
.pNext = nullptr,
.srcAccessMask = VK_ACCESS_COLOR_ATTACHMENT_WRITE_BIT,
.dstAccessMask = VK_ACCESS_SHADER_READ_BIT,
.oldLayout = VK_IMAGE_LAYOUT_COLOR_ATTACHMENT_OPTIMAL,
.newLayout = VK_IMAGE_LAYOUT_GENERAL,
.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED,
.image = vk_image,
.subresourceRange{
.aspectMask = aspect_mask,
.baseMipLevel = 0,
.levelCount = VK_REMAINING_MIP_LEVELS,
.baseArrayLayer = 0,
.layerCount = VK_REMAINING_ARRAY_LAYERS,
},
};
cmdbuf.PipelineBarrier(VK_PIPELINE_STAGE_COLOR_ATTACHMENT_OUTPUT_BIT,
vk::PIPELINE_STAGE_GRAPHICS_COMPUTE, 0, barrier);
});
scheduler.InvalidateState();
} }
constexpr u32 BL3D_BINDING_INPUT_BUFFER = 0; constexpr u32 BL3D_BINDING_INPUT_BUFFER = 0;
@@ -6,12 +6,15 @@
#pragma once #pragma once
#include <deque>
#include <optional> #include <optional>
#include <span> #include <span>
#include <utility> #include <utility>
#include <vector>
#include "common/common_types.h" #include "common/common_types.h"
#include "video_core/engines/maxwell_3d.h" #include "video_core/engines/maxwell_3d.h"
#include "video_core/surface.h"
#include "video_core/renderer_vulkan/vk_descriptor_pool.h" #include "video_core/renderer_vulkan/vk_descriptor_pool.h"
#include "video_core/renderer_vulkan/vk_update_descriptor.h" #include "video_core/renderer_vulkan/vk_update_descriptor.h"
#include "video_core/texture_cache/types.h" #include "video_core/texture_cache/types.h"
@@ -137,6 +140,44 @@ private:
MemoryAllocator& memory_allocator; MemoryAllocator& memory_allocator;
}; };
class RenderPassCache;
class ASTCDecoderFragmentPass final {
public:
explicit ASTCDecoderFragmentPass(const Device& device_, Scheduler& scheduler_,
DescriptorPool& descriptor_pool_,
ComputePassDescriptorQueue& compute_pass_descriptor_queue_,
RenderPassCache& render_pass_cache_);
~ASTCDecoderFragmentPass();
void Assemble(Image& image, const StagingBufferRef& map,
std::span<const VideoCommon::SwizzleParameters> swizzles,
VideoCore::Surface::PixelFormat decoded_format);
private:
VkPipeline FindOrEmplacePipeline(VkRenderPass render_pass);
const Device& device;
Scheduler& scheduler;
ComputePassDescriptorQueue& compute_pass_descriptor_queue;
RenderPassCache& render_pass_cache;
vk::DescriptorSetLayout descriptor_set_layout;
vk::DescriptorUpdateTemplate descriptor_template;
vk::PipelineLayout pipeline_layout;
DescriptorAllocator descriptor_allocator;
vk::ShaderModule vertex_shader;
vk::ShaderModule fragment_shader;
std::vector<VkRenderPass> pipeline_keys;
std::vector<vk::Pipeline> pipelines;
struct FrameResources {
u64 tick;
vk::ImageView view;
vk::Framebuffer framebuffer;
};
std::deque<FrameResources> frame_resources;
};
class BlockLinearUnswizzle3DPass final : public ComputePass { class BlockLinearUnswizzle3DPass final : public ComputePass {
public: public:
explicit BlockLinearUnswizzle3DPass(const Device& device_, Scheduler& scheduler_, explicit BlockLinearUnswizzle3DPass(const Device& device_, Scheduler& scheduler_,
@@ -167,6 +167,7 @@ Shader::RuntimeInfo MakeRuntimeInfo(std::span<const Shader::IR::Program> program
} }
const Shader::Stage stage{program.stage}; const Shader::Stage stage{program.stage};
const bool has_geometry{key.unique_hashes[4] != 0 && !programs[4].is_geometry_passthrough}; const bool has_geometry{key.unique_hashes[4] != 0 && !programs[4].is_geometry_passthrough};
const bool has_tessellation{key.unique_hashes[3] != 0};
const bool gl_ndc{key.state.ndc_minus_one_to_one != 0}; const bool gl_ndc{key.state.ndc_minus_one_to_one != 0};
const float point_size{std::bit_cast<float>(key.state.point_size)}; const float point_size{std::bit_cast<float>(key.state.point_size)};
switch (stage) { switch (stage) {
@@ -185,7 +186,9 @@ Shader::RuntimeInfo MakeRuntimeInfo(std::span<const Shader::IR::Program> program
LOG_WARNING(Render_Vulkan, "XFB requested in pipeline key but device lacks VK_EXT_transform_feedback; ignoring XFB decorations"); LOG_WARNING(Render_Vulkan, "XFB requested in pipeline key but device lacks VK_EXT_transform_feedback; ignoring XFB decorations");
} }
} }
info.convert_depth_mode = gl_ndc; if (!has_tessellation) {
info.convert_depth_mode = gl_ndc;
}
} }
if (key.state.dynamic_vertex_input) { if (key.state.dynamic_vertex_input) {
for (size_t index = 0; index < Maxwell::NumVertexAttributes; ++index) { for (size_t index = 0; index < Maxwell::NumVertexAttributes; ++index) {
@@ -224,6 +227,9 @@ Shader::RuntimeInfo MakeRuntimeInfo(std::span<const Shader::IR::Program> program
ASSERT(false); ASSERT(false);
return Shader::TessSpacing::Equal; return Shader::TessSpacing::Equal;
}(); }();
if (!has_geometry) {
info.convert_depth_mode = gl_ndc;
}
break; break;
case Shader::Stage::Geometry: case Shader::Stage::Geometry:
if (program.output_topology == Shader::OutputTopology::PointList) { if (program.output_topology == Shader::OutputTopology::PointList) {
@@ -305,12 +311,8 @@ size_t GetTotalPipelineWorkers() {
std::max<size_t>(static_cast<size_t>(std::thread::hardware_concurrency()), 2ULL) - 1ULL; std::max<size_t>(static_cast<size_t>(std::thread::hardware_concurrency()), 2ULL) - 1ULL;
#ifdef __ANDROID__ #ifdef __ANDROID__
const int configured = AndroidSettings::values.pipeline_worker_count.GetValue(); const int configured = AndroidSettings::values.pipeline_worker_count.GetValue();
const int clamped = std::clamp(configured, 4, 8); const size_t desired = static_cast<size_t>(std::max(configured, 1));
const size_t desired = static_cast<size_t>(clamped); return std::min<size_t>(max_core_threads, desired);
if (desired == 0) {
return 1ULL;
}
return std::min(max_core_threads, desired);
#else #else
return max_core_threads; return max_core_threads;
#endif #endif
@@ -241,11 +241,13 @@ void RasterizerVulkan::PrepareDraw(bool is_indexed, Func&& draw_func) {
if (!pipeline) { if (!pipeline) {
return; return;
} }
std::scoped_lock lock{buffer_cache.mutex, texture_cache.mutex}; {
// update engine as channel may be different. std::scoped_lock lock{buffer_cache.mutex, texture_cache.mutex};
pipeline->SetEngine(maxwell3d, gpu_memory); pipeline->SetEngine(maxwell3d, gpu_memory);
if (!pipeline->Configure(is_indexed)) if (!pipeline->Configure(is_indexed)) {
return; return;
}
}
UpdateDynamicStates(); UpdateDynamicStates();
@@ -673,7 +675,6 @@ void RasterizerVulkan::FlushRegion(DAddr addr, u64 size, VideoCommon::CacheType
texture_cache.DownloadMemory(addr, size); texture_cache.DownloadMemory(addr, size);
} }
if ((True(which & VideoCommon::CacheType::BufferCache))) { if ((True(which & VideoCommon::CacheType::BufferCache))) {
std::scoped_lock lock{buffer_cache.mutex};
buffer_cache.DownloadMemory(addr, size); buffer_cache.DownloadMemory(addr, size);
} }
if ((True(which & VideoCommon::CacheType::QueryCache))) { if ((True(which & VideoCommon::CacheType::QueryCache))) {
@@ -771,16 +772,22 @@ bool RasterizerVulkan::OnCPUWrite(DAddr addr, u64 size) {
return false; return false;
} }
static constexpr bool ENABLE_TEXTURE_CACHE_INVALIDATION_SKIP = true;
static constexpr bool ENABLE_FINE_GRAINED_TRACKER_LOCK = true;
void RasterizerVulkan::OnCacheInvalidation(DAddr addr, u64 size) { void RasterizerVulkan::OnCacheInvalidation(DAddr addr, u64 size) {
if (addr == 0 || size == 0) { if (addr == 0 || size == 0) {
return; return;
} }
{ if (!ENABLE_TEXTURE_CACHE_INVALIDATION_SKIP ||
device_memory.IsRegionTextureCached(addr, size)) {
std::scoped_lock lock{texture_cache.mutex}; std::scoped_lock lock{texture_cache.mutex};
texture_cache.WriteMemory(addr, size); texture_cache.WriteMemory(addr, size);
} }
{ if (ENABLE_FINE_GRAINED_TRACKER_LOCK) {
buffer_cache.CpuWriteInvalidate(addr, size);
} else {
std::scoped_lock lock{buffer_cache.mutex}; std::scoped_lock lock{buffer_cache.mutex};
buffer_cache.WriteMemory(addr, size); buffer_cache.WriteMemory(addr, size);
} }
@@ -129,6 +129,9 @@ constexpr VkBorderColor ConvertBorderColor(const std::array<float, 4>& color) {
if (info.storage) { if (info.storage) {
usage |= VK_IMAGE_USAGE_STORAGE_BIT; usage |= VK_IMAGE_USAGE_STORAGE_BIT;
} }
if (IsPixelFormatASTC(format)) {
usage |= VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT;
}
return usage; return usage;
} }
@@ -911,6 +914,10 @@ TextureCacheRuntime::TextureCacheRuntime(const Device& device_, Scheduler& sched
if (Settings::values.accelerate_astc.GetValue() == Settings::AstcDecodeMode::Gpu) { if (Settings::values.accelerate_astc.GetValue() == Settings::AstcDecodeMode::Gpu) {
astc_decoder_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool, astc_decoder_pass.emplace(device, scheduler, descriptor_pool, staging_buffer_pool,
compute_pass_descriptor_queue, memory_allocator); compute_pass_descriptor_queue, memory_allocator);
if (device.IsTiler()) {
astc_decoder_fragment_pass.emplace(device, scheduler, descriptor_pool,
compute_pass_descriptor_queue, render_pass_cache);
}
} }
if (!device.IsKhrImageFormatListSupported()) { if (!device.IsKhrImageFormatListSupported()) {
return; return;
@@ -2850,6 +2857,13 @@ void TextureCacheRuntime::AccelerateImageUpload(
u32 z_start, u32 z_count) { u32 z_start, u32 z_count) {
if (IsPixelFormatASTC(image.info.format)) { if (IsPixelFormatASTC(image.info.format)) {
if (astc_decoder_fragment_pass) {
const VideoCore::Surface::PixelFormat decoded_format =
WillUseWidenedAstcFormat(device, image.info)
? VideoCore::Surface::PixelFormat::R32G32B32A32_FLOAT
: VideoCore::Surface::PixelFormat::A8B8G8R8_UNORM;
return astc_decoder_fragment_pass->Assemble(image, map, swizzles, decoded_format);
}
return astc_decoder_pass->Assemble(image, map, swizzles); return astc_decoder_pass->Assemble(image, map, swizzles);
} }
@@ -93,7 +93,7 @@ public:
void AccelerateImageUpload(Image&, const StagingBufferRef&, void AccelerateImageUpload(Image&, const StagingBufferRef&,
std::span<const VideoCommon::SwizzleParameters>, std::span<const VideoCommon::SwizzleParameters>,
u32 z_start, u32 z_count); u32 z_start = 0, u32 z_count = 0);
void InsertUploadMemoryBarrier() {} void InsertUploadMemoryBarrier() {}
@@ -147,6 +147,7 @@ public:
BlitImageHelper& blit_image_helper; BlitImageHelper& blit_image_helper;
RenderPassCache& render_pass_cache; RenderPassCache& render_pass_cache;
std::optional<ASTCDecoderPass> astc_decoder_pass; std::optional<ASTCDecoderPass> astc_decoder_pass;
std::optional<ASTCDecoderFragmentPass> astc_decoder_fragment_pass;
std::optional<BlockLinearUnswizzle3DPass> bl3d_unswizzle_pass; std::optional<BlockLinearUnswizzle3DPass> bl3d_unswizzle_pass;
const Settings::ResolutionScalingInfo& resolution; const Settings::ResolutionScalingInfo& resolution;
@@ -2308,6 +2308,7 @@ void TextureCache<P>::TrackImage(ImageBase& image, ImageId image_id) {
if (False(image.flags & ImageFlagBits::Sparse)) { if (False(image.flags & ImageFlagBits::Sparse)) {
if (image.cpu_addr < ~(1ULL << 40)) { if (image.cpu_addr < ~(1ULL << 40)) {
device_memory.UpdatePagesCachedCount(image.cpu_addr, image.guest_size_bytes, 1); device_memory.UpdatePagesCachedCount(image.cpu_addr, image.guest_size_bytes, 1);
device_memory.UpdateTexturePagesCount(image.cpu_addr, image.guest_size_bytes, 1);
} }
return; return;
} }
@@ -2320,12 +2321,14 @@ void TextureCache<P>::TrackImage(ImageBase& image, ImageId image_id) {
const DAddr cpu_addr = map.cpu_addr; const DAddr cpu_addr = map.cpu_addr;
const std::size_t size = map.size; const std::size_t size = map.size;
device_memory.UpdatePagesCachedCount(cpu_addr, size, 1); device_memory.UpdatePagesCachedCount(cpu_addr, size, 1);
device_memory.UpdateTexturePagesCount(cpu_addr, size, 1);
} }
return; return;
} }
ForEachSparseSegment(image, ForEachSparseSegment(image,
[this]([[maybe_unused]] GPUVAddr gpu_addr, DAddr cpu_addr, size_t size) { [this]([[maybe_unused]] GPUVAddr gpu_addr, DAddr cpu_addr, size_t size) {
device_memory.UpdatePagesCachedCount(cpu_addr, size, 1); device_memory.UpdatePagesCachedCount(cpu_addr, size, 1);
device_memory.UpdateTexturePagesCount(cpu_addr, size, 1);
}); });
} }
@@ -2336,6 +2339,7 @@ void TextureCache<P>::UntrackImage(ImageBase& image, ImageId image_id) {
if (False(image.flags & ImageFlagBits::Sparse)) { if (False(image.flags & ImageFlagBits::Sparse)) {
if (image.cpu_addr < ~(1ULL << 40)) { if (image.cpu_addr < ~(1ULL << 40)) {
device_memory.UpdatePagesCachedCount(image.cpu_addr, image.guest_size_bytes, -1); device_memory.UpdatePagesCachedCount(image.cpu_addr, image.guest_size_bytes, -1);
device_memory.UpdateTexturePagesCount(image.cpu_addr, image.guest_size_bytes, -1);
} }
return; return;
} }
@@ -2348,6 +2352,7 @@ void TextureCache<P>::UntrackImage(ImageBase& image, ImageId image_id) {
const DAddr cpu_addr = map.cpu_addr; const DAddr cpu_addr = map.cpu_addr;
const std::size_t size = map.size; const std::size_t size = map.size;
device_memory.UpdatePagesCachedCount(cpu_addr, size, -1); device_memory.UpdatePagesCachedCount(cpu_addr, size, -1);
device_memory.UpdateTexturePagesCount(cpu_addr, size, -1);
} }
} }
@@ -257,7 +257,7 @@ public:
/// Prepare an image to be used /// Prepare an image to be used
void PrepareImage(ImageId image_id, bool is_modification, bool invalidate); void PrepareImage(ImageId image_id, bool is_modification, bool invalidate);
std::recursive_mutex mutex; std::mutex mutex;
private: private:
/// Iterate over all page indices in a range /// Iterate over all page indices in a range
+10 -4
View File
@@ -517,16 +517,22 @@ Device::Device(VkInstance instance_, vk::PhysicalDevice physical_, VkSurfaceKHR
VK_EXT_COLOR_WRITE_ENABLE_EXTENSION_NAME); VK_EXT_COLOR_WRITE_ENABLE_EXTENSION_NAME);
LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken shader float controls."); LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken shader float controls.");
RemoveExtension(extensions.shader_float_controls, VK_KHR_SHADER_FLOAT_CONTROLS_EXTENSION_NAME); RemoveExtension(extensions.shader_float_controls, VK_KHR_SHADER_FLOAT_CONTROLS_EXTENSION_NAME);
LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken workgroup memory explicit layout.");
RemoveExtensionFeature(extensions.workgroup_memory_explicit_layout,
features.workgroup_memory_explicit_layout,
VK_KHR_WORKGROUP_MEMORY_EXPLICIT_LAYOUT_EXTENSION_NAME);
LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken conservative rasterization.");
RemoveExtension(extensions.conservative_rasterization,
VK_EXT_CONSERVATIVE_RASTERIZATION_EXTENSION_NAME);
LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken depth clip control.");
RemoveExtensionFeature(extensions.depth_clip_control, features.depth_clip_control,
VK_EXT_DEPTH_CLIP_CONTROL_EXTENSION_NAME);
LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken shader atomic int64."); LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken shader atomic int64.");
RemoveExtensionFeature(extensions.shader_atomic_int64, features.shader_atomic_int64, RemoveExtensionFeature(extensions.shader_atomic_int64, features.shader_atomic_int64,
VK_KHR_SHADER_ATOMIC_INT64_EXTENSION_NAME); VK_KHR_SHADER_ATOMIC_INT64_EXTENSION_NAME);
features.shader_atomic_int64.shaderBufferInt64Atomics = false; features.shader_atomic_int64.shaderBufferInt64Atomics = false;
features.shader_atomic_int64.shaderSharedInt64Atomics = false; features.shader_atomic_int64.shaderSharedInt64Atomics = false;
features.features.shaderInt64 = false; features.features.shaderInt64 = false;
LOG_WARNING(Render_Vulkan, "Qualcomm drivers have broken workgroup memory explicit layout.");
RemoveExtensionFeature(extensions.workgroup_memory_explicit_layout,
features.workgroup_memory_explicit_layout,
VK_KHR_WORKGROUP_MEMORY_EXPLICIT_LAYOUT_EXTENSION_NAME);
#if defined(__ANDROID__) && defined(ARCHITECTURE_arm64) #if defined(__ANDROID__) && defined(ARCHITECTURE_arm64)
// BCn patching only safe on Android 9+ (API 28+). Older versions crash on driver load. // BCn patching only safe on Android 9+ (API 28+). Older versions crash on driver load.
@@ -230,6 +230,7 @@ void Load(VkDevice device, DeviceDispatch& dld) noexcept {
X(vkMapMemory); X(vkMapMemory);
X(vkQueueSubmit); X(vkQueueSubmit);
X(vkQueueSubmit2); X(vkQueueSubmit2);
X(vkResetCommandPool);
X(vkResetFences); X(vkResetFences);
X(vkResetQueryPool); X(vkResetQueryPool);
X(vkSetDebugUtilsObjectNameEXT); X(vkSetDebugUtilsObjectNameEXT);
@@ -346,6 +346,7 @@ struct DeviceDispatch : InstanceDispatch {
PFN_vkMapMemory vkMapMemory{}; PFN_vkMapMemory vkMapMemory{};
PFN_vkQueueSubmit vkQueueSubmit{}; PFN_vkQueueSubmit vkQueueSubmit{};
PFN_vkQueueSubmit2 vkQueueSubmit2{}; PFN_vkQueueSubmit2 vkQueueSubmit2{};
PFN_vkResetCommandPool vkResetCommandPool{};
PFN_vkResetFences vkResetFences{}; PFN_vkResetFences vkResetFences{};
PFN_vkResetQueryPool vkResetQueryPool{}; PFN_vkResetQueryPool vkResetQueryPool{};
PFN_vkSetDebugUtilsObjectNameEXT vkSetDebugUtilsObjectNameEXT{}; PFN_vkSetDebugUtilsObjectNameEXT vkSetDebugUtilsObjectNameEXT{};
@@ -925,6 +926,10 @@ public:
CommandBuffers Allocate(std::size_t num_buffers, CommandBuffers Allocate(std::size_t num_buffers,
VkCommandBufferLevel level = VK_COMMAND_BUFFER_LEVEL_PRIMARY) const; VkCommandBufferLevel level = VK_COMMAND_BUFFER_LEVEL_PRIMARY) const;
void Reset(VkCommandPoolResetFlags flags = 0) const {
Check(dld->vkResetCommandPool(owner, handle, flags));
}
/// Set object name. /// Set object name.
void SetObjectNameEXT(const char* name) const; void SetObjectNameEXT(const char* name) const;
}; };