Compare commits

..

12 Commits

Author SHA1 Message Date
lizzie aabaefaa73 Trigger Build 2026-05-19 09:39:09 +00:00
lizzie 0a5c7a3c17 Trigger Build 2026-05-19 07:54:23 +00:00
lizzie dd96ff15db fs 2026-05-19 09:01:31 +02:00
lizzie 810d857631 e 2026-05-19 09:01:31 +02:00
lizzie 50abdc8db9 fuck fix 2026-05-19 09:01:31 +02:00
lizzie 7abebe19db f 2026-05-19 09:01:31 +02:00
lizzie c3e5cd8fd1 suboptimal cgen for windows 2026-05-19 09:01:31 +02:00
lizzie d84244fa31 oops 2026-05-19 09:01:31 +02:00
lizzie 2691d27908 fix license 2026-05-19 09:01:31 +02:00
lizzie 3cef700513 [dynarmic] use constant resolution instead of clobbering a register when doing spinlocks
Signed-off-by: lizzie <lizzie@eden-emu.dev>
2026-05-19 09:01:31 +02:00
lizzie e875a3196b [core/hle/services/sockets] allow 'valid' range from [16,255] for IPv4 (#3491)
Signed-off-by: lizzie <lizzie@eden-emu.dev>
Reviewed-on: https://git.eden-emu.dev/eden-emu/eden/pulls/3491
Reviewed-by: Maufeat <sahyno1996@gmail.com>
Reviewed-by: CamilleLaVey <camillelavey99@gmail.com>
2026-05-18 23:54:47 +02:00
lizzie 4eb082485d [video_core] fix odr violation in formatter for pixelFormat (#3504)
Signed-off-by: lizzie <lizzie@eden-emu.dev>
Reviewed-on: https://git.eden-emu.dev/eden-emu/eden/pulls/3504
Reviewed-by: crueter <crueter@eden-emu.dev>
Reviewed-by: CamilleLaVey <camillelavey99@gmail.com>
2026-05-18 23:54:07 +02:00
21 changed files with 234 additions and 773 deletions
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
// SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
@@ -7,28 +10,30 @@
namespace Core { namespace Core {
DynarmicExclusiveMonitor::DynarmicExclusiveMonitor(Memory::Memory& memory_, std::size_t core_count_) DynarmicExclusiveMonitor::DynarmicExclusiveMonitor(Memory::Memory& memory_, std::size_t core_count_)
: monitor{core_count_}, memory{memory_} {} : monitor{}
, memory{memory_}
{}
DynarmicExclusiveMonitor::~DynarmicExclusiveMonitor() = default; DynarmicExclusiveMonitor::~DynarmicExclusiveMonitor() = default;
u8 DynarmicExclusiveMonitor::ExclusiveRead8(std::size_t core_index, VAddr addr) { u8 DynarmicExclusiveMonitor::ExclusiveRead8(std::size_t core_index, VAddr addr) {
return monitor.ReadAndMark<u8>(core_index, addr, [&]() -> u8 { return memory.Read8(addr); }); return monitor.ReadAndMark<u8>(core_index, addr, [=]() -> u8 { return memory.Read8(addr); });
} }
u16 DynarmicExclusiveMonitor::ExclusiveRead16(std::size_t core_index, VAddr addr) { u16 DynarmicExclusiveMonitor::ExclusiveRead16(std::size_t core_index, VAddr addr) {
return monitor.ReadAndMark<u16>(core_index, addr, [&]() -> u16 { return memory.Read16(addr); }); return monitor.ReadAndMark<u16>(core_index, addr, [=]() -> u16 { return memory.Read16(addr); });
} }
u32 DynarmicExclusiveMonitor::ExclusiveRead32(std::size_t core_index, VAddr addr) { u32 DynarmicExclusiveMonitor::ExclusiveRead32(std::size_t core_index, VAddr addr) {
return monitor.ReadAndMark<u32>(core_index, addr, [&]() -> u32 { return memory.Read32(addr); }); return monitor.ReadAndMark<u32>(core_index, addr, [=]() -> u32 { return memory.Read32(addr); });
} }
u64 DynarmicExclusiveMonitor::ExclusiveRead64(std::size_t core_index, VAddr addr) { u64 DynarmicExclusiveMonitor::ExclusiveRead64(std::size_t core_index, VAddr addr) {
return monitor.ReadAndMark<u64>(core_index, addr, [&]() -> u64 { return memory.Read64(addr); }); return monitor.ReadAndMark<u64>(core_index, addr, [=]() -> u64 { return memory.Read64(addr); });
} }
u128 DynarmicExclusiveMonitor::ExclusiveRead128(std::size_t core_index, VAddr addr) { u128 DynarmicExclusiveMonitor::ExclusiveRead128(std::size_t core_index, VAddr addr) {
return monitor.ReadAndMark<u128>(core_index, addr, [&]() -> u128 { return monitor.ReadAndMark<u128>(core_index, addr, [=]() -> u128 {
u128 result; u128 result;
result[0] = memory.Read64(addr); result[0] = memory.Read64(addr);
result[1] = memory.Read64(addr + 8); result[1] = memory.Read64(addr + 8);
@@ -41,31 +46,31 @@ void DynarmicExclusiveMonitor::ClearExclusive(std::size_t core_index) {
} }
bool DynarmicExclusiveMonitor::ExclusiveWrite8(std::size_t core_index, VAddr vaddr, u8 value) { bool DynarmicExclusiveMonitor::ExclusiveWrite8(std::size_t core_index, VAddr vaddr, u8 value) {
return monitor.DoExclusiveOperation<u8>(core_index, vaddr, [&](u8 expected) -> bool { return monitor.DoExclusiveOperation<u8>(core_index, vaddr, [=](u8 expected) -> bool {
return memory.WriteExclusive8(vaddr, value, expected); return memory.WriteExclusive8(vaddr, value, expected);
}); });
} }
bool DynarmicExclusiveMonitor::ExclusiveWrite16(std::size_t core_index, VAddr vaddr, u16 value) { bool DynarmicExclusiveMonitor::ExclusiveWrite16(std::size_t core_index, VAddr vaddr, u16 value) {
return monitor.DoExclusiveOperation<u16>(core_index, vaddr, [&](u16 expected) -> bool { return monitor.DoExclusiveOperation<u16>(core_index, vaddr, [=](u16 expected) -> bool {
return memory.WriteExclusive16(vaddr, value, expected); return memory.WriteExclusive16(vaddr, value, expected);
}); });
} }
bool DynarmicExclusiveMonitor::ExclusiveWrite32(std::size_t core_index, VAddr vaddr, u32 value) { bool DynarmicExclusiveMonitor::ExclusiveWrite32(std::size_t core_index, VAddr vaddr, u32 value) {
return monitor.DoExclusiveOperation<u32>(core_index, vaddr, [&](u32 expected) -> bool { return monitor.DoExclusiveOperation<u32>(core_index, vaddr, [=](u32 expected) -> bool {
return memory.WriteExclusive32(vaddr, value, expected); return memory.WriteExclusive32(vaddr, value, expected);
}); });
} }
bool DynarmicExclusiveMonitor::ExclusiveWrite64(std::size_t core_index, VAddr vaddr, u64 value) { bool DynarmicExclusiveMonitor::ExclusiveWrite64(std::size_t core_index, VAddr vaddr, u64 value) {
return monitor.DoExclusiveOperation<u64>(core_index, vaddr, [&](u64 expected) -> bool { return monitor.DoExclusiveOperation<u64>(core_index, vaddr, [=](u64 expected) -> bool {
return memory.WriteExclusive64(vaddr, value, expected); return memory.WriteExclusive64(vaddr, value, expected);
}); });
} }
bool DynarmicExclusiveMonitor::ExclusiveWrite128(std::size_t core_index, VAddr vaddr, u128 value) { bool DynarmicExclusiveMonitor::ExclusiveWrite128(std::size_t core_index, VAddr vaddr, u128 value) {
return monitor.DoExclusiveOperation<u128>(core_index, vaddr, [&](u128 expected) -> bool { return monitor.DoExclusiveOperation<u128>(core_index, vaddr, [=](u128 expected) -> bool {
return memory.WriteExclusive128(vaddr, value, expected); return memory.WriteExclusive128(vaddr, value, expected);
}); });
} }
-79
View File
@@ -14,17 +14,12 @@
#include <mutex> #include <mutex>
#include <vector> #include <vector>
#include <span> #include <span>
#include <utility>
#include "common/common_types.h" #include "common/common_types.h"
#include "common/range_mutex.h" #include "common/range_mutex.h"
#include "common/scratch_buffer.h" #include "common/scratch_buffer.h"
#include "common/virtual_buffer.h" #include "common/virtual_buffer.h"
#if defined(__linux__)
#include <sys/mman.h>
#endif
namespace Core { namespace Core {
constexpr size_t DEVICE_PAGEBITS = 12ULL; constexpr size_t DEVICE_PAGEBITS = 12ULL;
@@ -50,74 +45,6 @@ class DeviceMemoryManager {
using DeviceMethods = typename Traits::DeviceMethods; using DeviceMethods = typename Traits::DeviceMethods;
public: public:
class MirrorMapping {
public:
MirrorMapping() = default;
MirrorMapping(u8* mapped_base_, size_t mapped_size_, size_t data_offset_)
: mapped_base{mapped_base_}, mapped_size{mapped_size_}, data_offset{data_offset_} {}
MirrorMapping(const MirrorMapping&) = delete;
MirrorMapping& operator=(const MirrorMapping&) = delete;
MirrorMapping(MirrorMapping&& other) noexcept {
MoveFrom(other);
}
MirrorMapping& operator=(MirrorMapping&& other) noexcept {
if (this != &other) {
Release();
MoveFrom(other);
}
return *this;
}
~MirrorMapping() {
Release();
}
[[nodiscard]] bool IsValid() const noexcept {
return mapped_base != nullptr;
}
[[nodiscard]] explicit operator bool() const noexcept {
return IsValid();
}
[[nodiscard]] u8* Data() noexcept {
return mapped_base ? mapped_base + data_offset : nullptr;
}
[[nodiscard]] const u8* Data() const noexcept {
return mapped_base ? mapped_base + data_offset : nullptr;
}
[[nodiscard]] size_t Size() const noexcept {
return mapped_size >= data_offset ? mapped_size - data_offset : 0;
}
private:
void MoveFrom(MirrorMapping& other) noexcept {
mapped_base = std::exchange(other.mapped_base, nullptr);
mapped_size = std::exchange(other.mapped_size, 0);
data_offset = std::exchange(other.data_offset, 0);
}
void Release() noexcept {
#if defined(__linux__)
if (mapped_base) {
munmap(mapped_base, mapped_size);
}
#endif
mapped_base = nullptr;
mapped_size = 0;
data_offset = 0;
}
u8* mapped_base{};
size_t mapped_size{};
size_t data_offset{};
};
DeviceMemoryManager(const DeviceMemory& device_memory); DeviceMemoryManager(const DeviceMemory& device_memory);
~DeviceMemoryManager(); ~DeviceMemoryManager();
@@ -191,11 +118,6 @@ public:
void WriteBlock(DAddr address, const void* src_pointer, size_t size); void WriteBlock(DAddr address, const void* src_pointer, size_t size);
void WriteBlockUnsafe(DAddr address, const void* src_pointer, size_t size); void WriteBlockUnsafe(DAddr address, const void* src_pointer, size_t size);
[[nodiscard]] MirrorMapping CreateMirrorMapping(DAddr address, size_t size) const;
[[nodiscard]] u64 GetMappingVersion() const noexcept {
return mapping_version.load(std::memory_order_acquire);
}
Asid RegisterProcess(Memory::Memory* memory); Asid RegisterProcess(Memory::Memory* memory);
void UnregisterProcess(Asid id); void UnregisterProcess(Asid id);
@@ -314,7 +236,6 @@ private:
std::unique_ptr<CachedPages> cached_pages; std::unique_ptr<CachedPages> cached_pages;
Common::RangeMutex counter_guard; Common::RangeMutex counter_guard;
std::mutex mapping_guard; std::mutex mapping_guard;
std::atomic<u64> mapping_version{1};
}; };
-89
View File
@@ -4,10 +4,6 @@
// SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project // SPDX-FileCopyrightText: Copyright 2023 yuzu Emulator Project
// SPDX-License-Identifier: GPL-2.0-or-later // SPDX-License-Identifier: GPL-2.0-or-later
#if defined(__linux__) && !defined(_GNU_SOURCE)
#define _GNU_SOURCE
#endif
#include <atomic> #include <atomic>
#include <limits> #include <limits>
#include <memory> #include <memory>
@@ -15,17 +11,6 @@
#include <algorithm> #include <algorithm>
#include <vector> #include <vector>
#if defined(__linux__)
#include <sys/mman.h>
#ifndef MREMAP_MAYMOVE
#define MREMAP_MAYMOVE 1
#endif
#ifndef MREMAP_FIXED
#define MREMAP_FIXED 2
#endif
extern "C" void* mremap(void* old_address, size_t old_size, size_t new_size, int flags, ...);
#endif
#include "common/address_space.h" #include "common/address_space.h"
#include "common/address_space.inc" #include "common/address_space.inc"
#include "common/alignment.h" #include "common/alignment.h"
@@ -255,7 +240,6 @@ void DeviceMemoryManager<Traits>::Map(DAddr address, VAddr virtual_address, size
impl->multi_dev_address.Register(new_dev, start_id); impl->multi_dev_address.Register(new_dev, start_id);
} }
t_slot = {}; t_slot = {};
mapping_version.fetch_add(1, std::memory_order_release);
if (track) { if (track) {
TrackContinuityImpl(address, virtual_address, size, asid); TrackContinuityImpl(address, virtual_address, size, asid);
} }
@@ -288,7 +272,6 @@ void DeviceMemoryManager<Traits>::Unmap(DAddr address, size_t size) {
} }
} }
t_slot = {}; t_slot = {};
mapping_version.fetch_add(1, std::memory_order_release);
} }
template <typename Traits> template <typename Traits>
void DeviceMemoryManager<Traits>::TrackContinuityImpl(DAddr address, VAddr virtual_address, void DeviceMemoryManager<Traits>::TrackContinuityImpl(DAddr address, VAddr virtual_address,
@@ -332,78 +315,6 @@ const u8* DeviceMemoryManager<Traits>::GetSpan(const DAddr src_addr, const std::
return nullptr; return nullptr;
} }
template <typename Traits>
typename DeviceMemoryManager<Traits>::MirrorMapping DeviceMemoryManager<Traits>::CreateMirrorMapping(
DAddr address, size_t size) const {
#if !defined(__linux__)
return {};
#else
if (size == 0) {
return {};
}
const DAddr aligned_address = Common::AlignDown(address, DAddr{page_size});
const size_t data_offset = static_cast<size_t>(address - aligned_address);
const size_t mapped_size = Common::AlignUp(size + data_offset, page_size);
struct Segment {
const u8* source;
size_t size;
};
std::vector<Segment> segments;
segments.reserve(Common::DivCeil(mapped_size, page_size));
size_t remaining_size = mapped_size;
size_t page_index = aligned_address >> page_bits;
while (remaining_size > 0) {
const size_t next_pages = std::size_t(tracked_entries[page_index].continuity_tracker);
const size_t copy_amount = (std::min)(next_pages << page_bits, remaining_size);
const auto phys_addr = tracked_entries[page_index].compressed_physical_ptr;
if (phys_addr == 0) {
return {};
}
const auto* source =
GetPointerFromRaw<u8>(PAddr(phys_addr - 1U) << Memory::YUZU_PAGEBITS);
if (!segments.empty() && segments.back().source + segments.back().size == source) {
segments.back().size += copy_amount;
} else {
segments.push_back({source, copy_amount});
}
page_index += next_pages;
remaining_size -= copy_amount;
}
void* const mirror_base =
mmap(nullptr, mapped_size, PROT_NONE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0);
if (mirror_base == MAP_FAILED) {
return {};
}
size_t mirror_offset = 0;
for (const auto& segment : segments) {
void* const target = static_cast<u8*>(mirror_base) + mirror_offset;
void* const result = mremap(const_cast<u8*>(segment.source), 0, segment.size,
MREMAP_MAYMOVE | MREMAP_FIXED, target);
if (result == MAP_FAILED) {
munmap(mirror_base, mapped_size);
return {};
}
if (mprotect(result, segment.size, PROT_READ | PROT_WRITE) != 0) {
munmap(mirror_base, mapped_size);
return {};
}
mirror_offset += segment.size;
}
return MirrorMapping{static_cast<u8*>(mirror_base), mapped_size, data_offset};
#endif
}
template <typename Traits> template <typename Traits>
void DeviceMemoryManager<Traits>::InnerGatherDeviceAddresses(Common::ScratchBuffer<u32>& buffer, void DeviceMemoryManager<Traits>::InnerGatherDeviceAddresses(Common::ScratchBuffer<u32>& buffer,
PAddr address) { PAddr address) {
+4 -4
View File
@@ -629,7 +629,7 @@ Errno BSD::BindImpl(s32 fd, std::span<const u8> addr) {
if (!IsFileDescriptorValid(fd)) { if (!IsFileDescriptorValid(fd)) {
return Errno::BADF; return Errno::BADF;
} }
ASSERT(addr.size() == sizeof(SockAddrIn)); ASSERT(addr.size() >= 16);
auto addr_in = GetValue<SockAddrIn>(addr); auto addr_in = GetValue<SockAddrIn>(addr);
return Translate(file_descriptors[fd]->socket->Bind(Translate(addr_in))); return Translate(file_descriptors[fd]->socket->Bind(Translate(addr_in)));
@@ -640,7 +640,7 @@ Errno BSD::ConnectImpl(s32 fd, std::span<const u8> addr) {
return Errno::BADF; return Errno::BADF;
} }
UNIMPLEMENTED_IF(addr.size() != sizeof(SockAddrIn)); ASSERT(addr.size() >= 16);
auto addr_in = GetValue<SockAddrIn>(addr); auto addr_in = GetValue<SockAddrIn>(addr);
const Errno result = Translate(file_descriptors[fd]->socket->Connect(Translate(addr_in))); const Errno result = Translate(file_descriptors[fd]->socket->Connect(Translate(addr_in)));
@@ -874,7 +874,7 @@ std::pair<s32, Errno> BSD::RecvFromImpl(s32 fd, u32 flags, std::vector<u8>& mess
if (ret < 0) { if (ret < 0) {
addr.clear(); addr.clear();
} else { } else {
ASSERT(addr.size() == sizeof(SockAddrIn)); ASSERT(addr.size() >= 16);
const SockAddrIn result = Translate(addr_in); const SockAddrIn result = Translate(addr_in);
PutValue(addr, result); PutValue(addr, result);
} }
@@ -899,7 +899,7 @@ std::pair<s32, Errno> BSD::SendToImpl(s32 fd, u32 flags, std::span<const u8> mes
Network::SockAddrIn addr_in; Network::SockAddrIn addr_in;
Network::SockAddrIn* p_addr_in = nullptr; Network::SockAddrIn* p_addr_in = nullptr;
if (!addr.empty()) { if (!addr.empty()) {
ASSERT(addr.size() == sizeof(SockAddrIn)); ASSERT(addr.size() >= 16);
auto guest_addr_in = GetValue<SockAddrIn>(addr); auto guest_addr_in = GetValue<SockAddrIn>(addr);
addr_in = Translate(guest_addr_in); addr_in = Translate(guest_addr_in);
p_addr_in = &addr_in; p_addr_in = &addr_in;
+1 -1
View File
@@ -238,7 +238,7 @@ static std::vector<u8> SerializeAddrInfo(const std::vector<Network::AddrInfo>& v
Append<u32_be>(data, static_cast<u32>(Translate(addrinfo.family))); // ai_family Append<u32_be>(data, static_cast<u32>(Translate(addrinfo.family))); // ai_family
Append<u32_be>(data, static_cast<u32>(Translate(addrinfo.socket_type))); // ai_socktype Append<u32_be>(data, static_cast<u32>(Translate(addrinfo.socket_type))); // ai_socktype
Append<u32_be>(data, static_cast<u32>(Translate(addrinfo.protocol))); // ai_protocol Append<u32_be>(data, static_cast<u32>(Translate(addrinfo.protocol))); // ai_protocol
Append<u32_be>(data, sizeof(SockAddrIn)); // ai_addrlen Append<u32_be>(data, 16); // ai_addrlen
// ^ *not* sizeof(SerializedSockAddrIn), not that it matters since they're the same size // ^ *not* sizeof(SerializedSockAddrIn), not that it matters since they're the same size
// ai_addr: // ai_addr:
+2 -1
View File
@@ -110,8 +110,9 @@ struct SockAddrIn {
u8 family; u8 family;
u16 portno; u16 portno;
std::array<u8, 4> ip; std::array<u8, 4> ip;
std::array<u8, 8> zeroes; std::array<u8, 248> zeroes;
}; };
static_assert(sizeof(SockAddrIn) == 0x100);
enum class PollEvents : u16 { enum class PollEvents : u16 {
// Using Pascal case because IN is a macro on Windows. // Using Pascal case because IN is a macro on Windows.
@@ -265,13 +265,9 @@ PollEvents Translate(Network::PollEvents flags) {
} }
Network::SockAddrIn Translate(SockAddrIn value) { Network::SockAddrIn Translate(SockAddrIn value) {
if (value.len != 0 && value.len != sizeof(value) && value.len != 6) { // All lengths are valid, from [0 upto 256]
LOG_WARNING(Service, "Unexpected SockAddrIn len={}, expected 0, {}, or 6",
value.len, sizeof(value));
}
return { return {
.family = Translate(static_cast<Domain>(value.family)), .family = Translate(Domain(value.family)),
.ip = value.ip, .ip = value.ip,
.portno = static_cast<u16>(value.portno >> 8 | value.portno << 8), .portno = static_cast<u16>(value.portno >> 8 | value.portno << 8),
}; };
@@ -279,7 +275,7 @@ Network::SockAddrIn Translate(SockAddrIn value) {
SockAddrIn Translate(Network::SockAddrIn value) { SockAddrIn Translate(Network::SockAddrIn value) {
return { return {
.len = sizeof(SockAddrIn), .len = 16,
.family = static_cast<u8>(Translate(value.family)), .family = static_cast<u8>(Translate(value.family)),
.portno = static_cast<u16>(value.portno >> 8 | value.portno << 8), .portno = static_cast<u16>(value.portno >> 8 | value.portno << 8),
.ip = value.ip, .ip = value.ip,
-4
View File
@@ -169,8 +169,6 @@ if ("x86_64" IN_LIST ARCHITECTURE)
backend/x64/emit_x64_vector.cpp backend/x64/emit_x64_vector.cpp
backend/x64/emit_x64_vector_floating_point.cpp backend/x64/emit_x64_vector_floating_point.cpp
backend/x64/emit_x64_vector_saturation.cpp backend/x64/emit_x64_vector_saturation.cpp
backend/x64/exclusive_monitor.cpp
backend/x64/exclusive_monitor_friend.h
backend/x64/host_feature.h backend/x64/host_feature.h
backend/x64/hostloc.h backend/x64/hostloc.h
backend/x64/jitstate_info.h backend/x64/jitstate_info.h
@@ -231,7 +229,6 @@ if ("arm64" IN_LIST ARCHITECTURE)
backend/arm64/emit_arm64_vector_floating_point.cpp backend/arm64/emit_arm64_vector_floating_point.cpp
backend/arm64/emit_arm64_vector_saturation.cpp backend/arm64/emit_arm64_vector_saturation.cpp
backend/arm64/emit_context.h backend/arm64/emit_context.h
backend/arm64/exclusive_monitor.cpp
backend/arm64/fastmem.h backend/arm64/fastmem.h
backend/arm64/fpsr_manager.cpp backend/arm64/fpsr_manager.cpp
backend/arm64/fpsr_manager.h backend/arm64/fpsr_manager.h
@@ -278,7 +275,6 @@ if ("riscv64" IN_LIST ARCHITECTURE)
backend/riscv64/emit_riscv64_vector.cpp backend/riscv64/emit_riscv64_vector.cpp
backend/riscv64/emit_riscv64.cpp backend/riscv64/emit_riscv64.cpp
backend/riscv64/emit_riscv64.h backend/riscv64/emit_riscv64.h
backend/riscv64/exclusive_monitor.cpp
backend/riscv64/reg_alloc.cpp backend/riscv64/reg_alloc.cpp
backend/riscv64/reg_alloc.h backend/riscv64/reg_alloc.h
backend/riscv64/stack_layout.h backend/riscv64/stack_layout.h
@@ -135,12 +135,9 @@ static void* EmitExclusiveWriteCallTrampoline(oaknut::CodeGenerator& code, const
oaknut::Label l_addr, l_this; oaknut::Label l_addr, l_this;
auto fn = [](const A64::UserConfig& conf, A64::VAddr vaddr, T value) -> u32 { auto fn = [](const A64::UserConfig& conf, A64::VAddr vaddr, T value) -> u32 {
return conf.global_monitor->DoExclusiveOperation<T>(conf.processor_id, vaddr, return conf.global_monitor->DoExclusiveOperation<T>(conf.processor_id, vaddr, [&](T expected) -> bool {
[&](T expected) -> bool {
return (conf.callbacks->*callback)(vaddr, value, expected); return (conf.callbacks->*callback)(vaddr, value, expected);
}) }) ? 0 : 1;
? 0
: 1;
}; };
void* target = code.xptr<void*>(); void* target = code.xptr<void*>();
@@ -300,12 +297,9 @@ static void* EmitExclusiveWrite128CallTrampoline(oaknut::CodeGenerator& code, co
oaknut::Label l_addr, l_this; oaknut::Label l_addr, l_this;
auto fn = [](const A64::UserConfig& conf, A64::VAddr vaddr, Vector value) -> u32 { auto fn = [](const A64::UserConfig& conf, A64::VAddr vaddr, Vector value) -> u32 {
return conf.global_monitor->DoExclusiveOperation<Vector>(conf.processor_id, vaddr, return conf.global_monitor->DoExclusiveOperation<Vector>(conf.processor_id, vaddr, [&](Vector expected) -> bool {
[&](Vector expected) -> bool {
return conf.callbacks->MemoryWriteExclusive128(vaddr, value, expected); return conf.callbacks->MemoryWriteExclusive128(vaddr, value, expected);
}) }) ? 0 : 1;
? 0
: 1;
}; };
void* target = code.xptr<void*>(); void* target = code.xptr<void*>();
@@ -1,61 +0,0 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
/* This file is part of the dynarmic project.
* Copyright (c) 2022 MerryMage
* SPDX-License-Identifier: 0BSD
*/
#include "dynarmic/interface/exclusive_monitor.h"
#include <algorithm>
#include "common/assert.h"
namespace Dynarmic {
ExclusiveMonitor::ExclusiveMonitor(std::size_t processor_count)
: exclusive_addresses(processor_count, INVALID_EXCLUSIVE_ADDRESS), exclusive_values(processor_count) {}
size_t ExclusiveMonitor::GetProcessorCount() const {
return exclusive_addresses.size();
}
void ExclusiveMonitor::Lock() {
lock.Lock();
}
void ExclusiveMonitor::Unlock() {
lock.Unlock();
}
bool ExclusiveMonitor::CheckAndClear(std::size_t processor_id, VAddr address) {
const VAddr masked_address = address & RESERVATION_GRANULE_MASK;
Lock();
if (exclusive_addresses[processor_id] != masked_address) {
Unlock();
return false;
}
for (VAddr& other_address : exclusive_addresses) {
if (other_address == masked_address) {
other_address = INVALID_EXCLUSIVE_ADDRESS;
}
}
return true;
}
void ExclusiveMonitor::Clear() {
Lock();
std::fill(exclusive_addresses.begin(), exclusive_addresses.end(), INVALID_EXCLUSIVE_ADDRESS);
Unlock();
}
void ExclusiveMonitor::ClearProcessor(std::size_t processor_id) {
Lock();
exclusive_addresses[processor_id] = INVALID_EXCLUSIVE_ADDRESS;
Unlock();
}
} // namespace Dynarmic
@@ -1,54 +0,0 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
#include "dynarmic/interface/exclusive_monitor.h"
#include <algorithm>
namespace Dynarmic {
ExclusiveMonitor::ExclusiveMonitor(std::size_t processor_count)
: exclusive_addresses(processor_count, INVALID_EXCLUSIVE_ADDRESS), exclusive_values(processor_count) {}
size_t ExclusiveMonitor::GetProcessorCount() const {
return exclusive_addresses.size();
}
void ExclusiveMonitor::Lock() {
lock.Lock();
}
void ExclusiveMonitor::Unlock() {
lock.Unlock();
}
bool ExclusiveMonitor::CheckAndClear(size_t processor_id, VAddr address) {
const VAddr masked_address = address & RESERVATION_GRANULE_MASK;
Lock();
if (exclusive_addresses[processor_id] != masked_address) {
Unlock();
return false;
}
for (VAddr& other_address : exclusive_addresses) {
if (other_address == masked_address) {
other_address = INVALID_EXCLUSIVE_ADDRESS;
}
}
return true;
}
void ExclusiveMonitor::Clear() {
Lock();
std::fill(exclusive_addresses.begin(), exclusive_addresses.end(), INVALID_EXCLUSIVE_ADDRESS);
Unlock();
}
void ExclusiveMonitor::ClearProcessor(size_t processor_id) {
Lock();
exclusive_addresses[processor_id] = INVALID_EXCLUSIVE_ADDRESS;
Unlock();
}
} // namespace Dynarmic
@@ -20,7 +20,7 @@
#include "dynarmic/backend/x64/abi.h" #include "dynarmic/backend/x64/abi.h"
#include "dynarmic/backend/x64/devirtualize.h" #include "dynarmic/backend/x64/devirtualize.h"
#include "dynarmic/backend/x64/emit_x64_memory.h" #include "dynarmic/backend/x64/emit_x64_memory.h"
#include "dynarmic/backend/x64/exclusive_monitor_friend.h" #include "dynarmic/interface/exclusive_monitor.h"
#include "dynarmic/backend/x64/perf_map.h" #include "dynarmic/backend/x64/perf_map.h"
#include "dynarmic/interface/exclusive_monitor.h" #include "dynarmic/interface/exclusive_monitor.h"
@@ -174,67 +174,35 @@ void A32EmitX64::EmitA32ClearExclusive(A32EmitContext&, IR::Inst*) {
} }
void A32EmitX64::EmitA32ExclusiveReadMemory8(A32EmitContext& ctx, IR::Inst* inst) { void A32EmitX64::EmitA32ExclusiveReadMemory8(A32EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveReadMemoryInline<8, &A32::UserCallbacks::MemoryRead8>(ctx, inst); EmitExclusiveReadMemoryInline<8, &A32::UserCallbacks::MemoryRead8>(ctx, inst);
} else {
EmitExclusiveReadMemory<8, &A32::UserCallbacks::MemoryRead8>(ctx, inst);
}
} }
void A32EmitX64::EmitA32ExclusiveReadMemory16(A32EmitContext& ctx, IR::Inst* inst) { void A32EmitX64::EmitA32ExclusiveReadMemory16(A32EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveReadMemoryInline<16, &A32::UserCallbacks::MemoryRead16>(ctx, inst); EmitExclusiveReadMemoryInline<16, &A32::UserCallbacks::MemoryRead16>(ctx, inst);
} else {
EmitExclusiveReadMemory<16, &A32::UserCallbacks::MemoryRead16>(ctx, inst);
}
} }
void A32EmitX64::EmitA32ExclusiveReadMemory32(A32EmitContext& ctx, IR::Inst* inst) { void A32EmitX64::EmitA32ExclusiveReadMemory32(A32EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveReadMemoryInline<32, &A32::UserCallbacks::MemoryRead32>(ctx, inst); EmitExclusiveReadMemoryInline<32, &A32::UserCallbacks::MemoryRead32>(ctx, inst);
} else {
EmitExclusiveReadMemory<32, &A32::UserCallbacks::MemoryRead32>(ctx, inst);
}
} }
void A32EmitX64::EmitA32ExclusiveReadMemory64(A32EmitContext& ctx, IR::Inst* inst) { void A32EmitX64::EmitA32ExclusiveReadMemory64(A32EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveReadMemoryInline<64, &A32::UserCallbacks::MemoryRead64>(ctx, inst); EmitExclusiveReadMemoryInline<64, &A32::UserCallbacks::MemoryRead64>(ctx, inst);
} else {
EmitExclusiveReadMemory<64, &A32::UserCallbacks::MemoryRead64>(ctx, inst);
}
} }
void A32EmitX64::EmitA32ExclusiveWriteMemory8(A32EmitContext& ctx, IR::Inst* inst) { void A32EmitX64::EmitA32ExclusiveWriteMemory8(A32EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveWriteMemoryInline<8, &A32::UserCallbacks::MemoryWriteExclusive8>(ctx, inst); EmitExclusiveWriteMemoryInline<8, &A32::UserCallbacks::MemoryWriteExclusive8>(ctx, inst);
} else {
EmitExclusiveWriteMemory<8, &A32::UserCallbacks::MemoryWriteExclusive8>(ctx, inst);
}
} }
void A32EmitX64::EmitA32ExclusiveWriteMemory16(A32EmitContext& ctx, IR::Inst* inst) { void A32EmitX64::EmitA32ExclusiveWriteMemory16(A32EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveWriteMemoryInline<16, &A32::UserCallbacks::MemoryWriteExclusive16>(ctx, inst); EmitExclusiveWriteMemoryInline<16, &A32::UserCallbacks::MemoryWriteExclusive16>(ctx, inst);
} else {
EmitExclusiveWriteMemory<16, &A32::UserCallbacks::MemoryWriteExclusive16>(ctx, inst);
}
} }
void A32EmitX64::EmitA32ExclusiveWriteMemory32(A32EmitContext& ctx, IR::Inst* inst) { void A32EmitX64::EmitA32ExclusiveWriteMemory32(A32EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveWriteMemoryInline<32, &A32::UserCallbacks::MemoryWriteExclusive32>(ctx, inst); EmitExclusiveWriteMemoryInline<32, &A32::UserCallbacks::MemoryWriteExclusive32>(ctx, inst);
} else {
EmitExclusiveWriteMemory<32, &A32::UserCallbacks::MemoryWriteExclusive32>(ctx, inst);
}
} }
void A32EmitX64::EmitA32ExclusiveWriteMemory64(A32EmitContext& ctx, IR::Inst* inst) { void A32EmitX64::EmitA32ExclusiveWriteMemory64(A32EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveWriteMemoryInline<64, &A32::UserCallbacks::MemoryWriteExclusive64>(ctx, inst); EmitExclusiveWriteMemoryInline<64, &A32::UserCallbacks::MemoryWriteExclusive64>(ctx, inst);
} else {
EmitExclusiveWriteMemory<64, &A32::UserCallbacks::MemoryWriteExclusive64>(ctx, inst);
}
} }
void A32EmitX64::EmitCheckMemoryAbort(A32EmitContext& ctx, IR::Inst* inst, Xbyak::Label* end) { void A32EmitX64::EmitCheckMemoryAbort(A32EmitContext& ctx, IR::Inst* inst, Xbyak::Label* end) {
@@ -20,7 +20,7 @@
#include "dynarmic/backend/x64/abi.h" #include "dynarmic/backend/x64/abi.h"
#include "dynarmic/backend/x64/devirtualize.h" #include "dynarmic/backend/x64/devirtualize.h"
#include "dynarmic/backend/x64/emit_x64_memory.h" #include "dynarmic/backend/x64/emit_x64_memory.h"
#include "dynarmic/backend/x64/exclusive_monitor_friend.h" #include "dynarmic/interface/exclusive_monitor.h"
#include "dynarmic/backend/x64/perf_map.h" #include "dynarmic/backend/x64/perf_map.h"
#include "dynarmic/common/spin_lock_x64.h" #include "dynarmic/common/spin_lock_x64.h"
#include "dynarmic/interface/exclusive_monitor.h" #include "dynarmic/interface/exclusive_monitor.h"
@@ -330,83 +330,43 @@ void A64EmitX64::EmitA64ClearExclusive(A64EmitContext&, IR::Inst*) {
} }
void A64EmitX64::EmitA64ExclusiveReadMemory8(A64EmitContext& ctx, IR::Inst* inst) { void A64EmitX64::EmitA64ExclusiveReadMemory8(A64EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveReadMemoryInline<8, &A64::UserCallbacks::MemoryRead8>(ctx, inst); EmitExclusiveReadMemoryInline<8, &A64::UserCallbacks::MemoryRead8>(ctx, inst);
} else {
EmitExclusiveReadMemory<8, &A64::UserCallbacks::MemoryRead8>(ctx, inst);
}
} }
void A64EmitX64::EmitA64ExclusiveReadMemory16(A64EmitContext& ctx, IR::Inst* inst) { void A64EmitX64::EmitA64ExclusiveReadMemory16(A64EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveReadMemoryInline<16, &A64::UserCallbacks::MemoryRead16>(ctx, inst); EmitExclusiveReadMemoryInline<16, &A64::UserCallbacks::MemoryRead16>(ctx, inst);
} else {
EmitExclusiveReadMemory<16, &A64::UserCallbacks::MemoryRead16>(ctx, inst);
}
} }
void A64EmitX64::EmitA64ExclusiveReadMemory32(A64EmitContext& ctx, IR::Inst* inst) { void A64EmitX64::EmitA64ExclusiveReadMemory32(A64EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveReadMemoryInline<32, &A64::UserCallbacks::MemoryRead32>(ctx, inst); EmitExclusiveReadMemoryInline<32, &A64::UserCallbacks::MemoryRead32>(ctx, inst);
} else {
EmitExclusiveReadMemory<32, &A64::UserCallbacks::MemoryRead32>(ctx, inst);
}
} }
void A64EmitX64::EmitA64ExclusiveReadMemory64(A64EmitContext& ctx, IR::Inst* inst) { void A64EmitX64::EmitA64ExclusiveReadMemory64(A64EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveReadMemoryInline<64, &A64::UserCallbacks::MemoryRead64>(ctx, inst); EmitExclusiveReadMemoryInline<64, &A64::UserCallbacks::MemoryRead64>(ctx, inst);
} else {
EmitExclusiveReadMemory<64, &A64::UserCallbacks::MemoryRead64>(ctx, inst);
}
} }
void A64EmitX64::EmitA64ExclusiveReadMemory128(A64EmitContext& ctx, IR::Inst* inst) { void A64EmitX64::EmitA64ExclusiveReadMemory128(A64EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveReadMemoryInline<128, &A64::UserCallbacks::MemoryRead128>(ctx, inst); EmitExclusiveReadMemoryInline<128, &A64::UserCallbacks::MemoryRead128>(ctx, inst);
} else {
EmitExclusiveReadMemory<128, &A64::UserCallbacks::MemoryRead128>(ctx, inst);
}
} }
void A64EmitX64::EmitA64ExclusiveWriteMemory8(A64EmitContext& ctx, IR::Inst* inst) { void A64EmitX64::EmitA64ExclusiveWriteMemory8(A64EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveWriteMemoryInline<8, &A64::UserCallbacks::MemoryWriteExclusive8>(ctx, inst); EmitExclusiveWriteMemoryInline<8, &A64::UserCallbacks::MemoryWriteExclusive8>(ctx, inst);
} else {
EmitExclusiveWriteMemory<8, &A64::UserCallbacks::MemoryWriteExclusive8>(ctx, inst);
}
} }
void A64EmitX64::EmitA64ExclusiveWriteMemory16(A64EmitContext& ctx, IR::Inst* inst) { void A64EmitX64::EmitA64ExclusiveWriteMemory16(A64EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveWriteMemoryInline<16, &A64::UserCallbacks::MemoryWriteExclusive16>(ctx, inst); EmitExclusiveWriteMemoryInline<16, &A64::UserCallbacks::MemoryWriteExclusive16>(ctx, inst);
} else {
EmitExclusiveWriteMemory<16, &A64::UserCallbacks::MemoryWriteExclusive16>(ctx, inst);
}
} }
void A64EmitX64::EmitA64ExclusiveWriteMemory32(A64EmitContext& ctx, IR::Inst* inst) { void A64EmitX64::EmitA64ExclusiveWriteMemory32(A64EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveWriteMemoryInline<32, &A64::UserCallbacks::MemoryWriteExclusive32>(ctx, inst); EmitExclusiveWriteMemoryInline<32, &A64::UserCallbacks::MemoryWriteExclusive32>(ctx, inst);
} else {
EmitExclusiveWriteMemory<32, &A64::UserCallbacks::MemoryWriteExclusive32>(ctx, inst);
}
} }
void A64EmitX64::EmitA64ExclusiveWriteMemory64(A64EmitContext& ctx, IR::Inst* inst) { void A64EmitX64::EmitA64ExclusiveWriteMemory64(A64EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveWriteMemoryInline<64, &A64::UserCallbacks::MemoryWriteExclusive64>(ctx, inst); EmitExclusiveWriteMemoryInline<64, &A64::UserCallbacks::MemoryWriteExclusive64>(ctx, inst);
} else {
EmitExclusiveWriteMemory<64, &A64::UserCallbacks::MemoryWriteExclusive64>(ctx, inst);
}
} }
void A64EmitX64::EmitA64ExclusiveWriteMemory128(A64EmitContext& ctx, IR::Inst* inst) { void A64EmitX64::EmitA64ExclusiveWriteMemory128(A64EmitContext& ctx, IR::Inst* inst) {
if (conf.fastmem_exclusive_access) {
EmitExclusiveWriteMemoryInline<128, &A64::UserCallbacks::MemoryWriteExclusive128>(ctx, inst); EmitExclusiveWriteMemoryInline<128, &A64::UserCallbacks::MemoryWriteExclusive128>(ctx, inst);
} else {
EmitExclusiveWriteMemory<128, &A64::UserCallbacks::MemoryWriteExclusive128>(ctx, inst);
}
} }
void A64EmitX64::EmitCheckMemoryAbort(A64EmitContext&, IR::Inst* inst, Xbyak::Label* end) { void A64EmitX64::EmitCheckMemoryAbort(A64EmitContext&, IR::Inst* inst, Xbyak::Label* end) {
@@ -230,8 +230,7 @@ void AxxEmitX64::EmitExclusiveReadMemory(AxxEmitContext& ctx, IR::Inst* inst) {
if (ordered) { if (ordered) {
code.mfence(); code.mfence();
} }
code.CallLambda( code.CallLambda([](AxxUserConfig& conf, Axx::VAddr vaddr) -> T {
[](AxxUserConfig& conf, Axx::VAddr vaddr) -> T {
return conf.global_monitor->ReadAndMark<T>(conf.processor_id, vaddr, [&]() -> T { return conf.global_monitor->ReadAndMark<T>(conf.processor_id, vaddr, [&]() -> T {
return (conf.callbacks->*callback)(vaddr); return (conf.callbacks->*callback)(vaddr);
}); });
@@ -250,8 +249,7 @@ void AxxEmitX64::EmitExclusiveReadMemory(AxxEmitContext& ctx, IR::Inst* inst) {
if (ordered) { if (ordered) {
code.mfence(); code.mfence();
} }
code.CallLambda( code.CallLambda([](AxxUserConfig& conf, Axx::VAddr vaddr, Vector& ret) {
[](AxxUserConfig& conf, Axx::VAddr vaddr, Vector& ret) {
ret = conf.global_monitor->ReadAndMark<Vector>(conf.processor_id, vaddr, [&]() -> Vector { ret = conf.global_monitor->ReadAndMark<Vector>(conf.processor_id, vaddr, [&]() -> Vector {
return (conf.callbacks->*callback)(vaddr); return (conf.callbacks->*callback)(vaddr);
}); });
@@ -320,11 +318,12 @@ void AxxEmitX64::EmitExclusiveWriteMemory(AxxEmitContext& ctx, IR::Inst* inst) {
template<std::size_t bitsize, auto callback> template<std::size_t bitsize, auto callback>
void AxxEmitX64::EmitExclusiveReadMemoryInline(AxxEmitContext& ctx, IR::Inst* inst) { void AxxEmitX64::EmitExclusiveReadMemoryInline(AxxEmitContext& ctx, IR::Inst* inst) {
ASSERT(conf.global_monitor && conf.fastmem_pointer); ASSERT(conf.global_monitor);
if (!exception_handler.SupportsFastmem()) { if (!conf.fastmem_exclusive_access || !exception_handler.SupportsFastmem()) {
EmitExclusiveReadMemory<bitsize, callback>(ctx, inst); EmitExclusiveReadMemory<bitsize, callback>(ctx, inst);
return; return;
} }
ASSERT(conf.fastmem_pointer);
auto args = ctx.reg_alloc.GetArgumentInfo(inst); auto args = ctx.reg_alloc.GetArgumentInfo(inst);
constexpr bool ordered = true; constexpr bool ordered = true;
@@ -344,10 +343,10 @@ void AxxEmitX64::EmitExclusiveReadMemoryInline(AxxEmitContext& ctx, IR::Inst* in
const auto wrapped_fn = read_fallbacks[std::make_tuple(ordered, bitsize, vaddr.getIdx(), value_idx)]; const auto wrapped_fn = read_fallbacks[std::make_tuple(ordered, bitsize, vaddr.getIdx(), value_idx)];
EmitExclusiveLock(code, conf, tmp, tmp2.cvt32()); EmitExclusiveLock(code, conf, tmp2.cvt32());
code.mov(code.byte[code.ABI_JIT_PTR + offsetof(AxxJitState, exclusive_state)], u8(1)); code.mov(code.byte[code.ABI_JIT_PTR + offsetof(AxxJitState, exclusive_state)], u8(1));
code.mov(tmp, std::bit_cast<u64>(GetExclusiveMonitorAddressPointer(conf.global_monitor, conf.processor_id))); code.mov(tmp, std::bit_cast<u64>(conf.global_monitor->exclusive_addresses.data() + conf.processor_id));
code.mov(qword[tmp], vaddr); code.mov(qword[tmp], vaddr);
const auto fastmem_marker = ShouldFastmem(ctx, inst); const auto fastmem_marker = ShouldFastmem(ctx, inst);
@@ -381,10 +380,10 @@ void AxxEmitX64::EmitExclusiveReadMemoryInline(AxxEmitContext& ctx, IR::Inst* in
code.call(wrapped_fn); code.call(wrapped_fn);
} }
code.mov(tmp, std::bit_cast<u64>(GetExclusiveMonitorValuePointer(conf.global_monitor, conf.processor_id))); code.mov(tmp, std::bit_cast<u64>(conf.global_monitor->exclusive_values.data() + conf.processor_id));
EmitWriteMemoryMov<bitsize>(code, tmp, value_idx, false); EmitWriteMemoryMov<bitsize>(code, tmp, value_idx, false);
EmitExclusiveUnlock(code, conf, tmp, tmp2.cvt32()); EmitExclusiveUnlock(code, conf, tmp2.cvt32());
if constexpr (bitsize == 128) { if constexpr (bitsize == 128) {
ctx.reg_alloc.DefineValue(code, inst, Xbyak::Xmm{value_idx}); ctx.reg_alloc.DefineValue(code, inst, Xbyak::Xmm{value_idx});
@@ -397,11 +396,12 @@ void AxxEmitX64::EmitExclusiveReadMemoryInline(AxxEmitContext& ctx, IR::Inst* in
template<std::size_t bitsize, auto callback> template<std::size_t bitsize, auto callback>
void AxxEmitX64::EmitExclusiveWriteMemoryInline(AxxEmitContext& ctx, IR::Inst* inst) { void AxxEmitX64::EmitExclusiveWriteMemoryInline(AxxEmitContext& ctx, IR::Inst* inst) {
ASSERT(conf.global_monitor && conf.fastmem_pointer); ASSERT(conf.global_monitor);
if (!exception_handler.SupportsFastmem()) { if (!conf.fastmem_exclusive_access || !exception_handler.SupportsFastmem()) {
EmitExclusiveWriteMemory<bitsize, callback>(ctx, inst); EmitExclusiveWriteMemory<bitsize, callback>(ctx, inst);
return; return;
} }
ASSERT(conf.fastmem_pointer);
auto args = ctx.reg_alloc.GetArgumentInfo(inst); auto args = ctx.reg_alloc.GetArgumentInfo(inst);
constexpr bool ordered = true; constexpr bool ordered = true;
@@ -425,7 +425,7 @@ void AxxEmitX64::EmitExclusiveWriteMemoryInline(AxxEmitContext& ctx, IR::Inst* i
const auto wrapped_fn = exclusive_write_fallbacks[std::make_tuple(ordered, bitsize, vaddr.getIdx(), value.getIdx())]; const auto wrapped_fn = exclusive_write_fallbacks[std::make_tuple(ordered, bitsize, vaddr.getIdx(), value.getIdx())];
EmitExclusiveLock(code, conf, tmp, tmp2.cvt32()); EmitExclusiveLock(code, conf, tmp2.cvt32());
SharedLabel end = ctx.GenSharedLabel(); SharedLabel end = ctx.GenSharedLabel();
@@ -433,14 +433,14 @@ void AxxEmitX64::EmitExclusiveWriteMemoryInline(AxxEmitContext& ctx, IR::Inst* i
code.movzx(tmp.cvt32(), code.byte[code.ABI_JIT_PTR + offsetof(AxxJitState, exclusive_state)]); code.movzx(tmp.cvt32(), code.byte[code.ABI_JIT_PTR + offsetof(AxxJitState, exclusive_state)]);
code.test(tmp.cvt8(), tmp.cvt8()); code.test(tmp.cvt8(), tmp.cvt8());
code.je(*end, code.T_NEAR); code.je(*end, code.T_NEAR);
code.mov(tmp, std::bit_cast<u64>(GetExclusiveMonitorAddressPointer(conf.global_monitor, conf.processor_id))); code.mov(tmp, std::bit_cast<u64>(conf.global_monitor->exclusive_addresses.data() + conf.processor_id));
code.cmp(qword[tmp], vaddr); code.cmp(qword[tmp], vaddr);
code.jne(*end, code.T_NEAR); code.jne(*end, code.T_NEAR);
EmitExclusiveTestAndClear(code, conf, vaddr, tmp, rax); EmitExclusiveTestAndClear(code, conf, vaddr, tmp, rax);
code.mov(code.byte[code.ABI_JIT_PTR + offsetof(AxxJitState, exclusive_state)], u8(0)); code.mov(code.byte[code.ABI_JIT_PTR + offsetof(AxxJitState, exclusive_state)], u8(0));
code.mov(tmp, std::bit_cast<u64>(GetExclusiveMonitorValuePointer(conf.global_monitor, conf.processor_id))); code.mov(tmp, std::bit_cast<u64>(conf.global_monitor->exclusive_values.data() + conf.processor_id));
if constexpr (bitsize == 128) { if constexpr (bitsize == 128) {
code.mov(rax, qword[tmp + 0]); code.mov(rax, qword[tmp + 0]);
@@ -519,7 +519,7 @@ void AxxEmitX64::EmitExclusiveWriteMemoryInline(AxxEmitContext& ctx, IR::Inst* i
} }
code.L(*end); code.L(*end);
EmitExclusiveUnlock(code, conf, tmp, eax); EmitExclusiveUnlock(code, conf, eax);
ctx.reg_alloc.DefineValue(code, inst, status); ctx.reg_alloc.DefineValue(code, inst, status);
EmitCheckMemoryAbort(ctx, inst); EmitCheckMemoryAbort(ctx, inst);
} }
@@ -13,7 +13,7 @@
#include "dynarmic/backend/x64/a32_emit_x64.h" #include "dynarmic/backend/x64/a32_emit_x64.h"
#include "dynarmic/backend/x64/a64_emit_x64.h" #include "dynarmic/backend/x64/a64_emit_x64.h"
#include "dynarmic/backend/x64/exclusive_monitor_friend.h" #include "dynarmic/interface/exclusive_monitor.h"
#include "dynarmic/common/spin_lock_x64.h" #include "dynarmic/common/spin_lock_x64.h"
#include "dynarmic/interface/exclusive_monitor.h" #include "dynarmic/interface/exclusive_monitor.h"
#include "dynarmic/ir/acc_type.h" #include "dynarmic/ir/acc_type.h"
@@ -344,45 +344,38 @@ const void* EmitWriteMemoryMov(BlockOfCode& code, const Xbyak::RegExp& addr, int
} }
template<typename UserConfig> template<typename UserConfig>
void EmitExclusiveLock(BlockOfCode& code, const UserConfig& conf, Xbyak::Reg64 pointer, Xbyak::Reg32 tmp) { void EmitExclusiveLock(BlockOfCode& code, const UserConfig& conf, Xbyak::Reg32 tmp) {
if (conf.HasOptimization(OptimizationFlag::Unsafe_IgnoreGlobalMonitor)) { if (!conf.HasOptimization(OptimizationFlag::Unsafe_IgnoreGlobalMonitor)) {
return; u64 const slp = std::bit_cast<u64>(std::addressof(conf.global_monitor->lock.storage));
EmitSpinLockLock(code, dword[slp], tmp, code.HasHostFeature(HostFeature::WAITPKG));
} }
code.mov(pointer, std::bit_cast<u64>(GetExclusiveMonitorLockPointer(conf.global_monitor)));
EmitSpinLockLock(code, pointer, tmp, code.HasHostFeature(HostFeature::WAITPKG));
} }
template<typename UserConfig> template<typename UserConfig>
void EmitExclusiveUnlock(BlockOfCode& code, const UserConfig& conf, Xbyak::Reg64 pointer, Xbyak::Reg32 tmp) { void EmitExclusiveUnlock(BlockOfCode& code, const UserConfig& conf, Xbyak::Reg32 tmp) {
if (conf.HasOptimization(OptimizationFlag::Unsafe_IgnoreGlobalMonitor)) { if (!conf.HasOptimization(OptimizationFlag::Unsafe_IgnoreGlobalMonitor)) {
return; u64 const slp = std::bit_cast<u64>(std::addressof(conf.global_monitor->lock.storage));
EmitSpinLockUnlock(code, dword[slp], tmp);
} }
code.mov(pointer, std::bit_cast<u64>(GetExclusiveMonitorLockPointer(conf.global_monitor)));
EmitSpinLockUnlock(code, pointer, tmp);
} }
template<typename UserConfig> template<typename UserConfig>
void EmitExclusiveTestAndClear(BlockOfCode& code, const UserConfig& conf, Xbyak::Reg64 vaddr, Xbyak::Reg64 pointer, Xbyak::Reg64 tmp) { void EmitExclusiveTestAndClear(BlockOfCode& code, const UserConfig& conf, Xbyak::Reg64 vaddr, Xbyak::Reg64 pointer, Xbyak::Reg64 tmp) {
if (conf.HasOptimization(OptimizationFlag::Unsafe_IgnoreGlobalMonitor)) { if (!conf.HasOptimization(OptimizationFlag::Unsafe_IgnoreGlobalMonitor)) {
return;
}
code.mov(tmp, 0xDEAD'DEAD'DEAD'DEAD); code.mov(tmp, 0xDEAD'DEAD'DEAD'DEAD);
const size_t processor_count = GetExclusiveMonitorProcessorCount(conf.global_monitor); static_assert(ExclusiveMonitor::MAX_NUM_CPU_CORES == 4);
for (size_t processor_index = 0; processor_index < processor_count; processor_index++) { for (size_t i = 0; i < ExclusiveMonitor::MAX_NUM_CPU_CORES; i++) {
if (processor_index == conf.processor_id) { if (i != conf.processor_id) {
continue;
}
Xbyak::Label ok; Xbyak::Label ok;
code.mov(pointer, std::bit_cast<u64>(GetExclusiveMonitorAddressPointer(conf.global_monitor, processor_index))); code.mov(pointer, std::bit_cast<u64>(conf.global_monitor->exclusive_addresses.data() + i));
code.cmp(qword[pointer], vaddr); code.cmp(qword[pointer], vaddr);
code.jne(ok, code.T_NEAR); code.jne(ok, code.T_NEAR);
code.mov(qword[pointer], tmp); code.mov(qword[pointer], tmp);
code.L(ok); code.L(ok);
} }
} }
}
}
inline bool IsOrdered(IR::AccType acctype) { inline bool IsOrdered(IR::AccType acctype) {
return acctype == IR::AccType::ORDERED || acctype == IR::AccType::ORDEREDRW || acctype == IR::AccType::LIMITEDORDERED; return acctype == IR::AccType::ORDERED || acctype == IR::AccType::ORDEREDRW || acctype == IR::AccType::LIMITEDORDERED;
@@ -22,25 +22,33 @@ static const auto default_cg_mode = nullptr; //Allow RWE
namespace Dynarmic { namespace Dynarmic {
void EmitSpinLockLock(Xbyak::CodeGenerator& code, Xbyak::Reg64 ptr, Xbyak::Reg32 tmp, bool waitpkg) { /// @brief Emits a lock path for a given spinlock
/// @arg ptr Operand must be a dword[ptr]
/// @arg waitpkg Whetever or not the "UMWAIT" instruction can be used
void EmitSpinLockLock(Xbyak::CodeGenerator& code, Xbyak::Address ptr, Xbyak::Reg32 tmp, bool waitpkg) {
// TODO: this is because we lack regalloc - so better to be safe :(
// TODO: really involve regalloc when we require a 64 bit disp temporal... aside from the one we got handed of course
if (waitpkg) {
Xbyak::Label start, loop; Xbyak::Label start, loop;
code.jmp(start, code.T_NEAR); code.jmp(start, code.T_NEAR);
code.L(loop); code.L(loop);
code.push(Xbyak::util::eax);
if (waitpkg) { code.push(Xbyak::util::ebx);
// TODO: this is because we lack regalloc - so better to be safe :( code.push(Xbyak::util::edx);
code.push(Xbyak::util::rax);
code.push(Xbyak::util::rbx);
code.push(Xbyak::util::rdx);
// TODO: This clobbers EAX and EDX did we tell the regalloc? // TODO: This clobbers EAX and EDX did we tell the regalloc?
// ARM ptr for address-monitoring // ARM ptr for address-monitoring
// XBYAK BUG: code.umonitor(ptr); see issue #255 // XBYAK BUG: code.umonitor(ptr); see issue #255
// replace once xbyak has been fixed // replace once xbyak has been fixed
if (ptr.isREG()) {
code.db(0xF3); code.db(0xF3);
if (ptr.getIdx() >= 8) code.db(0x41); if (ptr.getIdx() >= 8) code.db(0x41);
code.db(0x0F); code.db(0xAE); code.db(0x0F); code.db(0xAE);
code.db(uint8_t((3 << 6) | ((6 & 7) << 3) | (ptr.getIdx() & 7))); code.db(uint8_t((3 << 6) | ((6 & 7) << 3) | (ptr.getIdx() & 7)));
} else {
code.mov(Xbyak::util::rax, ptr);
code.umonitor(Xbyak::util::rax);
}
// tmp.bit[0] = 0: C0.1 | Slow Wakup | Better Savings // tmp.bit[0] = 0: C0.1 | Slow Wakup | Better Savings
// tmp.bit[0] = 1: C0.2 | Fast Wakup | Lesser Savings // tmp.bit[0] = 1: C0.2 | Fast Wakup | Lesser Savings
@@ -55,22 +63,64 @@ void EmitSpinLockLock(Xbyak::CodeGenerator& code, Xbyak::Reg64 ptr, Xbyak::Reg32
code.umwait(Xbyak::util::ebx); code.umwait(Xbyak::util::ebx);
// CF == 1 if we hit the OS-timeout in IA32_UMWAIT_CONTROL without a write // CF == 1 if we hit the OS-timeout in IA32_UMWAIT_CONTROL without a write
// CF == 0 if we exited the wait for any other reason // CF == 0 if we exited the wait for any other reason
code.pop(Xbyak::util::rdx); code.pop(Xbyak::util::edx);
code.pop(Xbyak::util::rbx); code.pop(Xbyak::util::ebx);
code.pop(Xbyak::util::rax); code.pop(Xbyak::util::eax);
} else {
code.pause();
}
code.L(start); code.L(start);
code.mov(tmp, 1); code.mov(tmp, 1);
/*code.lock();*/ code.xchg(code.dword[ptr], tmp); if (ptr.is64bitDisp()) {
// if tmp is on eax, use ebx, otherwise use eax!
auto const other_tmp = tmp.cvt32() == Xbyak::util::eax
? Xbyak::util::rbx
: Xbyak::util::rax;
code.push(other_tmp);
code.mov(other_tmp, ptr.getDisp());
/*code.lock();*/ code.xchg(code.dword[other_tmp], tmp);
code.pop(other_tmp);
} else {
/*code.lock();*/ code.xchg(ptr, tmp);
}
code.test(tmp, tmp);
code.jnz(loop, code.T_NEAR);
} else {
Xbyak::Label start, loop;
code.jmp(start, code.T_NEAR);
code.L(loop);
code.pause();
code.L(start);
code.mov(tmp, 1);
if (ptr.is64bitDisp()) {
// if tmp is on eax, use ebx, otherwise use eax!
auto const other_tmp = tmp.cvt32() == Xbyak::util::eax
? Xbyak::util::rbx
: Xbyak::util::rax;
code.push(other_tmp);
code.mov(other_tmp, ptr.getDisp());
/*code.lock();*/ code.xchg(code.dword[other_tmp], tmp);
code.pop(other_tmp);
} else {
/*code.lock();*/ code.xchg(ptr, tmp);
}
code.test(tmp, tmp); code.test(tmp, tmp);
code.jnz(loop, code.T_NEAR); code.jnz(loop, code.T_NEAR);
} }
}
void EmitSpinLockUnlock(Xbyak::CodeGenerator& code, Xbyak::Reg64 ptr, Xbyak::Reg32 tmp) { // ptr operand must be a dword[ptr]
void EmitSpinLockUnlock(Xbyak::CodeGenerator& code, Xbyak::Address ptr, Xbyak::Reg32 tmp) {
code.xor_(tmp, tmp); code.xor_(tmp, tmp);
code.xchg(code.dword[ptr], tmp); if (ptr.is64bitDisp()) {
// if tmp is on eax, use ebx, otherwise use eax!
auto const other_tmp = tmp.cvt32() == Xbyak::util::eax
? Xbyak::util::rbx
: Xbyak::util::rax;
code.push(other_tmp);
code.mov(other_tmp, ptr.getDisp());
/*code.lock();*/ code.xchg(code.dword[other_tmp], tmp);
code.pop(other_tmp);
} else {
/*code.lock();*/ code.xchg(ptr, tmp);
}
code.mfence(); code.mfence();
} }
@@ -93,11 +143,12 @@ void SpinLockImpl::Initialize() noexcept {
Xbyak::Reg64 const ABI_PARAM1 = Backend::X64::HostLocToReg64(Backend::X64::ABI_PARAM1); Xbyak::Reg64 const ABI_PARAM1 = Backend::X64::HostLocToReg64(Backend::X64::ABI_PARAM1);
code.align(); code.align();
lock = code.getCurr<void (*)(volatile int*)>(); lock = code.getCurr<void (*)(volatile int*)>();
EmitSpinLockLock(code, ABI_PARAM1, code.eax, false); EmitSpinLockLock(code, code.dword[ABI_PARAM1], code.eax, false);
code.ret(); code.ret();
code.align(); code.align();
unlock = code.getCurr<void (*)(volatile int*)>(); unlock = code.getCurr<void (*)(volatile int*)>();
EmitSpinLockUnlock(code, ABI_PARAM1, code.eax); EmitSpinLockUnlock(code, code.dword[ABI_PARAM1], code.eax);
code.ret(); code.ret();
} }
@@ -12,7 +12,7 @@
namespace Dynarmic { namespace Dynarmic {
void EmitSpinLockLock(Xbyak::CodeGenerator& code, Xbyak::Reg64 ptr, Xbyak::Reg32 tmp, bool waitpkg); void EmitSpinLockLock(Xbyak::CodeGenerator& code, Xbyak::Address ptr, Xbyak::Reg32 tmp, bool waitpkg);
void EmitSpinLockUnlock(Xbyak::CodeGenerator& code, Xbyak::Reg64 ptr, Xbyak::Reg32 tmp); void EmitSpinLockUnlock(Xbyak::CodeGenerator& code, Xbyak::Address ptr, Xbyak::Reg32 tmp);
} // namespace Dynarmic } // namespace Dynarmic
@@ -1,3 +1,6 @@
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
// SPDX-License-Identifier: GPL-3.0-or-later
/* This file is part of the dynarmic project. /* This file is part of the dynarmic project.
* Copyright (c) 2018 MerryMage * Copyright (c) 2018 MerryMage
* SPDX-License-Identifier: 0BSD * SPDX-License-Identifier: 0BSD
@@ -20,68 +23,71 @@ using Vector = std::array<std::uint64_t, 2>;
class ExclusiveMonitor { class ExclusiveMonitor {
public: public:
/// @param processor_count Maximum number of processors using this global explicit ExclusiveMonitor() noexcept {
/// exclusive monitor. Each processor must have a std::fill(exclusive_addresses.begin(), exclusive_addresses.end(), INVALID_EXCLUSIVE_ADDRESS);
/// unique id. }
explicit ExclusiveMonitor(size_t processor_count);
size_t GetProcessorCount() const;
/// Marks a region containing [address, address+size) to be exclusive to /// Marks a region containing [address, address+size) to be exclusive to
/// processor processor_id. /// processor index.
template<typename T, typename Function> template<typename T, typename F>
T ReadAndMark(size_t processor_id, VAddr address, Function op) { [[nodiscard]] inline T ReadAndMark(std::size_t index, VAddr address, F f) {
static_assert(std::is_trivially_copyable_v<T>); static_assert(std::is_trivially_copyable_v<T>);
const VAddr masked_address = address & RESERVATION_GRANULE_MASK; const VAddr masked_address = address & RESERVATION_GRANULE_MASK;
lock.Lock();
Lock(); exclusive_addresses[index] = masked_address;
exclusive_addresses[processor_id] = masked_address; T const value = f();
const T value = op(); std::memcpy(exclusive_values[index].data(), std::addressof(value), sizeof(T));
std::memcpy(exclusive_values[processor_id].data(), &value, sizeof(T)); lock.Unlock();
Unlock();
return value; return value;
} }
/// Checks to see if processor processor_id has exclusive access to the [[nodiscard]] inline bool CheckAndClear(std::size_t index, VAddr address) {
const VAddr masked_address = address & RESERVATION_GRANULE_MASK;
if (exclusive_addresses[index] != masked_address)
return false;
for (VAddr& other_address : exclusive_addresses)
if (other_address == masked_address)
other_address = INVALID_EXCLUSIVE_ADDRESS;
return true;
}
/// Checks to see if processor index has exclusive access to the
/// specified region. If it does, executes the operation then clears /// specified region. If it does, executes the operation then clears
/// the exclusive state for processors if their exclusive region(s) /// the exclusive state for processors if their exclusive region(s)
/// contain [address, address+size). /// contain [address, address+size).
template<typename T, typename Function> template<typename T, typename F>
bool DoExclusiveOperation(size_t processor_id, VAddr address, Function op) { [[nodiscard]] inline bool DoExclusiveOperation(std::size_t index, VAddr address, F&& f) {
static_assert(std::is_trivially_copyable_v<T>); static_assert(std::is_trivially_copyable_v<T>);
if (!CheckAndClear(processor_id, address)) { bool result = false;
return false; lock.Lock();
if (CheckAndClear(index, address)) {
T saved_value{};
std::memcpy(std::addressof(saved_value), exclusive_values[index].data(), sizeof(T));
result = f(saved_value);
} }
lock.Unlock();
T saved_value;
std::memcpy(&saved_value, exclusive_values[processor_id].data(), sizeof(T));
const bool result = op(saved_value);
Unlock();
return result; return result;
} }
/// Unmark everything. /// Unmark everything.
void Clear(); inline void Clear() {
lock.Lock();
std::fill(exclusive_addresses.begin(), exclusive_addresses.end(), INVALID_EXCLUSIVE_ADDRESS);
lock.Unlock();
}
/// Unmark processor id /// Unmark processor id
void ClearProcessor(size_t processor_id); inline void ClearProcessor(size_t index) {
lock.Lock();
private: exclusive_addresses[index] = INVALID_EXCLUSIVE_ADDRESS;
bool CheckAndClear(size_t processor_id, VAddr address); lock.Unlock();
}
void Lock();
void Unlock();
friend volatile int* GetExclusiveMonitorLockPointer(ExclusiveMonitor*);
friend size_t GetExclusiveMonitorProcessorCount(ExclusiveMonitor*);
friend VAddr* GetExclusiveMonitorAddressPointer(ExclusiveMonitor*, size_t index);
friend Vector* GetExclusiveMonitorValuePointer(ExclusiveMonitor*, size_t index);
static constexpr VAddr RESERVATION_GRANULE_MASK = 0xFFFF'FFFF'FFFF'FFFFull; static constexpr VAddr RESERVATION_GRANULE_MASK = 0xFFFF'FFFF'FFFF'FFFFull;
static constexpr VAddr INVALID_EXCLUSIVE_ADDRESS = 0xDEAD'DEAD'DEAD'DEADull; static constexpr VAddr INVALID_EXCLUSIVE_ADDRESS = 0xDEAD'DEAD'DEAD'DEADull;
static constexpr size_t MAX_NUM_CPU_CORES = 4; // Sync with src/core/hardware_properties static constexpr size_t MAX_NUM_CPU_CORES = 4; // Sync with src/core/hardware_properties
boost::container::static_vector<VAddr, MAX_NUM_CPU_CORES> exclusive_addresses; std::array<VAddr, MAX_NUM_CPU_CORES> exclusive_addresses;
boost::container::static_vector<Vector, MAX_NUM_CPU_CORES> exclusive_values; std::array<Vector, MAX_NUM_CPU_CORES> exclusive_values;
SpinLock lock; SpinLock lock;
}; };
+2 -215
View File
@@ -104,51 +104,6 @@ void BufferCache<P>::TickFrame() {
RunGarbageCollector(); RunGarbageCollector();
} }
++frame_tick; ++frame_tick;
static constexpr u64 mirror_stats_log_interval = 300;
if ((frame_tick % mirror_stats_log_interval) == 0) {
const u64 upload_hit_copies = mirror_upload_hit_copies - mirror_upload_hit_copies_last;
const u64 upload_miss_copies = mirror_upload_miss_copies - mirror_upload_miss_copies_last;
const u64 upload_hit_bytes = mirror_upload_hit_bytes - mirror_upload_hit_bytes_last;
const u64 upload_miss_bytes = mirror_upload_miss_bytes - mirror_upload_miss_bytes_last;
const u64 download_hit_copies =
mirror_download_hit_copies - mirror_download_hit_copies_last;
const u64 download_miss_copies =
mirror_download_miss_copies - mirror_download_miss_copies_last;
const u64 download_hit_bytes = mirror_download_hit_bytes - mirror_download_hit_bytes_last;
const u64 download_miss_bytes =
mirror_download_miss_bytes - mirror_download_miss_bytes_last;
const u64 upload_total_copies = upload_hit_copies + upload_miss_copies;
const u64 download_total_copies = download_hit_copies + download_miss_copies;
if (upload_total_copies > 0 || download_total_copies > 0) {
const double upload_hit_ratio = upload_total_copies > 0
? (100.0 * static_cast<double>(upload_hit_copies) /
static_cast<double>(upload_total_copies))
: 0.0;
const double download_hit_ratio =
download_total_copies > 0
? (100.0 * static_cast<double>(download_hit_copies) /
static_cast<double>(download_total_copies))
: 0.0;
LOG_INFO(HW_GPU,
"Buffer mirror counters (last {} frames): upload hit/miss copies = {}/{}, "
"hit ratio = {:.2f}%, bytes hit/miss = {}/{}, download hit/miss copies = "
"{}/{}, hit ratio = {:.2f}%, bytes hit/miss = {}/{}",
mirror_stats_log_interval, upload_hit_copies, upload_miss_copies,
upload_hit_ratio, upload_hit_bytes, upload_miss_bytes, download_hit_copies,
download_miss_copies, download_hit_ratio, download_hit_bytes,
download_miss_bytes);
}
mirror_upload_hit_copies_last = mirror_upload_hit_copies;
mirror_upload_miss_copies_last = mirror_upload_miss_copies;
mirror_upload_hit_bytes_last = mirror_upload_hit_bytes;
mirror_upload_miss_bytes_last = mirror_upload_miss_bytes;
mirror_download_hit_copies_last = mirror_download_hit_copies;
mirror_download_miss_copies_last = mirror_download_miss_copies;
mirror_download_hit_bytes_last = mirror_download_hit_bytes;
mirror_download_miss_bytes_last = mirror_download_miss_bytes;
}
delayed_destruction_ring.Tick(); delayed_destruction_ring.Tick();
for (auto& buffer : async_buffers_death_ring) { for (auto& buffer : async_buffers_death_ring) {
@@ -1609,21 +1564,6 @@ BufferId BufferCache<P>::CreateBuffer(DAddr device_addr, u32 wanted_size) {
const u32 size = static_cast<u32>(overlap.end - overlap.begin); const u32 size = static_cast<u32>(overlap.end - overlap.begin);
const BufferId new_buffer_id = slot_buffers.insert(runtime, overlap.begin, size); const BufferId new_buffer_id = slot_buffers.insert(runtime, overlap.begin, size);
auto& new_buffer = slot_buffers[new_buffer_id]; auto& new_buffer = slot_buffers[new_buffer_id];
const u64 current_mapping_version = device_memory.GetMappingVersion();
if (mirror_mapping_version != current_mapping_version) {
buffer_mirrors.clear();
mirror_mapping_version = current_mapping_version;
}
buffer_mirrors.erase(new_buffer.CpuAddr());
if (auto mirror =
device_memory.CreateMirrorMapping(new_buffer.CpuAddr(), new_buffer.SizeBytes());
mirror) {
buffer_mirrors.emplace(new_buffer.CpuAddr(), std::move(mirror));
if (!mirror_creation_logged) [[unlikely]] {
LOG_INFO(HW_GPU, "Buffer mirror mapping enabled (first successful mapping)");
mirror_creation_logged = true;
}
}
const size_t size_bytes = new_buffer.SizeBytes(); const size_t size_bytes = new_buffer.SizeBytes();
runtime.ClearBuffer(new_buffer, 0, size_bytes, 0); runtime.ClearBuffer(new_buffer, 0, size_bytes, 0);
new_buffer.MarkUsage(0, size_bytes); new_buffer.MarkUsage(0, size_bytes);
@@ -1717,52 +1657,15 @@ void BufferCache<P>::ImmediateUploadMemory([[maybe_unused]] Buffer& buffer,
[[maybe_unused]] std::span<const BufferCopy> copies) { [[maybe_unused]] std::span<const BufferCopy> copies) {
if constexpr (!USE_MEMORY_MAPS_FOR_UPLOADS) { if constexpr (!USE_MEMORY_MAPS_FOR_UPLOADS) {
std::span<u8> immediate_buffer; std::span<u8> immediate_buffer;
const auto resolve_mirror_pointer = [&]() -> const u8* {
const u64 current_mapping_version = device_memory.GetMappingVersion();
if (mirror_mapping_version != current_mapping_version) {
buffer_mirrors.clear();
mirror_mapping_version = current_mapping_version;
}
auto mirror_it = buffer_mirrors.find(buffer.CpuAddr());
if (mirror_it == buffer_mirrors.end()) {
if (auto mirror =
device_memory.CreateMirrorMapping(buffer.CpuAddr(), buffer.SizeBytes());
mirror) {
auto [it, inserted] =
buffer_mirrors.emplace(buffer.CpuAddr(), std::move(mirror));
mirror_it = it;
if (inserted && !mirror_creation_logged) [[unlikely]] {
LOG_INFO(HW_GPU, "Buffer mirror mapping enabled (first successful mapping)");
mirror_creation_logged = true;
}
}
}
return mirror_it != buffer_mirrors.end() ? mirror_it->second.Data() : nullptr;
};
const u8* const mirror_pointer = resolve_mirror_pointer();
for (const BufferCopy& copy : copies) { for (const BufferCopy& copy : copies) {
std::span<const u8> upload_span; std::span<const u8> upload_span;
const DAddr device_addr = buffer.CpuAddr() + copy.dst_offset; const DAddr device_addr = buffer.CpuAddr() + copy.dst_offset;
if (mirror_pointer != nullptr) { if (IsRangeGranular(device_addr, copy.size)) {
mirror_upload_hit_copies++;
mirror_upload_hit_bytes += copy.size;
if (!mirror_upload_logged) [[unlikely]] {
LOG_INFO(HW_GPU, "Buffer mirror fast path active for upload sync");
mirror_upload_logged = true;
}
upload_span =
std::span(mirror_pointer + static_cast<size_t>(copy.dst_offset), copy.size);
} else if (IsRangeGranular(device_addr, copy.size)) {
mirror_upload_miss_copies++;
mirror_upload_miss_bytes += copy.size;
auto* const ptr = device_memory.GetPointer<u8>(device_addr); auto* const ptr = device_memory.GetPointer<u8>(device_addr);
if (ptr != nullptr) { if (ptr != nullptr) {
upload_span = std::span(ptr, copy.size); upload_span = std::span(ptr, copy.size);
} }
} else { } else {
mirror_upload_miss_copies++;
mirror_upload_miss_bytes += copy.size;
if (immediate_buffer.empty()) { if (immediate_buffer.empty()) {
immediate_buffer = ImmediateBuffer(largest_copy); immediate_buffer = ImmediateBuffer(largest_copy);
} }
@@ -1781,47 +1684,10 @@ void BufferCache<P>::MappedUploadMemory([[maybe_unused]] Buffer& buffer,
if constexpr (USE_MEMORY_MAPS) { if constexpr (USE_MEMORY_MAPS) {
auto upload_staging = runtime.UploadStagingBuffer(total_size_bytes); auto upload_staging = runtime.UploadStagingBuffer(total_size_bytes);
const std::span<u8> staging_pointer = upload_staging.mapped_span; const std::span<u8> staging_pointer = upload_staging.mapped_span;
const auto resolve_mirror_pointer = [&]() -> const u8* {
const u64 current_mapping_version = device_memory.GetMappingVersion();
if (mirror_mapping_version != current_mapping_version) {
buffer_mirrors.clear();
mirror_mapping_version = current_mapping_version;
}
auto mirror_it = buffer_mirrors.find(buffer.CpuAddr());
if (mirror_it == buffer_mirrors.end()) {
if (auto mirror =
device_memory.CreateMirrorMapping(buffer.CpuAddr(), buffer.SizeBytes());
mirror) {
auto [it, inserted] =
buffer_mirrors.emplace(buffer.CpuAddr(), std::move(mirror));
mirror_it = it;
if (inserted && !mirror_creation_logged) [[unlikely]] {
LOG_INFO(HW_GPU, "Buffer mirror mapping enabled (first successful mapping)");
mirror_creation_logged = true;
}
}
}
return mirror_it != buffer_mirrors.end() ? mirror_it->second.Data() : nullptr;
};
const u8* const mirror_pointer = resolve_mirror_pointer();
for (BufferCopy& copy : copies) { for (BufferCopy& copy : copies) {
u8* const src_pointer = staging_pointer.data() + copy.src_offset; u8* const src_pointer = staging_pointer.data() + copy.src_offset;
const DAddr device_addr = buffer.CpuAddr() + copy.dst_offset; const DAddr device_addr = buffer.CpuAddr() + copy.dst_offset;
if (mirror_pointer != nullptr) {
mirror_upload_hit_copies++;
mirror_upload_hit_bytes += copy.size;
if (!mirror_upload_logged) [[unlikely]] {
LOG_INFO(HW_GPU, "Buffer mirror fast path active for upload sync");
mirror_upload_logged = true;
}
std::memcpy(src_pointer, mirror_pointer + static_cast<size_t>(copy.dst_offset),
copy.size);
} else {
mirror_upload_miss_copies++;
mirror_upload_miss_bytes += copy.size;
device_memory.ReadBlockUnsafe(device_addr, src_pointer, copy.size); device_memory.ReadBlockUnsafe(device_addr, src_pointer, copy.size);
}
// Apply the staging offset // Apply the staging offset
copy.src_offset += upload_staging.offset; copy.src_offset += upload_staging.offset;
@@ -1914,30 +1780,6 @@ void BufferCache<P>::DownloadBufferMemory(Buffer& buffer, DAddr device_addr, u64
if constexpr (USE_MEMORY_MAPS) { if constexpr (USE_MEMORY_MAPS) {
auto download_staging = runtime.DownloadStagingBuffer(total_size_bytes); auto download_staging = runtime.DownloadStagingBuffer(total_size_bytes);
const u8* const mapped_memory = download_staging.mapped_span.data(); const u8* const mapped_memory = download_staging.mapped_span.data();
const auto resolve_mirror_pointer = [&]() -> u8* {
const u64 current_mapping_version = device_memory.GetMappingVersion();
if (mirror_mapping_version != current_mapping_version) {
buffer_mirrors.clear();
mirror_mapping_version = current_mapping_version;
}
auto mirror_it = buffer_mirrors.find(buffer.CpuAddr());
if (mirror_it == buffer_mirrors.end()) {
if (auto mirror =
device_memory.CreateMirrorMapping(buffer.CpuAddr(), buffer.SizeBytes());
mirror) {
auto [it, inserted] =
buffer_mirrors.emplace(buffer.CpuAddr(), std::move(mirror));
mirror_it = it;
if (inserted && !mirror_creation_logged) [[unlikely]] {
LOG_INFO(HW_GPU, "Buffer mirror mapping enabled (first successful mapping)");
mirror_creation_logged = true;
}
}
}
return mirror_it != buffer_mirrors.end() ? mirror_it->second.Data() : nullptr;
};
u8* const mirror_pointer = resolve_mirror_pointer();
const std::span<BufferCopy> copies_span(copies.data(), copies.data() + copies.size()); const std::span<BufferCopy> copies_span(copies.data(), copies.data() + copies.size());
for (BufferCopy& copy : copies) { for (BufferCopy& copy : copies) {
// Modify copies to have the staging offset in mind // Modify copies to have the staging offset in mind
@@ -1951,65 +1793,14 @@ void BufferCache<P>::DownloadBufferMemory(Buffer& buffer, DAddr device_addr, u64
// Undo the modified offset // Undo the modified offset
const u64 dst_offset = copy.dst_offset - download_staging.offset; const u64 dst_offset = copy.dst_offset - download_staging.offset;
const u8* copy_mapped_memory = mapped_memory + dst_offset; const u8* copy_mapped_memory = mapped_memory + dst_offset;
if (mirror_pointer != nullptr) {
mirror_download_hit_copies++;
mirror_download_hit_bytes += copy.size;
if (!mirror_download_logged) [[unlikely]] {
LOG_INFO(HW_GPU, "Buffer mirror fast path active for download sync");
mirror_download_logged = true;
}
std::memcpy(mirror_pointer + static_cast<size_t>(copy.src_offset),
copy_mapped_memory, copy.size);
} else {
mirror_download_miss_copies++;
mirror_download_miss_bytes += copy.size;
device_memory.WriteBlockUnsafe(copy_device_addr, copy_mapped_memory, copy.size); device_memory.WriteBlockUnsafe(copy_device_addr, copy_mapped_memory, copy.size);
} }
}
} else { } else {
const std::span<u8> immediate_buffer = ImmediateBuffer(largest_copy); const std::span<u8> immediate_buffer = ImmediateBuffer(largest_copy);
const auto resolve_mirror_pointer = [&]() -> u8* {
const u64 current_mapping_version = device_memory.GetMappingVersion();
if (mirror_mapping_version != current_mapping_version) {
buffer_mirrors.clear();
mirror_mapping_version = current_mapping_version;
}
auto mirror_it = buffer_mirrors.find(buffer.CpuAddr());
if (mirror_it == buffer_mirrors.end()) {
if (auto mirror =
device_memory.CreateMirrorMapping(buffer.CpuAddr(), buffer.SizeBytes());
mirror) {
auto [it, inserted] =
buffer_mirrors.emplace(buffer.CpuAddr(), std::move(mirror));
mirror_it = it;
if (inserted && !mirror_creation_logged) [[unlikely]] {
LOG_INFO(HW_GPU, "Buffer mirror mapping enabled (first successful mapping)");
mirror_creation_logged = true;
}
}
}
return mirror_it != buffer_mirrors.end() ? mirror_it->second.Data() : nullptr;
};
u8* const mirror_pointer = resolve_mirror_pointer();
for (const BufferCopy& copy : copies) { for (const BufferCopy& copy : copies) {
buffer.ImmediateDownload(copy.src_offset, immediate_buffer.subspan(0, copy.size)); buffer.ImmediateDownload(copy.src_offset, immediate_buffer.subspan(0, copy.size));
const DAddr copy_device_addr = buffer.CpuAddr() + copy.src_offset; const DAddr copy_device_addr = buffer.CpuAddr() + copy.src_offset;
if (mirror_pointer != nullptr) { device_memory.WriteBlockUnsafe(copy_device_addr, immediate_buffer.data(), copy.size);
mirror_download_hit_copies++;
mirror_download_hit_bytes += copy.size;
if (!mirror_download_logged) [[unlikely]] {
LOG_INFO(HW_GPU, "Buffer mirror fast path active for download sync");
mirror_download_logged = true;
}
std::memcpy(mirror_pointer + static_cast<size_t>(copy.src_offset),
immediate_buffer.data(), copy.size);
} else {
mirror_download_miss_copies++;
mirror_download_miss_bytes += copy.size;
device_memory.WriteBlockUnsafe(copy_device_addr, immediate_buffer.data(),
copy.size);
}
} }
} }
} }
@@ -2050,10 +1841,6 @@ void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) {
if (!do_not_mark) { if (!do_not_mark) {
Buffer& buffer = slot_buffers[buffer_id]; Buffer& buffer = slot_buffers[buffer_id];
memory_tracker.MarkRegionAsCpuModified(buffer.CpuAddr(), buffer.SizeBytes()); memory_tracker.MarkRegionAsCpuModified(buffer.CpuAddr(), buffer.SizeBytes());
buffer_mirrors.erase(buffer.CpuAddr());
} else {
const Buffer& buffer = slot_buffers[buffer_id];
buffer_mirrors.erase(buffer.CpuAddr());
} }
Unregister(buffer_id); Unregister(buffer_id);
@@ -473,8 +473,6 @@ private:
Tegra::MaxwellDeviceMemoryManager& device_memory; Tegra::MaxwellDeviceMemoryManager& device_memory;
Common::SlotVector<Buffer> slot_buffers; Common::SlotVector<Buffer> slot_buffers;
ankerl::unordered_dense::map<DAddr, Tegra::MaxwellDeviceMemoryManager::MirrorMapping>
buffer_mirrors;
#ifdef YUZU_LEGACY #ifdef YUZU_LEGACY
static constexpr size_t TICKS_TO_DESTROY = 6; static constexpr size_t TICKS_TO_DESTROY = 6;
#else #else
@@ -524,26 +522,6 @@ private:
std::array<BufferId, ((1ULL << 34) >> CACHING_PAGEBITS)> page_table; std::array<BufferId, ((1ULL << 34) >> CACHING_PAGEBITS)> page_table;
Common::ScratchBuffer<u8> tmp_buffer; Common::ScratchBuffer<u8> tmp_buffer;
bool mirror_creation_logged = false;
bool mirror_upload_logged = false;
bool mirror_download_logged = false;
u64 mirror_mapping_version = 0;
u64 mirror_upload_hit_copies = 0;
u64 mirror_upload_miss_copies = 0;
u64 mirror_upload_hit_bytes = 0;
u64 mirror_upload_miss_bytes = 0;
u64 mirror_download_hit_copies = 0;
u64 mirror_download_miss_copies = 0;
u64 mirror_download_hit_bytes = 0;
u64 mirror_download_miss_bytes = 0;
u64 mirror_upload_hit_copies_last = 0;
u64 mirror_upload_miss_copies_last = 0;
u64 mirror_upload_hit_bytes_last = 0;
u64 mirror_upload_miss_bytes_last = 0;
u64 mirror_download_hit_copies_last = 0;
u64 mirror_download_miss_copies_last = 0;
u64 mirror_download_hit_bytes_last = 0;
u64 mirror_download_miss_bytes_last = 0;
}; };
} // namespace VideoCommon } // namespace VideoCommon
+10 -1
View File
@@ -5,6 +5,8 @@
#pragma once #pragma once
#undef PIXEL_FORMAT_LIST
#include <climits> #include <climits>
#include <utility> #include <utility>
#include "common/assert.h" #include "common/assert.h"
@@ -128,7 +130,7 @@ namespace VideoCore::Surface {
PIXEL_FORMAT_ELEM(S8_UINT_D24_UNORM, 1, 1, 32) \ PIXEL_FORMAT_ELEM(S8_UINT_D24_UNORM, 1, 1, 32) \
PIXEL_FORMAT_ELEM(D32_FLOAT_S8_UINT, 1, 1, 64) PIXEL_FORMAT_ELEM(D32_FLOAT_S8_UINT, 1, 1, 64)
enum class PixelFormat { enum class PixelFormat : u32 {
#define PIXEL_FORMAT_ELEM(name, ...) name, #define PIXEL_FORMAT_ELEM(name, ...) name,
PIXEL_FORMAT_LIST PIXEL_FORMAT_LIST
#undef PIXEL_FORMAT_ELEM #undef PIXEL_FORMAT_ELEM
@@ -192,6 +194,13 @@ constexpr u32 BytesPerBlock(PixelFormat pixel_format) {
return BitsPerBlock(pixel_format) / CHAR_BIT; return BitsPerBlock(pixel_format) / CHAR_BIT;
} }
}
#include "video_core/gpu.h"
#include "video_core/textures/texture.h"
namespace VideoCore::Surface {
SurfaceTarget SurfaceTargetFromTextureType(Tegra::Texture::TextureType texture_type); SurfaceTarget SurfaceTargetFromTextureType(Tegra::Texture::TextureType texture_type);
bool SurfaceTargetIsLayered(SurfaceTarget target); bool SurfaceTargetIsLayered(SurfaceTarget target);
bool SurfaceTargetIsArray(SurfaceTarget target); bool SurfaceTargetIsArray(SurfaceTarget target);