Port UMA part 1

This commit is contained in:
CamilleLaVey
2026-10-01 00:19:46 -04:00
parent 451b8eb80c
commit 0a88e761eb
5 changed files with 75 additions and 3 deletions
@@ -1876,6 +1876,20 @@ void BufferCache<P>::MappedUploadMemory([[maybe_unused]] Buffer& buffer,
[[maybe_unused]] u64 total_size_bytes, [[maybe_unused]] u64 total_size_bytes,
[[maybe_unused]] std::span<BufferCopy> copies) { [[maybe_unused]] std::span<BufferCopy> copies) {
if constexpr (USE_MEMORY_MAPS) { if constexpr (USE_MEMORY_MAPS) {
if constexpr (requires { runtime.DirectUploadSpan(buffer, copies); }) {
const std::span<u8> direct = runtime.DirectUploadSpan(buffer, copies);
if (!direct.empty()) {
for (const BufferCopy& copy : copies) {
const DAddr device_addr = buffer.CpuAddr() + copy.dst_offset;
if (Settings::values.enable_gpu_buffer_readback.GetValue()) {
DownloadBufferMemory(buffer, device_addr, copy.size);
}
device_memory.ReadBlockUnsafe(device_addr, direct.data() + copy.dst_offset,
copy.size);
}
return;
}
}
auto upload_staging = runtime.UploadStagingBuffer(total_size_bytes); auto upload_staging = runtime.UploadStagingBuffer(total_size_bytes);
const std::span<u8> staging_pointer = upload_staging.mapped_span; const std::span<u8> staging_pointer = upload_staging.mapped_span;
for (BufferCopy& copy : copies) { for (BufferCopy& copy : copies) {
@@ -97,10 +97,14 @@ vk::Buffer CreateBuffer(const Device& device, const MemoryAllocator& memory_allo
.queueFamilyIndexCount = 0, .queueFamilyIndexCount = 0,
.pQueueFamilyIndices = nullptr, .pQueueFamilyIndices = nullptr,
}; };
if (sparse_alignment > 1) { MemoryUsage usage = MemoryUsage::DeviceLocal;
return memory_allocator.CreateBuffer(buffer_ci, MemoryUsage::DeviceLocal, sparse_alignment); if (device.IsUMA()) {
usage = MemoryUsage::Stream;
} }
return memory_allocator.CreateBuffer(buffer_ci, MemoryUsage::DeviceLocal); if (sparse_alignment > 1) {
return memory_allocator.CreateBuffer(buffer_ci, usage, sparse_alignment);
}
return memory_allocator.CreateBuffer(buffer_ci, usage);
} }
} // Anonymous namespace } // Anonymous namespace
@@ -138,6 +142,17 @@ void Buffer::MarkUsage(u64 offset, u64 size) noexcept {
last_usage_tick = scheduler->CurrentTick(); last_usage_tick = scheduler->CurrentTick();
} }
void Buffer::MarkUpload() noexcept {
last_upload_tick = scheduler->CurrentTick();
}
std::span<u8> Buffer::CoherentMapping() noexcept {
if (!buffer.IsHostCoherent()) {
return {};
}
return buffer.Mapped();
}
VkBufferView Buffer::View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format) { VkBufferView Buffer::View(u32 offset, u32 size, VideoCore::Surface::PixelFormat format) {
if (!device) { if (!device) {
// Null buffer supported, return a null descriptor // Null buffer supported, return a null descriptor
@@ -496,6 +511,16 @@ bool BufferCacheRuntime::CanReorderUpload(const Buffer& buffer,
return can_use_upload_cmdbuf; return can_use_upload_cmdbuf;
} }
std::span<u8> BufferCacheRuntime::DirectUploadSpan(
Buffer& buffer, std::span<const VideoCommon::BufferCopy> copies) {
const std::span<u8> mapping = buffer.CoherentMapping();
if (mapping.empty() || !scheduler.IsFree(buffer.LastUploadTick()) ||
!CanReorderUpload(buffer, copies)) {
return {};
}
return mapping;
}
void BufferCacheRuntime::CopyBuffer(VkBuffer dst_buffer, VkBuffer src_buffer, void BufferCacheRuntime::CopyBuffer(VkBuffer dst_buffer, VkBuffer src_buffer,
std::span<const VideoCommon::BufferCopy> copies, bool barrier, std::span<const VideoCommon::BufferCopy> copies, bool barrier,
bool can_reorder_upload) { bool can_reorder_upload) {
@@ -69,6 +69,14 @@ public:
return last_usage_tick; return last_usage_tick;
} }
void MarkUpload() noexcept;
[[nodiscard]] u64 LastUploadTick() const noexcept {
return last_upload_tick;
}
[[nodiscard]] std::span<u8> CoherentMapping() noexcept;
operator VkBuffer() const noexcept { operator VkBuffer() const noexcept {
return *buffer; return *buffer;
} }
@@ -88,6 +96,7 @@ private:
VideoCommon::UsageTracker tracker; VideoCommon::UsageTracker tracker;
VkDeviceAddress device_address{}; VkDeviceAddress device_address{};
u64 last_usage_tick{}; u64 last_usage_tick{};
u64 last_upload_tick{};
bool is_null{}; bool is_null{};
bool sparse_compatible{}; bool sparse_compatible{};
}; };
@@ -140,6 +149,9 @@ public:
bool CanReorderUpload(const Buffer& buffer, std::span<const VideoCommon::BufferCopy> copies); bool CanReorderUpload(const Buffer& buffer, std::span<const VideoCommon::BufferCopy> copies);
[[nodiscard]] std::span<u8> DirectUploadSpan(Buffer& buffer,
std::span<const VideoCommon::BufferCopy> copies);
void FreeDeferredStagingBuffer(StagingBufferRef& ref); void FreeDeferredStagingBuffer(StagingBufferRef& ref);
void PreCopyBarrier(); void PreCopyBarrier();
@@ -148,6 +160,13 @@ public:
std::span<const VideoCommon::BufferCopy> copies, bool barrier, std::span<const VideoCommon::BufferCopy> copies, bool barrier,
bool can_reorder_upload = false); bool can_reorder_upload = false);
void CopyBuffer(Buffer& dst_buffer, VkBuffer src_buffer,
std::span<const VideoCommon::BufferCopy> copies, bool barrier,
bool can_reorder_upload = false) {
dst_buffer.MarkUpload();
CopyBuffer(dst_buffer.Handle(), src_buffer, copies, barrier, can_reorder_upload);
}
void PostCopyBarrier(); void PostCopyBarrier();
void ClearBuffer(VkBuffer dest_buffer, u32 offset, size_t size, u32 value); void ClearBuffer(VkBuffer dest_buffer, u32 offset, size_t size, u32 value);
@@ -487,6 +487,15 @@ Device::Device(VkInstance instance_, vk::PhysicalDevice physical_, VkSurfaceKHR
properties.subgroup_size_control.maxSubgroupSize > GuestWarpSize; properties.subgroup_size_control.maxSubgroupSize > GuestWarpSize;
is_integrated = properties.properties.deviceType == VK_PHYSICAL_DEVICE_TYPE_INTEGRATED_GPU; is_integrated = properties.properties.deviceType == VK_PHYSICAL_DEVICE_TYPE_INTEGRATED_GPU;
const VkPhysicalDeviceMemoryProperties memory_properties =
physical.GetMemoryProperties().memoryProperties;
is_uma = std::all_of(memory_properties.memoryTypes,
memory_properties.memoryTypes + memory_properties.memoryTypeCount,
[](const VkMemoryType& type) {
return (type.propertyFlags & VK_MEMORY_PROPERTY_DEVICE_LOCAL_BIT) ==
0 ||
(type.propertyFlags & VK_MEMORY_PROPERTY_HOST_VISIBLE_BIT) != 0;
});
supports_d24_depth = supports_d24_depth =
IsFormatSupported(VK_FORMAT_D24_UNORM_S8_UINT, IsFormatSupported(VK_FORMAT_D24_UNORM_S8_UINT,
@@ -984,6 +984,10 @@ FN_MAX_LIMIT_LIST
return must_emulate_scaled_formats; return must_emulate_scaled_formats;
} }
bool IsUMA() const {
return is_uma;
}
bool HasNullDescriptor() const { bool HasNullDescriptor() const {
return features.robustness2.nullDescriptor; return features.robustness2.nullDescriptor;
} }
@@ -1198,6 +1202,7 @@ private:
bool is_blit_depth32_stencil8_supported{}; ///< Support for blitting from and to D32S8. bool is_blit_depth32_stencil8_supported{}; ///< Support for blitting from and to D32S8.
bool is_warp_potentially_bigger{}; ///< Host warp size can be bigger than guest. bool is_warp_potentially_bigger{}; ///< Host warp size can be bigger than guest.
bool is_integrated{}; ///< Is GPU an iGPU. bool is_integrated{}; ///< Is GPU an iGPU.
bool is_uma{};
bool has_broken_compute{}; ///< Compute shaders can cause crashes bool has_broken_compute{}; ///< Compute shaders can cause crashes
bool has_broken_cube_compatibility{}; ///< Has broken cube compatibility bit bool has_broken_cube_compatibility{}; ///< Has broken cube compatibility bit
bool has_broken_float16_math{}; bool has_broken_float16_math{};