Address some regressions with UMA

This commit is contained in:
CamilleLaVey
2026-10-01 20:56:23 -04:00
parent 7db354a25a
commit 5c7520f572
6 changed files with 62 additions and 54 deletions
+12 -7
View File
@@ -211,14 +211,19 @@ DAddr NvMap::PinHandle(NvMap::Handle::Id handle, bool low_area_pin) {
while ((address = smmu.Allocate(aligned_up)) == 0) {
// Free handles until the allocation succeeds
std::scoped_lock queueLock(unmap_queue_lock);
if (auto freeHandleDesc{unmap_queue.front()}) {
// Handles in the unmap queue are guaranteed not to be pinned so don't bother
// checking if they are before unmapping
std::scoped_lock freeLock(freeHandleDesc->mutex);
if (handle_description->d_address)
UnmapHandle(*freeHandleDesc);
} else {
if (unmap_queue.empty()) {
LOG_CRITICAL(Service_NVDRV, "Ran out of SMMU address space!");
return 0;
}
// Handles in the unmap queue are guaranteed not to be pinned so don't bother
// checking if they are before unmapping
const std::shared_ptr<Handle> freeHandleDesc = unmap_queue.front();
std::scoped_lock freeLock(freeHandleDesc->mutex);
if (freeHandleDesc->d_address) {
UnmapHandle(*freeHandleDesc);
} else {
unmap_queue.pop_front();
freeHandleDesc->unmap_queue_entry.reset();
}
}
@@ -233,9 +233,9 @@ NvResult nvhost_as_gpu::FreeSpace(IoctlFreeSpace& params) {
if (allocation.page_size != params.page_size || allocation.size != (u64(params.pages) * params.page_size))
return NvResult::BadValue;
for (const auto mapping_offset : allocation.mappings)
if (!FreeMappingLocked(mapping_offset))
return NvResult::BadValue;
for (const auto mapping_offset : allocation.mappings) {
void(FreeMappingLocked(mapping_offset));
}
// Unset sparse flag if required
if (allocation.sparse)
@@ -375,39 +375,16 @@ NvResult nvhost_as_gpu::MapBufferEx(IoctlMapBufferEx& params) {
mapping_map.insert_or_assign(params.offset, Mapping(params.handle, device_address, params.offset, size, false, big_page, false));
}
map_buffer_offsets.insert(params.offset);
return NvResult::Success;
}
NvResult nvhost_as_gpu::UnmapBuffer(IoctlUnmapBuffer& params) {
LOG_DEBUG(Service_NVDRV, "called, offset={:#x}", params.offset);
std::scoped_lock lock(mutex);
if (auto const offset_it = map_buffer_offsets.find(params.offset); offset_it != map_buffer_offsets.end()) {
LOG_DEBUG(Service_NVDRV, "called, offset={:#x}", params.offset);
if (!vm.initialised) {
return NvResult::BadValue;
}
auto const it = mapping_map.find(params.offset);
auto const mapping = it->second;
if (!mapping.fixed) {
auto& allocator{mapping.big_page ? *vm.big_page_allocator : *vm.small_page_allocator};
u32 page_size_bits{mapping.big_page ? vm.big_page_size_bits : VM::PAGE_SIZE_BITS};
allocator.Free(u32(mapping.offset >> page_size_bits), u32(mapping.size >> page_size_bits));
}
// Sparse mappings shouldn't be fully unmapped, just returned to their sparse state
// Only FreeSpace can unmap them fully
if (mapping.sparse_alloc) {
gmmu->MapSparse(params.offset, mapping.size, mapping.big_page);
} else {
gmmu->Unmap(params.offset, mapping.size);
}
nvmap.UnpinHandle(mapping.handle);
mapping_map.erase(params.offset);
map_buffer_offsets.erase(params.offset);
if (!vm.initialised) {
return NvResult::BadValue;
}
void(FreeMappingLocked(params.offset));
return NvResult::Success;
}
@@ -113,8 +113,6 @@ private:
};
static_assert(sizeof(IoctlRemapEntry) == 20, "IoctlRemapEntry is incorrect size");
::Common::unordered_set<s64_le> map_buffer_offsets{};
struct IoctlMapBufferEx {
MappingFlags flags{}; // bit0: fixed_offset, bit2: cacheable
u32_le kind{}; // -1 is default
+32 -13
View File
@@ -25,22 +25,22 @@ class UsageTracker {
};
public:
explicit UsageTracker(size_t size) : pages((size >> BUFFER_PAGE_SHIFT) + 1) {}
explicit UsageTracker(size_t size)
: pages((size >> BUFFER_PAGE_SHIFT) + 1), full_ticks(pages.size()) {}
void Track(u64 offset, u64 size, u64 tick, u64 gpu_tick) noexcept {
const u64 end = offset + size;
if (size == 0 || ((end - 1) >> BUFFER_PAGE_SHIFT) >= pages.size()) {
return;
}
for (u64 page = offset >> BUFFER_PAGE_SHIFT; page <= (end - 1) >> BUFFER_PAGE_SHIFT;
++page) {
Page& entry = pages[page];
if (entry.tick <= gpu_tick) {
entry.bits = 0;
}
entry.bits |= PageMask(page, offset, end);
entry.tick = (std::max)(entry.tick, tick);
const u64 first_page = offset >> BUFFER_PAGE_SHIFT;
const u64 last_page = (end - 1) >> BUFFER_PAGE_SHIFT;
TrackPage(first_page, PageMask(first_page, offset, end), tick, gpu_tick);
if (last_page == first_page) {
return;
}
std::fill(full_ticks.begin() + first_page + 1, full_ticks.begin() + last_page, tick);
TrackPage(last_page, PageMask(last_page, offset, end), tick, gpu_tick);
}
[[nodiscard]] bool IsUsed(u64 offset, u64 size, u64 gpu_tick) const noexcept {
@@ -48,10 +48,14 @@ public:
if (size == 0 || ((end - 1) >> BUFFER_PAGE_SHIFT) >= pages.size()) {
return false;
}
for (u64 page = offset >> BUFFER_PAGE_SHIFT; page <= (end - 1) >> BUFFER_PAGE_SHIFT;
++page) {
const Page& entry = pages[page];
if (entry.tick > gpu_tick && (entry.bits & PageMask(page, offset, end)) != 0) {
const u64 first_page = offset >> BUFFER_PAGE_SHIFT;
const u64 last_page = (end - 1) >> BUFFER_PAGE_SHIFT;
if (IsPageUsed(first_page, PageMask(first_page, offset, end), gpu_tick) ||
IsPageUsed(last_page, PageMask(last_page, offset, end), gpu_tick)) {
return true;
}
for (u64 page = first_page + 1; page < last_page; ++page) {
if (IsPageUsed(page, ~u64{0}, gpu_tick)) {
return true;
}
}
@@ -59,6 +63,20 @@ public:
}
private:
void TrackPage(u64 page, u64 mask, u64 tick, u64 gpu_tick) noexcept {
Page& entry = pages[page];
if (entry.tick <= gpu_tick) {
entry.bits = 0;
}
entry.bits |= mask;
entry.tick = (std::max)(entry.tick, tick);
}
[[nodiscard]] bool IsPageUsed(u64 page, u64 mask, u64 gpu_tick) const noexcept {
const Page& entry = pages[page];
return full_ticks[page] > gpu_tick || (entry.tick > gpu_tick && (entry.bits & mask) != 0);
}
[[nodiscard]] static u64 PageMask(u64 page, u64 offset, u64 end) noexcept {
const u64 page_begin = page << BUFFER_PAGE_SHIFT;
const u64 first =
@@ -69,6 +87,7 @@ private:
}
std::vector<Page> pages;
std::vector<u64> full_ticks;
};
} // namespace VideoCommon
@@ -147,8 +147,14 @@ bool Buffer::IsRegionUploading(u64 offset, u64 size) const noexcept {
}
void Buffer::MarkUsage(u64 offset, u64 size) noexcept {
tracker.Track(offset, size, scheduler->CurrentTick(),
scheduler->GetMasterSemaphore().KnownGpuTick());
const u64 tick = scheduler->CurrentTick();
if (tick == usage_tick && offset >= usage_begin && offset + size <= usage_end) {
return;
}
tracker.Track(offset, size, tick, scheduler->GetMasterSemaphore().KnownGpuTick());
usage_tick = tick;
usage_begin = offset;
usage_end = offset + size;
}
void Buffer::MarkUpload(std::span<const VideoCommon::BufferCopy> copies) noexcept {
@@ -88,6 +88,9 @@ private:
VideoCommon::UsageTracker tracker;
VideoCommon::UsageTracker uploads;
VkDeviceAddress device_address{};
u64 usage_tick{};
u64 usage_begin{};
u64 usage_end{};
u64 last_upload_tick{};
bool is_null{};
bool sparse_compatible{};