fix lsfg build desktop + adjustments

This commit is contained in:
CamilleLaVey
2026-09-30 15:06:43 -04:00
parent 5714c3e477
commit 940568bb33
14 changed files with 123 additions and 76 deletions
+1 -1
View File
@@ -157,7 +157,7 @@ private:
[[nodiscard]] std::optional<size_t> RvaToFileOffset(std::span<const Section> sections, u32 rva) {
for (const Section& section : sections) {
const u32 span = std::max(section.virtual_size, section.raw_size);
const u32 span = (std::max)(section.virtual_size, section.raw_size);
if (span == 0 || rva < section.virtual_address) {
continue;
}
@@ -9,9 +9,12 @@
#include "common/fs/fs.h"
#include "common/fs/path_util.h"
#include "common/settings.h"
#include "video_core/host_shaders/vulkan_fidelityfx_fsr_vert_spv.h"
#include "video_core/host_shaders/vulkan_present_frag_spv.h"
#include "video_core/renderer_vulkan/present/frame_gen.h"
#include "video_core/renderer_vulkan/present/util.h"
#include "video_core/renderer_vulkan/vk_scheduler.h"
#include "video_core/renderer_vulkan/vk_shader_util.h"
#include "video_core/vulkan_common/vulkan_device.h"
namespace Vulkan {
@@ -24,7 +27,7 @@ constexpr u32 LSFG_RECURRENCE_FRAMES = 2;
[[nodiscard]] f32 ConfiguredFlowScale() {
if (Settings::values.frame_gen_flow_scale_auto.GetValue()) {
return 1.0f;
return std::clamp(1.0f / Settings::values.resolution_info.up_factor, 0.25f, 1.0f);
}
return static_cast<f32>(Settings::values.frame_gen_flow_scale.GetValue()) / 100.0f;
}
@@ -65,7 +68,7 @@ void WritePortablePixmap(const std::filesystem::path& path, const std::string& m
void WriteGrayscalePgm(const std::filesystem::path& path, VkExtent2D extent,
std::span<const u8> pixels) {
const size_t expected = static_cast<size_t>(extent.width) * extent.height;
WritePortablePixmap(path, "P5", extent, pixels.subspan(0, std::min(expected, pixels.size())));
WritePortablePixmap(path, "P5", extent, pixels.subspan(0, (std::min)(expected, pixels.size())));
}
void WriteRaw(const std::filesystem::path& path, std::span<const u8> pixels) {
@@ -125,58 +128,65 @@ VkImageMemoryBarrier2 MakeTransitionBarrier(VkImage image, VkPipelineStageFlags2
};
}
VkImageBlit2 MakeBlitRegion(VkExtent2D extent) {
const VkImageSubresourceLayers layers{
.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT,
.mipLevel = 0,
.baseArrayLayer = 0,
.layerCount = 1,
};
const VkOffset3D end{
.x = static_cast<s32>(extent.width),
.y = static_cast<s32>(extent.height),
.z = 1,
};
return VkImageBlit2{
.sType = VK_STRUCTURE_TYPE_IMAGE_BLIT_2,
.pNext = nullptr,
.srcSubresource = layers,
.srcOffsets = {VkOffset3D{}, end},
.dstSubresource = layers,
.dstOffsets = {VkOffset3D{}, end},
};
}
void DrawSourceFrame(vk::CommandBuffer cmdbuf, VkPipeline pipeline, VkPipelineLayout layout,
VkDescriptorSet set, LsfgImage& destination, VkExtent2D extent) {
const VkImageMemoryBarrier2 before = MakeTransitionBarrier(
destination.Handle(), VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT, VK_ACCESS_2_NONE,
VK_PIPELINE_STAGE_2_COLOR_ATTACHMENT_OUTPUT_BIT, VK_ACCESS_2_COLOR_ATTACHMENT_WRITE_BIT,
destination.Layout(), VK_IMAGE_LAYOUT_GENERAL);
cmdbuf.PipelineBarrier(before);
void CopySourceFrame(vk::CommandBuffer cmdbuf, VkImage source, LsfgImage& destination,
VkExtent2D extent) {
const std::array before{
MakeTransitionBarrier(source, vk::PIPELINE_STAGE_IMAGE_USERS, vk::ACCESS_IMAGE_WRITES,
VK_PIPELINE_STAGE_2_TRANSFER_BIT, VK_ACCESS_2_TRANSFER_READ_BIT,
VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_GENERAL),
MakeTransitionBarrier(destination.Handle(), VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT,
VK_ACCESS_2_NONE, VK_PIPELINE_STAGE_2_TRANSFER_BIT,
VK_ACCESS_2_TRANSFER_WRITE_BIT, destination.Layout(),
VK_IMAGE_LAYOUT_GENERAL),
const VkViewport viewport{
.x = 0.0f,
.y = 0.0f,
.width = static_cast<f32>(extent.width),
.height = static_cast<f32>(extent.height),
.minDepth = 0.0f,
.maxDepth = 1.0f,
};
cmdbuf.PipelineBarrier(0, {}, {}, before);
cmdbuf.BlitImage(source, VK_IMAGE_LAYOUT_GENERAL, destination.Handle(),
VK_IMAGE_LAYOUT_GENERAL, MakeBlitRegion(extent), VK_FILTER_NEAREST);
const std::array after{
MakeTransitionBarrier(source, VK_PIPELINE_STAGE_2_TRANSFER_BIT, VK_ACCESS_2_NONE,
vk::PIPELINE_STAGE_IMAGE_USERS, VK_ACCESS_2_NONE,
VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_GENERAL),
MakeTransitionBarrier(destination.Handle(), VK_PIPELINE_STAGE_2_TRANSFER_BIT,
VK_ACCESS_2_TRANSFER_WRITE_BIT,
VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT, VK_ACCESS_2_SHADER_READ_BIT,
VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_GENERAL),
const VkRect2D scissor{
.offset = {0, 0},
.extent = extent,
};
cmdbuf.PipelineBarrier(0, {}, {}, after);
BeginRendering(cmdbuf, destination.View(), extent, VK_ATTACHMENT_LOAD_OP_DONT_CARE);
cmdbuf.BindPipeline(VK_PIPELINE_BIND_POINT_GRAPHICS, pipeline);
cmdbuf.BindDescriptorSets(VK_PIPELINE_BIND_POINT_GRAPHICS, layout, 0, set, {});
cmdbuf.SetViewport(0, viewport);
cmdbuf.SetScissor(0, scissor);
cmdbuf.Draw(3, 1, 0, 0);
cmdbuf.EndRendering();
const VkImageMemoryBarrier2 after = MakeTransitionBarrier(
destination.Handle(), VK_PIPELINE_STAGE_2_COLOR_ATTACHMENT_OUTPUT_BIT,
VK_ACCESS_2_COLOR_ATTACHMENT_WRITE_BIT, VK_PIPELINE_STAGE_2_COMPUTE_SHADER_BIT,
VK_ACCESS_2_SHADER_READ_BIT, VK_IMAGE_LAYOUT_GENERAL, VK_IMAGE_LAYOUT_GENERAL);
cmdbuf.PipelineBarrier(after);
destination.SetLayout(VK_IMAGE_LAYOUT_GENERAL);
}
void UpdateSourceSet(const Device& device, VkDescriptorSet set, VkSampler sampler,
VkImageView view) {
const VkDescriptorImageInfo image_info{
.sampler = sampler,
.imageView = view,
.imageLayout = VK_IMAGE_LAYOUT_GENERAL,
};
const VkWriteDescriptorSet write{
.sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET,
.pNext = nullptr,
.dstSet = set,
.dstBinding = 0,
.dstArrayElement = 0,
.descriptorCount = 1,
.descriptorType = VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER,
.pImageInfo = &image_info,
.pBufferInfo = nullptr,
.pTexelBufferView = nullptr,
};
device.GetLogical().UpdateDescriptorSets(std::array{write}, {});
}
} // Anonymous namespace
FrameGen::FrameGen(MemoryAllocator& memory_allocator_, Scheduler& scheduler_)
@@ -184,11 +194,12 @@ FrameGen::FrameGen(MemoryAllocator& memory_allocator_, Scheduler& scheduler_)
FrameGen::~FrameGen() = default;
void FrameGen::Process(const Device& device, VkImage source, VkExtent2D extent) {
void FrameGen::Process(const Device& device, VkImageView source, VkExtent2D extent) {
generated = false;
if (!shaders) {
shaders.emplace(device);
CreateInputPass(device);
}
if (unavailable || !Settings::values.frame_gen.GetValue()) {
@@ -221,14 +232,20 @@ void FrameGen::Process(const Device& device, VkImage source, VkExtent2D extent)
last_count = count;
last_generations = plan.generations;
const bool warm = plan.warm && count + 1 >= LSFG_REQUIRED_FRAMES;
const size_t input_slot = count % INPUT_SLOTS;
const bool warm = plan.warm && count + 1 >= LSFG_REQUIRED_FRAMES &&
scheduler.IsFree(input_ticks[input_slot]);
warm_streak = warm ? warm_streak + 1 : 0;
generated = warm && warm_streak >= LSFG_RECURRENCE_FRAMES && plan.generations > 0;
if (warm) {
const VkDescriptorSet set = input_sets[input_slot];
input_ticks[input_slot] = scheduler.CurrentTick();
UpdateSourceSet(device, set, *input_sampler, source);
scheduler.RequestOutsideRenderPassOperationContext();
scheduler.Record([this, source, extent, count](vk::CommandBuffer cmdbuf) {
CopySourceFrame(cmdbuf, source, chain->Input(count), extent);
scheduler.Record([this, set, extent, count](vk::CommandBuffer cmdbuf) {
DrawSourceFrame(cmdbuf, *input_pipeline, *input_layout, set, chain->Input(count),
extent);
chain->DispatchShared(cmdbuf, count);
});
}
@@ -267,6 +284,21 @@ const LsfgImage& FrameGen::Generate(const Device& device, size_t generation) {
return output;
}
void FrameGen::CreateInputPass(const Device& device) {
input_set_layout =
CreateWrappedDescriptorSetLayout(device, {VK_DESCRIPTOR_TYPE_COMBINED_IMAGE_SAMPLER});
input_layout = CreateWrappedPipelineLayout(device, input_set_layout);
input_pool = CreateWrappedDescriptorPool(device, INPUT_SLOTS, INPUT_SLOTS);
std::array<VkDescriptorSetLayout, INPUT_SLOTS> layouts;
layouts.fill(*input_set_layout);
input_sets = CreateWrappedDescriptorSets(input_pool, layouts);
input_vertex_shader = BuildShader(device, VULKAN_FIDELITYFX_FSR_VERT_SPV);
input_fragment_shader = BuildShader(device, VULKAN_PRESENT_FRAG_SPV);
input_pipeline = CreateWrappedPipeline(device, LSFG_DEFAULT_FORMAT, input_layout,
std::tie(input_vertex_shader, input_fragment_shader));
input_sampler = CreateNearestNeighborSampler(device);
}
void FrameGen::Rebuild(const Device& device, VkExtent2D extent, f32 flow_scale) {
scheduler.Finish();
chain.reset();
@@ -22,7 +22,7 @@ public:
explicit FrameGen(MemoryAllocator& memory_allocator, Scheduler& scheduler);
~FrameGen();
void Process(const Device& device, VkImage source, VkExtent2D extent);
void Process(const Device& device, VkImageView source, VkExtent2D extent);
[[nodiscard]] size_t WantedGenerations(size_t capacity);
@@ -31,6 +31,9 @@ public:
[[nodiscard]] const LsfgImage& Generate(const Device& device, size_t generation);
private:
static constexpr size_t INPUT_SLOTS = 4;
void CreateInputPass(const Device& device);
void Rebuild(const Device& device, VkExtent2D extent, f32 flow_scale);
void DumpDebugImages(u64 count);
@@ -40,6 +43,15 @@ private:
std::optional<LsfgShaders> shaders;
std::optional<LsfgChain> chain;
std::array<LsfgImage, LSFG_MAX_GENERATIONS> outputs;
vk::DescriptorSetLayout input_set_layout;
vk::PipelineLayout input_layout;
vk::DescriptorPool input_pool;
vk::DescriptorSets input_sets;
vk::ShaderModule input_vertex_shader;
vk::ShaderModule input_fragment_shader;
vk::Pipeline input_pipeline;
vk::Sampler input_sampler;
std::array<u64, INPUT_SLOTS> input_ticks{};
FrameGenPacer pacer;
FrameGenPlan plan{};
VkExtent2D built_extent{};
@@ -47,7 +47,7 @@ constexpr auto PROBE_STEP_DELAY = std::chrono::milliseconds(250);
} // Anonymous namespace
FrameGenPlan FrameGenPacer::Plan(size_t capacity) {
const size_t ceiling = std::min(capacity, Settings::FrameGenMaxGenerations());
const size_t ceiling = (std::min)(capacity, Settings::FrameGenMaxGenerations());
if (ceiling == 0) {
Reset();
return {};
@@ -74,7 +74,7 @@ FrameGenPlan FrameGenPacer::Plan(size_t capacity) {
if (smoothed_interval > 0.0f) {
f32 burst_threshold = BURST_CADENCE_RATIO / smoothed_interval;
if (target_rate > 0.0f) {
burst_threshold = std::max(burst_threshold, target_rate * BURST_TARGET_RATIO);
burst_threshold = (std::max)(burst_threshold, target_rate * BURST_TARGET_RATIO);
}
if (1.0f / interval_seconds > burst_threshold) {
DeferEvaluations(interval);
@@ -109,7 +109,7 @@ FrameGenPlan FrameGenPacer::Plan(size_t capacity) {
}
if (target_rate == 0.0f) {
limit = std::min(Settings::FrameGenGenerations(), ceiling);
limit = (std::min)(Settings::FrameGenGenerations(), ceiling);
output_credit = 0.0f;
issued_generations = limit;
return {.generations = limit, .warm = limit > 0};
@@ -117,7 +117,7 @@ FrameGenPlan FrameGenPacer::Plan(size_t capacity) {
UpdateLimit(now, 1.0f / smoothed_interval, target_rate, ceiling);
const size_t allowed = std::min(limit, ceiling);
const size_t allowed = (std::min)(limit, ceiling);
const f32 desired_outputs = smoothed_interval * target_rate;
if (allowed == 0 || desired_outputs <= 1.0f) {
output_credit = 0.0f;
@@ -127,7 +127,7 @@ FrameGenPlan FrameGenPacer::Plan(size_t capacity) {
output_credit += desired_outputs;
const size_t outputs =
std::max<size_t>(1, static_cast<size_t>(std::floor(output_credit + CREDIT_EPSILON)));
const size_t generations = std::min(outputs - 1, allowed);
const size_t generations = (std::min)(outputs - 1, allowed);
output_credit -= static_cast<f32>(generations + 1);
if (output_credit < 0.0f) {
@@ -142,7 +142,7 @@ FrameGenPlan FrameGenPacer::Plan(size_t capacity) {
void FrameGenPacer::UpdateLimit(Clock::time_point now, f32 base_rate, f32 target_rate,
size_t ceiling) {
limit = std::min(limit, ceiling);
limit = (std::min)(limit, ceiling);
if (probe_until) {
if (now < *probe_until) {
@@ -152,9 +152,9 @@ void FrameGenPacer::UpdateLimit(Clock::time_point now, f32 base_rate, f32 target
output_credit = 0.0f;
const f32 previous_output =
std::min(target_rate, probe_base_rate * static_cast<f32>(probe_previous_limit + 1));
(std::min)(target_rate, probe_base_rate * static_cast<f32>(probe_previous_limit + 1));
const f32 current_output =
std::min(target_rate, base_rate * static_cast<f32>(limit + 1));
(std::min)(target_rate, base_rate * static_cast<f32>(limit + 1));
const bool throughput_regressed =
current_output < previous_output * PROBE_THROUGHPUT_TOLERANCE;
@@ -166,7 +166,7 @@ void FrameGenPacer::UpdateLimit(Clock::time_point now, f32 base_rate, f32 target
if (throughput_regressed || collapsed_for_marginal_gain || emulation_slowed) {
limit = probe_previous_limit;
probe_failures = std::min(probe_failures + 1, MAX_PROBE_FAILURES);
probe_failures = (std::min)(probe_failures + 1, MAX_PROBE_FAILURES);
next_probe = now + ProbeBackoff(probe_failures);
deficit_since.reset();
return;
@@ -150,7 +150,7 @@ void Layer::ConfigureDraw(const Device& device, PresentPushConstants* out_push_c
#endif
auto crop_rect = Tegra::NormalizeCrop(framebuffer, texture_width, texture_height);
generation_source = source_image;
generation_source = source_image_view;
generation_extent = source_extent;
generation_render_extent = render_extent;
generation_crop = crop_rect;
@@ -68,7 +68,7 @@ public:
[[nodiscard]] bool IsGenerationFree(size_t generation) const;
[[nodiscard]] VkImage GenerationSource() const {
[[nodiscard]] VkImageView GenerationSource() const {
return generation_source;
}
@@ -125,7 +125,7 @@ private:
#endif
std::vector<u64> resource_ticks{};
std::vector<u64> generation_ticks{};
VkImage generation_source{};
VkImageView generation_source{};
VkExtent2D generation_extent{};
VkExtent2D generation_render_extent{};
Common::Rectangle<f32> generation_crop{};
@@ -47,7 +47,7 @@ LsfgChain::LsfgChain(const Device& device, MemoryAllocator& memory_allocator,
const size_t level = LSFG_MIP_LEVELS - 1 - i;
gamma[i] = LsfgGamma(device, memory_allocator, shaders, resources, descriptor_pool,
alpha[level].Outputs(),
beta.Output(std::min(level, LSFG_BETA_OUTPUTS - 1)),
beta.Output((std::min)(level, LSFG_BETA_OUTPUTS - 1)),
i == 0 ? nullptr : &gamma[i - 1].Output());
if (i < FIRST_DELTA_LEVEL) {
@@ -44,7 +44,8 @@ vk::Image CreateChainImage(MemoryAllocator& memory_allocator, VkExtent2D extent,
.samples = VK_SAMPLE_COUNT_1_BIT,
.tiling = VK_IMAGE_TILING_OPTIMAL,
.usage = VK_IMAGE_USAGE_TRANSFER_SRC_BIT | VK_IMAGE_USAGE_TRANSFER_DST_BIT |
VK_IMAGE_USAGE_STORAGE_BIT | VK_IMAGE_USAGE_SAMPLED_BIT,
VK_IMAGE_USAGE_STORAGE_BIT | VK_IMAGE_USAGE_SAMPLED_BIT |
VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT,
.sharingMode = VK_SHARING_MODE_EXCLUSIVE,
.queueFamilyIndexCount = 0,
.pQueueFamilyIndices = nullptr,
@@ -95,7 +96,7 @@ VkImageMemoryBarrier2 MakeBarrier(const LsfgImage& image, VkAccessFlags2 src_acc
LsfgImage::LsfgImage(const Device& device, MemoryAllocator& memory_allocator, VkExtent2D extent_,
VkFormat format_)
: extent{std::max(1u, extent_.width), std::max(1u, extent_.height)}, format{format_} {
: extent{(std::max)(1u, extent_.width), (std::max)(1u, extent_.height)}, format{format_} {
image = CreateChainImage(memory_allocator, extent, format);
view = CreateWrappedImageView(device, image, format);
}
@@ -40,8 +40,10 @@ LsfgMipmaps::LsfgMipmaps(const Device& device, MemoryAllocator& memory_allocator
const VkExtent2D input_extent = (*frames)[0].Extent();
flow_extent = VkExtent2D{
.width = std::max(1u, static_cast<u32>(static_cast<f32>(input_extent.width) * flow_scale)),
.height = std::max(1u, static_cast<u32>(static_cast<f32>(input_extent.height) * flow_scale)),
.width = (std::max)(1u,
static_cast<u32>(static_cast<f32>(input_extent.width) * flow_scale)),
.height = (std::max)(1u,
static_cast<u32>(static_cast<f32>(input_extent.height) * flow_scale)),
};
for (size_t i = 0; i < LSFG_MIP_LEVELS; ++i) {
@@ -208,7 +208,7 @@ void RendererVulkan::Composite(std::span<const Tegra::FramebufferConfig> framebu
void(frame_gen.WantedGenerations(present_manager.MaxExtraFrames()));
const FrameGenSource source = blit_swapchain.GenerationSource();
frame_gen.Process(device, source.image, source.extent);
frame_gen.Process(device, source.view, source.extent);
const Layout::FramebufferLayout layout = render_window.GetFramebufferLayout();
const size_t generated_frames = frame_gen.GeneratedFrameCount();
@@ -105,7 +105,7 @@ FrameGenSource BlitScreen::GenerationSource() const {
return {};
}
return FrameGenSource{
.image = layers.front().GenerationSource(),
.view = layers.front().GenerationSource(),
.extent = layers.front().GenerationExtent(),
};
}
@@ -49,7 +49,7 @@ struct FramebufferTextureInfo {
};
struct FrameGenSource {
VkImage image{};
VkImageView view{};
VkExtent2D extent{};
};
@@ -175,7 +175,7 @@ Frame* PresentManager::GetRenderFrame() {
Frame* PresentManager::TryGetRenderFrame() {
std::scoped_lock lock{free_mutex};
if (free_queue.empty() || free_queue.front()->present_done.GetStatus() != VK_SUCCESS) {
if (free_queue.size() < 2 || free_queue.front()->present_done.GetStatus() != VK_SUCCESS) {
return nullptr;
}
Frame* const frame = free_queue.front();
@@ -316,7 +316,7 @@ void PresentManager::SetImageCount() {
const size_t generations = Settings::FrameGenMaxGenerations();
const size_t queued_composites = Settings::values.frame_gen_queue_target.GetValue() + 1;
const size_t baseline = swapchain.GetImageCount() + generations;
image_count = std::min<size_t>(std::max((generations + 1) * queued_composites, baseline),
image_count = std::min<size_t>((std::max)((generations + 1) * queued_composites, baseline),
MAX_FRAMES_IN_FLIGHT);
#else
image_count = std::min<size_t>(swapchain.GetImageCount(), MAX_FRAMES_IN_FLIGHT);
@@ -55,7 +55,7 @@ namespace {
QString LosslessSearchDirectory() {
const QString suffix = QStringLiteral("/steamapps/common/Lossless Scaling");
const std::array roots{
const std::array<QString, 4> roots{
QDir::homePath() + QStringLiteral("/.local/share/Steam"),
QDir::homePath() + QStringLiteral("/.steam/steam"),
QDir::homePath() + QStringLiteral("/.var/app/com.valvesoftware.Steam/.local/share/Steam"),