mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-10-06 14:09:58 +00:00
Compare commits
2 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| c9fa7a7505 | |||
| a873fde17f |
-1
@@ -27,7 +27,6 @@ enum class BooleanSetting(override val key: String) : AbstractBooleanSetting {
|
||||
RENDERER_USE_DISK_SHADER_CACHE("use_disk_shader_cache"),
|
||||
RENDERER_FORCE_MAX_CLOCK("force_max_clock"),
|
||||
RENDERER_ASYNCHRONOUS_GPU_EMULATION("use_asynchronous_gpu_emulation"),
|
||||
RENDERER_ASYNC_PRESENTATION("async_presentation"),
|
||||
RENDERER_ASYNCHRONOUS_SHADERS("use_asynchronous_shaders"),
|
||||
RENDERER_REACTIVE_FLUSHING("use_reactive_flushing"),
|
||||
ENABLE_BUFFER_HISTORY("enable_buffer_history"),
|
||||
|
||||
-7
@@ -753,13 +753,6 @@ abstract class SettingsItem(
|
||||
descriptionId = R.string.renderer_asynchronous_gpu_emulation_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.RENDERER_ASYNC_PRESENTATION,
|
||||
titleId = R.string.renderer_async_presentation,
|
||||
descriptionId = R.string.renderer_async_presentation_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SingleChoiceSetting(
|
||||
IntSetting.DMA_ACCURACY,
|
||||
|
||||
-1
@@ -561,7 +561,6 @@ class SettingsFragmentPresenter(
|
||||
add(BooleanSetting.RENDERER_ASYNCHRONOUS_SHADERS.key)
|
||||
add(IntSetting.ANDROID_PIPELINE_WORKERS.key)
|
||||
add(BooleanSetting.RENDERER_ASYNCHRONOUS_GPU_EMULATION.key)
|
||||
add(BooleanSetting.RENDERER_ASYNC_PRESENTATION.key)
|
||||
add(SettingsItem.GPU_UNSWIZZLE_COMBINED)
|
||||
|
||||
add(HeaderSetting(R.string.extensions))
|
||||
|
||||
@@ -574,8 +574,6 @@
|
||||
<string name="renderer_force_max_clock_description">يجبر وحدة معالجة الرسومات على العمل بأقصى سرعة ممكنة (سيظل يتم تطبيق القيود الحرارية).</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">محاكاة غير متزامنة لوحدة معالجة الرسومات</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">يمكن لهذه الحيلة أن تزيد الأداء عن طريق تشغيل محاكاة وحدة معالجة الرسومات بشكل غير متزامن على حساب مشاكل الرسومات وزيادة معدلات الأعطال بسبب العمليات المتعلقة بالتوقيت.</string>
|
||||
<string name="renderer_async_presentation">عرض غير متزامن</string>
|
||||
<string name="renderer_async_presentation_description">يمكن لهذه الحيلة أن تزيد من الأداء عن طريق نقل عملية العرض إلى خيط معالجة منفصل على حساب مشاكل الرسوميات.</string>
|
||||
<string name="renderer_reactive_flushing">استخدم التنظيف التفاعلي</string>
|
||||
<string name="renderer_reactive_flushing_description">يحسن دقة العرض في بعض الألعاب على حساب الأداء.</string>
|
||||
<string name="enable_buffer_history">تفعيل سجل التخزين المؤقت</string>
|
||||
|
||||
@@ -518,8 +518,6 @@
|
||||
<string name="renderer_force_max_clock_description">Fuerza a la GPU a ejecutarse a la velocidad máxima de reloj posible (se seguirán aplicando restricciones térmicas).</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">Emulación de GPU asíncrona</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">Este hack puede aumentar el rendimiento ejecutando la emulación de la GPU de forma asíncrona, a costa de problemas gráficos y un aumento en la tasa de fallos debido a operaciones relacionadas con la sincronización.</string>
|
||||
<string name="renderer_async_presentation">Presentación asíncrona</string>
|
||||
<string name="renderer_async_presentation_description">Este hack puede aumentar el rendimiento al mover la presentación a un hilo independiente de la CPU a costa de problemas gráficos.</string>
|
||||
<string name="renderer_reactive_flushing">Usar limpieza reactiva</string>
|
||||
<string name="renderer_reactive_flushing_description">Mejora la precisión de renderizado en algunos juegos, pero reduce el rendimiento.</string>
|
||||
<string name="enable_buffer_history">Activar el historial del búfer</string>
|
||||
|
||||
@@ -563,8 +563,6 @@
|
||||
<string name="renderer_force_max_clock_description">Заставляет ГПУ работать на максимально возможных тактовых частотах (тепловые ограничения все равно будут применяться).</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">Асинхронная эмуляция ГПУ</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">Может повысить производительность за счёт асинхронного запуска эмуляции ГПУ, но ценой появления графических ошибок и увеличения частоты вылетов из-за операций, зависящих от синхронизации.</string>
|
||||
<string name="renderer_async_presentation">Асинхронная презентация</string>
|
||||
<string name="renderer_async_presentation_description">Может повысить производительность за счёт перемещения вывода кадров в отдельный поток ЦП, но ценой возникновения графических проблем.</string>
|
||||
<string name="renderer_reactive_flushing">Реактивная очистка</string>
|
||||
<string name="renderer_reactive_flushing_description">Повышение точности рендеринга в некоторых играх за счет снижения производительности.</string>
|
||||
<string name="enable_buffer_history">Включить историю буфера</string>
|
||||
|
||||
@@ -481,8 +481,6 @@
|
||||
<string name="renderer_force_max_clock_description">Змушує GPU працювати на максимальній тактовій частоті.</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">Асинхронна емуляція ГП</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">Це обхідне рішення може покращити продуктивність завдяки асинхронному виконанню емуляції ГП, але спричинить проблеми з графікою та збільшить частоту збоїв чутливих до таймінгів операцій.</string>
|
||||
<string name="renderer_async_presentation">Асинхронне подання</string>
|
||||
<string name="renderer_async_presentation_description">Це обхідне рішення може покращити продуктивність завдяки переміщенню подання на окремий потік ЦП, але спричинить проблеми з графікою.</string>
|
||||
<string name="renderer_reactive_flushing">Реактивне очищення</string>
|
||||
<string name="renderer_reactive_flushing_description">Покращує точність рендерингу в деяких іграх.</string>
|
||||
<string name="enable_buffer_history">Увімкнути історію буфера</string>
|
||||
|
||||
@@ -564,8 +564,6 @@
|
||||
<string name="renderer_force_max_clock_description">强制 GPU 以最大时钟运行 (温控依然生效)。</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">GPU 异步模拟</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">此技巧可通过异步运行 GPU 模拟来提升性能,但在执行与时序相关的操作时,可能带来图形显示问题以及增加崩溃概率。</string>
|
||||
<string name="renderer_async_presentation">异步呈现</string>
|
||||
<string name="renderer_async_presentation_description">此技巧通过将图形呈现移至独立的 CPU 线程来提升性能,但可能会带来图形显示问题。</string>
|
||||
<string name="renderer_reactive_flushing">启用反应性刷新</string>
|
||||
<string name="renderer_reactive_flushing_description">通过牺牲性能来提升某些游戏的渲染精度。</string>
|
||||
<string name="enable_buffer_history">启用缓冲区历史</string>
|
||||
|
||||
@@ -560,12 +560,10 @@
|
||||
<string name="sync_memory_operations_description">確保計算和記憶體操作之間的資料一致性,此選項應能修復某些遊戲中的問題,但在某些情況下可能會降低效能。使用Unreal Engine 4的遊戲似乎受影響最大</string>
|
||||
<string name="use_disk_shader_cache">磁碟著色器快取</string>
|
||||
<string name="use_disk_shader_cache_description">將產生的著色器快取儲存至硬碟以減少中斷</string>
|
||||
<string name="renderer_force_max_clock">強制使用最大時脈(僅限Adreno)</string>
|
||||
<string name="renderer_force_max_clock_description">強制GPU以可能的最大時脈運作(熱溫限制仍會被套用)</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">GPU非同步模擬</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">此修改可透過使用GPU非同步模擬來提升性能,不過可能會導致圖形問題且會提高在進行時序相關操作時當機的機率</string>
|
||||
<string name="renderer_async_presentation">非同步呈現</string>
|
||||
<string name="renderer_async_presentation_description">此修改可以透過將渲染移至單獨的CPU執行緒來提升性能,但可能會導致圖形問題</string>
|
||||
<string name="renderer_force_max_clock">強制使用最大時脈 (僅限Adreno)</string>
|
||||
<string name="renderer_force_max_clock_description">強制 GPU 以可能的最大時脈執行 (熱溫限制仍會被套用)</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">GPU 非同步模擬</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">此修改可透過使用 GPU 非同步模擬來提升性能,不過可能會導致圖形問題且會提高在進行時序相關操作時當機的機率</string>
|
||||
<string name="renderer_reactive_flushing">使用重新啟用排清</string>
|
||||
<string name="renderer_reactive_flushing_description">犧牲效能以改善部分遊戲的轉譯準確度</string>
|
||||
<string name="enable_buffer_history">啟用緩衝區歷史</string>
|
||||
|
||||
@@ -580,8 +580,6 @@
|
||||
<string name="renderer_force_max_clock_description">Forces the GPU to run at the maximum possible clocks (thermal constraints will still be applied).</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">GPU async emulation</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">This hack can increase performance by running GPU emulation asynchronously at the cost of graphical issues and increased crash rates by timing-related operations.</string>
|
||||
<string name="renderer_async_presentation">Asynchronous presentation</string>
|
||||
<string name="renderer_async_presentation_description">This hack can increase performance by moving presentation to a separate CPU thread at the cost of graphical issues.</string>
|
||||
<string name="renderer_reactive_flushing">Use reactive flushing</string>
|
||||
<string name="renderer_reactive_flushing_description">Improves rendering accuracy in some games at the cost of performance.</string>
|
||||
<string name="enable_buffer_history">Enable buffer history</string>
|
||||
|
||||
@@ -472,13 +472,8 @@ struct Values {
|
||||
SwitchableSetting<bool> frame_gen_dump_flow{linkage, false, "frame_gen_dump_flow",
|
||||
Category::Renderer};
|
||||
|
||||
SwitchableSetting<bool> use_asynchronous_gpu_emulation{linkage,
|
||||
#ifdef __ANDROID__
|
||||
false,
|
||||
#else
|
||||
true,
|
||||
#endif
|
||||
"use_asynchronous_gpu_emulation", Category::Renderer};
|
||||
SwitchableSetting<bool> use_asynchronous_gpu_emulation{linkage, true, "use_asynchronous_gpu_emulation",
|
||||
Category::Renderer};
|
||||
// *nix platforms may have issues with the borderless windowed fullscreen mode.
|
||||
// Default to exclusive fullscreen on these platforms for now.
|
||||
SwitchableSetting<FullscreenMode, true> fullscreen_mode{linkage,
|
||||
@@ -645,7 +640,7 @@ struct Values {
|
||||
linkage, false, "nce_runtime_nro_patch", Category::RendererHacks};
|
||||
SwitchableSetting<bool> async_presentation{linkage,
|
||||
#ifdef __ANDROID__
|
||||
false,
|
||||
true,
|
||||
#else
|
||||
false,
|
||||
#endif
|
||||
|
||||
@@ -172,7 +172,6 @@ add_library(shader_recompiler STATIC
|
||||
ir_opt/constant_propagation_pass.cpp
|
||||
ir_opt/dead_code_elimination_pass.cpp
|
||||
ir_opt/dual_vertex_pass.cpp
|
||||
ir_opt/geometry_compaction_pass.cpp
|
||||
ir_opt/global_memory_to_storage_buffer_pass.cpp
|
||||
ir_opt/identity_removal_pass.cpp
|
||||
ir_opt/layer_pass.cpp
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <algorithm>
|
||||
#include <span>
|
||||
#include <tuple>
|
||||
#include <type_traits>
|
||||
@@ -440,13 +439,10 @@ void SetupCapabilities(const Profile& profile, const Info& info, EmitContext& ct
|
||||
ctx.AddCapability(spv::Capability::DrawParameters);
|
||||
}
|
||||
if ((info.uses_subgroup_vote || info.uses_subgroup_invocation_id ||
|
||||
info.uses_subgroup_shuffles || info.uses_subgroup_mask) &&
|
||||
info.uses_subgroup_shuffles) &&
|
||||
profile.support_vote && profile.SupportsSubgroupStage(ctx.stage)) {
|
||||
ctx.AddCapability(spv::Capability::GroupNonUniformBallot);
|
||||
ctx.AddCapability(spv::Capability::GroupNonUniformShuffle);
|
||||
if (info.uses_subgroup_shuffles && profile.support_shuffle_relative) {
|
||||
ctx.AddCapability(spv::Capability::GroupNonUniformShuffleRelative);
|
||||
}
|
||||
if (!profile.warp_size_potentially_larger_than_guest) {
|
||||
// vote ops are only used when not taking the long path
|
||||
ctx.AddCapability(spv::Capability::GroupNonUniformVote);
|
||||
@@ -525,29 +521,6 @@ void PatchPhiNodes(IR::Program& program, EmitContext& ctx) {
|
||||
return { ctx.Def(phi->Arg(phi_arg)), parent };
|
||||
});
|
||||
}
|
||||
|
||||
void RewriteOpcodes(std::vector<u32>& code, std::span<const std::pair<u32, spv::Op>> rewrites) {
|
||||
if (rewrites.empty()) {
|
||||
return;
|
||||
}
|
||||
size_t offset = 5;
|
||||
while (offset + 2 < code.size()) {
|
||||
const u32 word_count{code[offset] >> 16};
|
||||
const auto opcode{static_cast<spv::Op>(code[offset] & 0xFFFFu)};
|
||||
if (word_count == 0) {
|
||||
return;
|
||||
}
|
||||
if (opcode == spv::Op::OpGroupNonUniformShuffleXor ||
|
||||
opcode == spv::Op::OpGroupNonUniformQuadBroadcast) {
|
||||
const auto it{std::ranges::find(rewrites, code[offset + 2],
|
||||
&std::pair<u32, spv::Op>::first)};
|
||||
if (it != rewrites.end()) {
|
||||
code[offset] = (code[offset] & 0xFFFF0000u) | static_cast<u32>(it->second);
|
||||
}
|
||||
}
|
||||
offset += word_count;
|
||||
}
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
std::vector<u32> EmitSPIRV(const Profile& profile, const RuntimeInfo& runtime_info, IR::Program& program, Bindings& bindings) {
|
||||
@@ -562,9 +535,7 @@ std::vector<u32> EmitSPIRV(const Profile& profile, const RuntimeInfo& runtime_in
|
||||
SetupCapabilities(profile, program.info, ctx);
|
||||
SetupTransformFeedbackCapabilities(ctx, main);
|
||||
PatchPhiNodes(program, ctx);
|
||||
std::vector<u32> code{ctx.Assemble()};
|
||||
RewriteOpcodes(code, ctx.opcode_rewrites);
|
||||
return code;
|
||||
return ctx.Assemble();
|
||||
}
|
||||
|
||||
Id EmitPhi(EmitContext& ctx, IR::Inst* inst) {
|
||||
|
||||
@@ -77,57 +77,20 @@ Id GetMaxThreadId(EmitContext& ctx, Id thread_id, Id clamp, Id segmentation_mask
|
||||
return ComputeMaxThreadId(ctx, min_thread_id, clamp, not_seg_mask);
|
||||
}
|
||||
|
||||
Id HostThreadId(EmitContext& ctx, Id thread_id) {
|
||||
if (!ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
return thread_id;
|
||||
Id SelectValue(EmitContext& ctx, Id in_range, Id value, Id src_thread_id) {
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return value;
|
||||
}
|
||||
return ctx.OpSelect(
|
||||
ctx.U32[1], in_range,
|
||||
ctx.OpGroupNonUniformShuffle(ctx.U32[1], SubgroupScope(ctx), value, src_thread_id), value);
|
||||
}
|
||||
|
||||
Id AddPartitionBase(EmitContext& ctx, Id thread_id) {
|
||||
const Id partition_idx{ctx.OpShiftRightLogical(ctx.U32[1], GetThreadId(ctx), ctx.Const(5u))};
|
||||
const Id partition_base{ctx.OpShiftLeftLogical(ctx.U32[1], partition_idx, ctx.Const(5u))};
|
||||
return ctx.OpIAdd(ctx.U32[1], thread_id, partition_base);
|
||||
}
|
||||
|
||||
Id GuestLane(EmitContext& ctx, Id index) {
|
||||
return ctx.OpBitwiseAnd(ctx.U32[1], index, ctx.Const(31U));
|
||||
}
|
||||
|
||||
Id ShuffleAbsolute(EmitContext& ctx, Id value, Id src_thread_id) {
|
||||
if (!ctx.profile.has_broken_spirv_subgroup_shuffle) {
|
||||
return ctx.OpGroupNonUniformShuffle(ctx.U32[1], SubgroupScope(ctx), value, src_thread_id);
|
||||
}
|
||||
Id result{ctx.u32_zero_value};
|
||||
for (u32 lane = 0; lane < ctx.profile.max_subgroup_size; ++lane) {
|
||||
const Id read{
|
||||
ctx.OpGroupNonUniformBroadcast(ctx.U32[1], SubgroupScope(ctx), value, ctx.Const(lane))};
|
||||
const Id matches{ctx.OpIEqual(ctx.U1, src_thread_id, ctx.Const(lane))};
|
||||
result = ctx.OpSelect(ctx.U32[1], matches, read, result);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
Id ShuffleRelative(EmitContext& ctx, Id value, Id delta, Id src_thread_id, spv::Op op) {
|
||||
if (!ctx.profile.support_shuffle_relative) {
|
||||
return ShuffleAbsolute(ctx, value, HostThreadId(ctx, src_thread_id));
|
||||
}
|
||||
const Id result{ctx.OpGroupNonUniformShuffleXor(ctx.U32[1], SubgroupScope(ctx), value, delta)};
|
||||
ctx.opcode_rewrites.emplace_back(result.value, op);
|
||||
return result;
|
||||
}
|
||||
|
||||
Id BroadcastLane(EmitContext& ctx, Id value, u32 lane) {
|
||||
Id result{
|
||||
ctx.OpGroupNonUniformBroadcast(ctx.U32[1], SubgroupScope(ctx), value, ctx.Const(lane))};
|
||||
if (!ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
return result;
|
||||
}
|
||||
const Id partition_idx{ctx.OpShiftRightLogical(ctx.U32[1], GetThreadId(ctx), ctx.Const(5u))};
|
||||
for (u32 base = 32; base < ctx.profile.max_subgroup_size; base += 32) {
|
||||
const Id read{ctx.OpGroupNonUniformBroadcast(ctx.U32[1], SubgroupScope(ctx), value,
|
||||
ctx.Const(base + lane))};
|
||||
const Id matches{ctx.OpIEqual(ctx.U1, partition_idx, ctx.Const(base >> 5))};
|
||||
result = ctx.OpSelect(ctx.U32[1], matches, read, result);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
Id EmitLaneId(EmitContext& ctx) {
|
||||
@@ -240,75 +203,61 @@ Id EmitShuffleIndex(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id cla
|
||||
const Id min_thread_id{ComputeMinThreadId(ctx, thread_id, segmentation_mask)};
|
||||
const Id max_thread_id{ComputeMaxThreadId(ctx, min_thread_id, clamp, not_seg_mask)};
|
||||
|
||||
const Id lhs{ctx.OpBitwiseAnd(ctx.U32[1], GuestLane(ctx, index), not_seg_mask)};
|
||||
const Id src_thread_id{ctx.OpBitwiseOr(ctx.U32[1], lhs, min_thread_id)};
|
||||
const Id lhs{ctx.OpBitwiseAnd(ctx.U32[1], index, not_seg_mask)};
|
||||
Id src_thread_id{ctx.OpBitwiseOr(ctx.U32[1], lhs, min_thread_id)};
|
||||
const Id in_range{ctx.OpSLessThanEqual(ctx.U1, src_thread_id, max_thread_id)};
|
||||
|
||||
if (ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
src_thread_id = AddPartitionBase(ctx, src_thread_id);
|
||||
}
|
||||
|
||||
SetInBoundsFlag(inst, in_range);
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return value;
|
||||
}
|
||||
const IR::Value lane{inst->Arg(1).Resolve()};
|
||||
const IR::Value segment{inst->Arg(3).Resolve()};
|
||||
if (lane.IsImmediate() && segment.IsImmediate() && segment.U32() == 0) {
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, BroadcastLane(ctx, value, lane.U32() & 31),
|
||||
value);
|
||||
}
|
||||
const Id shuffled{ShuffleAbsolute(ctx, value, HostThreadId(ctx, src_thread_id))};
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, shuffled, value);
|
||||
return SelectValue(ctx, in_range, value, src_thread_id);
|
||||
}
|
||||
|
||||
Id EmitShuffleUp(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id delta{GuestLane(ctx, index)};
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id max_thread_id{GetMaxThreadId(ctx, thread_id, clamp, segmentation_mask)};
|
||||
const Id src_thread_id{ctx.OpISub(ctx.U32[1], thread_id, delta)};
|
||||
Id src_thread_id{ctx.OpISub(ctx.U32[1], thread_id, index)};
|
||||
const Id in_range{ctx.OpSGreaterThanEqual(ctx.U1, src_thread_id, max_thread_id)};
|
||||
|
||||
if (ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
src_thread_id = AddPartitionBase(ctx, src_thread_id);
|
||||
}
|
||||
|
||||
SetInBoundsFlag(inst, in_range);
|
||||
const Id shuffled{ShuffleRelative(ctx, value, delta, src_thread_id,
|
||||
spv::Op::OpGroupNonUniformShuffleUp)};
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, shuffled, value);
|
||||
return SelectValue(ctx, in_range, value, src_thread_id);
|
||||
}
|
||||
|
||||
Id EmitShuffleDown(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id delta{GuestLane(ctx, index)};
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id max_thread_id{GetMaxThreadId(ctx, thread_id, clamp, segmentation_mask)};
|
||||
const Id src_thread_id{ctx.OpIAdd(ctx.U32[1], thread_id, delta)};
|
||||
Id src_thread_id{ctx.OpIAdd(ctx.U32[1], thread_id, index)};
|
||||
const Id in_range{ctx.OpSLessThanEqual(ctx.U1, src_thread_id, max_thread_id)};
|
||||
|
||||
if (ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
src_thread_id = AddPartitionBase(ctx, src_thread_id);
|
||||
}
|
||||
|
||||
SetInBoundsFlag(inst, in_range);
|
||||
const Id shuffled{ShuffleRelative(ctx, value, delta, src_thread_id,
|
||||
spv::Op::OpGroupNonUniformShuffleDown)};
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, shuffled, value);
|
||||
return SelectValue(ctx, in_range, value, src_thread_id);
|
||||
}
|
||||
|
||||
Id EmitShuffleButterfly(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id mask{GuestLane(ctx, index)};
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id max_thread_id{GetMaxThreadId(ctx, thread_id, clamp, segmentation_mask)};
|
||||
const Id src_thread_id{ctx.OpBitwiseXor(ctx.U32[1], thread_id, mask)};
|
||||
Id src_thread_id{ctx.OpBitwiseXor(ctx.U32[1], thread_id, index)};
|
||||
const Id in_range{ctx.OpSLessThanEqual(ctx.U1, src_thread_id, max_thread_id)};
|
||||
|
||||
if (ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
src_thread_id = AddPartitionBase(ctx, src_thread_id);
|
||||
}
|
||||
|
||||
SetInBoundsFlag(inst, in_range);
|
||||
const Id shuffled{ctx.OpGroupNonUniformShuffleXor(ctx.U32[1], SubgroupScope(ctx), value, mask)};
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, shuffled, value);
|
||||
return SelectValue(ctx, in_range, value, src_thread_id);
|
||||
}
|
||||
|
||||
Id EmitQuadBroadcast(EmitContext& ctx, Id value, Id lane) {
|
||||
@@ -318,16 +267,10 @@ Id EmitQuadBroadcast(EmitContext& ctx, Id value, Id lane) {
|
||||
const Id base{ctx.OpBitwiseAnd(ctx.U32[1], GetThreadId(ctx), ctx.Const(~3u))};
|
||||
const Id local_lane{ctx.OpBitwiseAnd(ctx.U32[1], lane, ctx.Const(3u))};
|
||||
const Id src_thread_id{ctx.OpBitwiseOr(ctx.U32[1], base, local_lane)};
|
||||
return ShuffleAbsolute(ctx, value, src_thread_id);
|
||||
return ctx.OpGroupNonUniformShuffle(ctx.U32[1], SubgroupScope(ctx), value, src_thread_id);
|
||||
}
|
||||
|
||||
Id EmitQuadSwap(EmitContext& ctx, Id value, Id direction) {
|
||||
if (ctx.profile.support_quad_shuffles) {
|
||||
const Id result{
|
||||
ctx.OpGroupNonUniformQuadBroadcast(ctx.U32[1], SubgroupScope(ctx), value, direction)};
|
||||
ctx.opcode_rewrites.emplace_back(result.value, spv::Op::OpGroupNonUniformQuadSwap);
|
||||
return result;
|
||||
}
|
||||
const Id xor_mask{ctx.OpIAdd(ctx.U32[1], direction, ctx.Const(1u))};
|
||||
return ctx.OpGroupNonUniformShuffleXor(ctx.U32[1], SubgroupScope(ctx), value, xor_mask);
|
||||
}
|
||||
|
||||
@@ -361,7 +361,6 @@ public:
|
||||
Id frag_depth{};
|
||||
|
||||
std::vector<Id> interfaces;
|
||||
std::vector<std::pair<u32, spv::Op>> opcode_rewrites;
|
||||
|
||||
Id load_const_func_u8{};
|
||||
Id load_const_func_u16{};
|
||||
|
||||
@@ -5,10 +5,7 @@
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <bitset>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include <vector>
|
||||
#include <queue>
|
||||
|
||||
@@ -175,17 +172,7 @@ std::map<IR::Attribute, IR::Attribute> GenerateLegacyToGenericMappings(
|
||||
void EmitGeometryPassthrough(IR::IREmitter& ir, const IR::Program& program,
|
||||
const Shader::VaryingState& passthrough_mask,
|
||||
bool passthrough_position,
|
||||
std::optional<IR::Attribute> passthrough_layer_attr,
|
||||
std::optional<IR::Reg> viewport_mask_reg) {
|
||||
constexpr std::array CULLED_POSITION{2.0f, 2.0f, 2.0f, 1.0f};
|
||||
IR::U1 culled{ir.Imm1(false)};
|
||||
IR::U32 viewport{ir.Imm32(0)};
|
||||
if (viewport_mask_reg) {
|
||||
const IR::U32 mask{ir.GetReg(*viewport_mask_reg)};
|
||||
const IR::U32 lowest_bit{ir.BitwiseAnd(mask, IR::U32{ir.INeg(mask)})};
|
||||
culled = ir.IEqual(mask, ir.Imm32(0));
|
||||
viewport = IR::U32{ir.Select(culled, ir.Imm32(0), ir.FindUMsb(lowest_bit))};
|
||||
}
|
||||
std::optional<IR::Attribute> passthrough_layer_attr) {
|
||||
for (u32 i = 0; i < program.output_vertices; i++) {
|
||||
// Assign generics from input
|
||||
for (u32 j = 0; j < 32; j++) {
|
||||
@@ -203,19 +190,10 @@ void EmitGeometryPassthrough(IR::IREmitter& ir, const IR::Program& program,
|
||||
if (passthrough_position) {
|
||||
// Assign position from input
|
||||
const IR::Attribute attr = IR::Attribute::PositionX;
|
||||
for (u32 component = 0; component < 4; ++component) {
|
||||
IR::F32 value{ir.GetAttribute(attr + component, ir.Imm32(i))};
|
||||
if (viewport_mask_reg) {
|
||||
value = IR::F32{
|
||||
ir.Select(culled, ir.Imm32(CULLED_POSITION[component]), value)};
|
||||
}
|
||||
ir.SetAttribute(attr + component, value, ir.Imm32(0));
|
||||
}
|
||||
}
|
||||
|
||||
if (viewport_mask_reg) {
|
||||
ir.SetAttribute(IR::Attribute::ViewportIndex, ir.BitCast<IR::F32>(viewport),
|
||||
ir.Imm32(0));
|
||||
ir.SetAttribute(attr + 0, ir.GetAttribute(attr + 0, ir.Imm32(i)), ir.Imm32(0));
|
||||
ir.SetAttribute(attr + 1, ir.GetAttribute(attr + 1, ir.Imm32(i)), ir.Imm32(0));
|
||||
ir.SetAttribute(attr + 2, ir.GetAttribute(attr + 2, ir.Imm32(i)), ir.Imm32(0));
|
||||
ir.SetAttribute(attr + 3, ir.GetAttribute(attr + 3, ir.Imm32(i)), ir.Imm32(0));
|
||||
}
|
||||
|
||||
if (passthrough_layer_attr) {
|
||||
@@ -241,91 +219,19 @@ u32 GetOutputTopologyVertices(OutputTopology output_topology) {
|
||||
}
|
||||
}
|
||||
|
||||
std::optional<IR::Reg> FindFreeRegister(const IR::Program& program) {
|
||||
std::bitset<IR::NUM_REGS> used;
|
||||
for (IR::Block* const block : program.blocks) {
|
||||
for (const IR::Inst& inst : block->Instructions()) {
|
||||
const IR::Opcode opcode{inst.GetOpcode()};
|
||||
if (opcode == IR::Opcode::GetRegister || opcode == IR::Opcode::SetRegister) {
|
||||
used.set(IR::RegIndex(inst.Arg(0).Reg()));
|
||||
}
|
||||
}
|
||||
}
|
||||
for (size_t index = IR::NUM_USER_REGS; index-- > 0;) {
|
||||
if (!used.test(index)) {
|
||||
return static_cast<IR::Reg>(index);
|
||||
}
|
||||
}
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
std::optional<IR::Reg> LowerViewportMask(const IR::Program& program) {
|
||||
const std::optional<IR::Reg> reg{FindFreeRegister(program)};
|
||||
if (!reg || program.blocks.empty()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
bool stores_mask{};
|
||||
for (IR::Block* const block : program.blocks) {
|
||||
for (IR::Inst& inst : block->Instructions()) {
|
||||
if (inst.GetOpcode() != IR::Opcode::SetAttribute ||
|
||||
inst.Arg(0).Attribute() != IR::Attribute::ViewportMask) {
|
||||
continue;
|
||||
}
|
||||
IR::IREmitter ir{*block, IR::Block::InstructionList::s_iterator_to(inst)};
|
||||
ir.SetReg(*reg, ir.BitCast<IR::U32>(IR::F32{inst.Arg(1)}));
|
||||
inst.Invalidate();
|
||||
stores_mask = true;
|
||||
}
|
||||
}
|
||||
if (!stores_mask) {
|
||||
return std::nullopt;
|
||||
}
|
||||
IR::Block& entry{*program.blocks.front()};
|
||||
IR::IREmitter ir{entry, entry.begin()};
|
||||
ir.SetReg(*reg, ir.Imm32(1));
|
||||
return reg;
|
||||
}
|
||||
|
||||
void LowerGeometryPassthrough(const IR::Program& program, const HostTranslateInfo& host_info) {
|
||||
std::optional<IR::Reg> viewport_mask_reg;
|
||||
if (!host_info.support_viewport_mask) {
|
||||
viewport_mask_reg = LowerViewportMask(program);
|
||||
}
|
||||
for (IR::Block* const block : program.blocks) {
|
||||
for (IR::Inst& inst : block->Instructions()) {
|
||||
if (inst.GetOpcode() == IR::Opcode::Epilogue) {
|
||||
IR::IREmitter ir{*block, IR::Block::InstructionList::s_iterator_to(inst)};
|
||||
EmitGeometryPassthrough(
|
||||
ir, program, program.info.passthrough,
|
||||
program.info.passthrough.AnyComponent(IR::Attribute::PositionX), {},
|
||||
viewport_mask_reg);
|
||||
program.info.passthrough.AnyComponent(IR::Attribute::PositionX), {});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void TightenOutputVertices(IR::Program& program) {
|
||||
if (program.stage != Stage::Geometry || program.is_geometry_passthrough) {
|
||||
return;
|
||||
}
|
||||
const bool has_loops = std::ranges::any_of(program.syntax_list, [](const auto& node) {
|
||||
return node.type == IR::AbstractSyntaxNode::Type::Loop;
|
||||
});
|
||||
if (has_loops) {
|
||||
return;
|
||||
}
|
||||
u32 num_emits = 0;
|
||||
for (IR::Block* const block : program.blocks) {
|
||||
for (const IR::Inst& inst : block->Instructions()) {
|
||||
if (inst.GetOpcode() == IR::Opcode::EmitVertex) {
|
||||
++num_emits;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (num_emits != 0) {
|
||||
program.output_vertices = (std::min)(program.output_vertices, num_emits);
|
||||
}
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
IR::Program TranslateProgram(ObjectPool<IR::Inst>& inst_pool, ObjectPool<IR::Block>& block_pool,
|
||||
@@ -393,14 +299,12 @@ IR::Program TranslateProgram(ObjectPool<IR::Inst>& inst_pool, ObjectPool<IR::Blo
|
||||
Optimization::PositionPass(env, program);
|
||||
|
||||
Optimization::GlobalMemoryToStorageBufferPass(program, normalized_host_info);
|
||||
Optimization::GeometryCompactionPass(program, normalized_host_info);
|
||||
Optimization::TexturePass(env, program, normalized_host_info);
|
||||
|
||||
if (Settings::values.resolution_info.active || Settings::values.rescale_hack.GetValue()) {
|
||||
Optimization::RescalingPass(program);
|
||||
}
|
||||
Optimization::DeadCodeEliminationPass(program);
|
||||
TightenOutputVertices(program);
|
||||
if (Settings::values.renderer_debug) {
|
||||
Optimization::VerificationPass(program);
|
||||
}
|
||||
@@ -529,7 +433,7 @@ IR::Program GenerateGeometryPassthrough(ObjectPool<IR::Inst>& inst_pool,
|
||||
|
||||
IR::IREmitter ir{*current_block};
|
||||
EmitGeometryPassthrough(ir, program, program.info.stores, true,
|
||||
source_program.info.emulated_layer, std::nullopt);
|
||||
source_program.info.emulated_layer);
|
||||
|
||||
IR::Block* return_block{block_pool.Create(inst_pool)};
|
||||
IR::IREmitter{*return_block}.Epilogue();
|
||||
|
||||
@@ -34,12 +34,10 @@ struct HostTranslateInfo {
|
||||
bool needs_demote_reorder{}; ///< True when the device needs DemoteToHelperInvocation reordered
|
||||
bool support_snorm_render_buffer{}; ///< True when the device supports SNORM render buffers
|
||||
bool support_viewport_index_layer{}; ///< True when the device supports gl_Layer in VS
|
||||
bool support_viewport_mask{};
|
||||
bool support_geometry_shader_passthrough{}; ///< True when the device supports geometry
|
||||
///< passthrough shaders
|
||||
bool support_conditional_barrier{}; ///< True when the device supports barriers in conditional
|
||||
///< control flow
|
||||
bool single_lane_geometry_subgroups{};
|
||||
|
||||
void ApplyDescriptorLimitPolicy() noexcept {
|
||||
if (min_ssbo_alignment == 0) {
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
@@ -651,29 +651,6 @@ IR::Value GetThroughCast(IR::Value value, IR::Opcode expected_cast) {
|
||||
return value;
|
||||
}
|
||||
|
||||
u32 QuadButterflyMask(const IR::Inst& inst) {
|
||||
if (inst.GetOpcode() == IR::Opcode::QuadSwap) {
|
||||
const IR::Value direction{inst.Arg(1)};
|
||||
if (!direction.IsImmediate()) {
|
||||
return 0;
|
||||
}
|
||||
return direction.U32() + 1;
|
||||
}
|
||||
if (inst.GetOpcode() != IR::Opcode::ShuffleButterfly) {
|
||||
return 0;
|
||||
}
|
||||
const IR::Value index{inst.Arg(1)};
|
||||
const IR::Value clamp{inst.Arg(2)};
|
||||
const IR::Value segmentation_mask{inst.Arg(3)};
|
||||
if (!index.IsImmediate() || !clamp.IsImmediate() || !segmentation_mask.IsImmediate()) {
|
||||
return 0;
|
||||
}
|
||||
if (clamp.U32() != 3 || segmentation_mask.U32() != 28) {
|
||||
return 0;
|
||||
}
|
||||
return index.U32();
|
||||
}
|
||||
|
||||
void FoldFSwizzleAdd(IR::Block& block, IR::Inst& inst) {
|
||||
const IR::Value swizzle{inst.Arg(2)};
|
||||
if (!swizzle.IsImmediate()) {
|
||||
@@ -689,8 +666,7 @@ void FoldFSwizzleAdd(IR::Block& block, IR::Inst& inst) {
|
||||
return;
|
||||
}
|
||||
IR::Inst* const inst2{value_1.InstRecursive()};
|
||||
const u32 lane_mask{QuadButterflyMask(*inst2)};
|
||||
if (lane_mask == 0) {
|
||||
if (inst2->GetOpcode() != IR::Opcode::ShuffleButterfly) {
|
||||
return;
|
||||
}
|
||||
const IR::Value value_3{GetThroughCast(inst2->Arg(0).Resolve(), IR::Opcode::BitCastU32F32)};
|
||||
@@ -702,15 +678,24 @@ void FoldFSwizzleAdd(IR::Block& block, IR::Inst& inst) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
const IR::Value index{inst2->Arg(1)};
|
||||
const IR::Value clamp{inst2->Arg(2)};
|
||||
const IR::Value segmentation_mask{inst2->Arg(3)};
|
||||
if (!index.IsImmediate() || !clamp.IsImmediate() || !segmentation_mask.IsImmediate()) {
|
||||
return;
|
||||
}
|
||||
if (clamp.U32() != 3 || segmentation_mask.U32() != 28) {
|
||||
return;
|
||||
}
|
||||
if (swizzle_value == 0x99) {
|
||||
// DPdxFine
|
||||
if (lane_mask == 1) {
|
||||
if (index.U32() == 1) {
|
||||
IR::IREmitter ir{block, IR::Block::InstructionList::s_iterator_to(inst)};
|
||||
inst.ReplaceUsesWith(ir.DPdxFine(IR::F32{inst.Arg(1)}));
|
||||
}
|
||||
} else if (swizzle_value == 0xA5) {
|
||||
// DPdyFine
|
||||
if (lane_mask == 2) {
|
||||
if (index.U32() == 2) {
|
||||
IR::IREmitter ir{block, IR::Block::InstructionList::s_iterator_to(inst)};
|
||||
inst.ReplaceUsesWith(ir.DPdyFine(IR::F32{inst.Arg(1)}));
|
||||
}
|
||||
@@ -724,13 +709,6 @@ bool FindGradient3DDerivatives(std::array<IR::Value, 3>& results, IR::Value coor
|
||||
const auto check_through_shuffle = [](IR::Value input, IR::Value& result) {
|
||||
const IR::Value value_1{GetThroughCast(input.Resolve(), IR::Opcode::BitCastF32U32)};
|
||||
IR::Inst* const inst2{value_1.InstRecursive()};
|
||||
if (inst2->GetOpcode() == IR::Opcode::QuadBroadcast) {
|
||||
if (!inst2->Arg(1).Resolve().IsImmediate()) {
|
||||
return false;
|
||||
}
|
||||
result = GetThroughCast(inst2->Arg(0).Resolve(), IR::Opcode::BitCastU32F32);
|
||||
return true;
|
||||
}
|
||||
if (inst2->GetOpcode() != IR::Opcode::ShuffleIndex) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -1,129 +0,0 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#include <boost/container/small_vector.hpp>
|
||||
|
||||
#include "shader_recompiler/frontend/ir/ir_emitter.h"
|
||||
#include "shader_recompiler/host_translate_info.h"
|
||||
#include "shader_recompiler/ir_opt/passes.h"
|
||||
|
||||
namespace Shader::Optimization {
|
||||
namespace {
|
||||
constexpr int MAX_BALLOT_DEPTH = 8;
|
||||
|
||||
struct SlotAtomic {
|
||||
IR::Block* block;
|
||||
IR::Inst* atomic;
|
||||
IR::Inst* ballot;
|
||||
};
|
||||
|
||||
IR::Inst* FindBallot(const IR::Value& value, int depth) {
|
||||
if (depth > MAX_BALLOT_DEPTH || value.IsImmediate()) {
|
||||
return nullptr;
|
||||
}
|
||||
IR::Inst* const inst{value.InstRecursive()};
|
||||
if (inst->GetOpcode() == IR::Opcode::SubgroupBallot) {
|
||||
return inst;
|
||||
}
|
||||
if (inst->GetOpcode() == IR::Opcode::Phi) {
|
||||
return nullptr;
|
||||
}
|
||||
for (size_t index = 0; index < inst->NumArgs(); ++index) {
|
||||
if (IR::Inst* const ballot{FindBallot(inst->Arg(index), depth + 1)}) {
|
||||
return ballot;
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
bool IsShuffle(IR::Opcode opcode) {
|
||||
switch (opcode) {
|
||||
case IR::Opcode::ShuffleIndex:
|
||||
case IR::Opcode::ShuffleUp:
|
||||
case IR::Opcode::ShuffleDown:
|
||||
case IR::Opcode::ShuffleButterfly:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
IR::U32 PrimitiveSlot(IR::IREmitter& ir, u32 invocations, const IR::U32& amount) {
|
||||
IR::U32 key{ir.GetAttributeU32(IR::Attribute::PrimitiveId)};
|
||||
if (invocations > 1) {
|
||||
key = ir.IAdd(ir.IMul(key, ir.Imm32(invocations)), ir.InvocationId());
|
||||
}
|
||||
return ir.IMul(key, amount);
|
||||
}
|
||||
|
||||
bool MergesAtomic(const IR::Inst& phi, const IR::Block& block, const IR::Inst& atomic) {
|
||||
for (size_t index = 0; index < phi.NumArgs(); ++index) {
|
||||
const IR::Value arg{phi.Arg(index)};
|
||||
if (phi.PhiBlock(index) == &block && !arg.IsImmediate() &&
|
||||
arg.InstRecursive() == &atomic) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void RewriteMergedSlots(IR::Block& block, IR::Inst& atomic, u32 invocations) {
|
||||
for (IR::Block* const successor : block.ImmSuccessors()) {
|
||||
for (IR::Inst& phi : successor->Instructions()) {
|
||||
if (phi.GetOpcode() != IR::Opcode::Phi || !MergesAtomic(phi, block, atomic)) {
|
||||
continue;
|
||||
}
|
||||
for (size_t index = 0; index < phi.NumArgs(); ++index) {
|
||||
const IR::Value arg{phi.Arg(index)};
|
||||
if (arg.IsImmediate() || !IsShuffle(arg.InstRecursive()->GetOpcode())) {
|
||||
continue;
|
||||
}
|
||||
IR::IREmitter ir{*phi.PhiBlock(index)};
|
||||
const IR::U32 amount{arg.InstRecursive()->Arg(0)};
|
||||
phi.SetArg(index, PrimitiveSlot(ir, invocations, amount));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void RewriteAtomic(IR::Block& block, IR::Inst& atomic, u32 invocations) {
|
||||
const auto insert_point{IR::Block::InstructionList::s_iterator_to(atomic)};
|
||||
IR::IREmitter ir{block, insert_point};
|
||||
const IR::U32 amount{atomic.Arg(2)};
|
||||
const IR::U32 slot{PrimitiveSlot(ir, invocations, amount)};
|
||||
block.PrependNewInst(insert_point, IR::Opcode::StorageAtomicUMax32,
|
||||
{atomic.Arg(0), atomic.Arg(1), ir.IAdd(slot, amount)});
|
||||
atomic.ReplaceUsesWith(slot);
|
||||
}
|
||||
|
||||
void NeutralizePredicate(IR::Inst& ballot) {
|
||||
const IR::Value pred{ballot.Arg(0)};
|
||||
if (!pred.IsImmediate()) {
|
||||
pred.InstRecursive()->ReplaceUsesWith(IR::Value{true});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void GeometryCompactionPass(IR::Program& program, const HostTranslateInfo& host_info) {
|
||||
if (program.stage != Stage::Geometry || !host_info.single_lane_geometry_subgroups) {
|
||||
return;
|
||||
}
|
||||
boost::container::small_vector<SlotAtomic, 4> slot_atomics;
|
||||
for (IR::Block* const block : program.post_order_blocks) {
|
||||
for (IR::Inst& inst : block->Instructions()) {
|
||||
if (inst.GetOpcode() != IR::Opcode::StorageAtomicIAdd32) {
|
||||
continue;
|
||||
}
|
||||
if (IR::Inst* const ballot{FindBallot(inst.Arg(2), 0)}) {
|
||||
slot_atomics.push_back({block, &inst, ballot});
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const SlotAtomic& slot_atomic : slot_atomics) {
|
||||
RewriteMergedSlots(*slot_atomic.block, *slot_atomic.atomic, program.invocations);
|
||||
RewriteAtomic(*slot_atomic.block, *slot_atomic.atomic, program.invocations);
|
||||
NeutralizePredicate(*slot_atomic.ballot);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
@@ -1,6 +1,3 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
@@ -19,7 +16,6 @@ void CollectShaderInfoPass(Environment& env, IR::Program& program);
|
||||
void ConditionalBarrierPass(IR::Program& program);
|
||||
void ConstantPropagationPass(Environment& env, IR::Program& program);
|
||||
void DeadCodeEliminationPass(IR::Program& program);
|
||||
void GeometryCompactionPass(IR::Program& program, const HostTranslateInfo& host_info);
|
||||
void GlobalMemoryToStorageBufferPass(IR::Program& program, const HostTranslateInfo& host_info);
|
||||
void IdentityRemovalPass(IR::Program& program);
|
||||
void LowerFp64ToFp32(IR::Program& program);
|
||||
|
||||
@@ -40,7 +40,6 @@ struct Profile {
|
||||
bool support_shader_quad_control{};
|
||||
bool support_quad_shuffles{};
|
||||
bool support_vote{};
|
||||
bool support_shuffle_relative{};
|
||||
u32 supported_subgroup_stages{0x7F};
|
||||
bool support_viewport_index_layer_non_geometry{};
|
||||
bool support_viewport_mask{};
|
||||
@@ -103,8 +102,6 @@ struct Profile {
|
||||
bool ignore_nan_fp_comparisons{};
|
||||
/// Some drivers have broken support for OpVectorExtractDynamic on subgroup mask inputs
|
||||
bool has_broken_spirv_subgroup_mask_vector_extract_dynamic{};
|
||||
bool has_broken_spirv_subgroup_shuffle{};
|
||||
u32 max_subgroup_size{};
|
||||
|
||||
u32 gl_max_compute_smem_size{};
|
||||
|
||||
|
||||
@@ -217,13 +217,15 @@ void RendererVulkan::Composite(std::span<const Tegra::FramebufferConfig> framebu
|
||||
|
||||
scheduler.RequestOutsideRenderPassOperationContext();
|
||||
blit_swapchain.DrawToFrame(device, rasterizer, frame, framebuffers,
|
||||
render_window.GetFramebufferLayout(), swapchain.GetImageCount(),
|
||||
render_window.GetFramebufferLayout(),
|
||||
present_manager.SwapchainImageCount(),
|
||||
swapchain.GetImageViewFormat());
|
||||
|
||||
#ifdef HAS_LSFG
|
||||
void(frame_gen.WantedGenerations(present_manager.MaxExtraFrames()));
|
||||
|
||||
frame_gen.Process(device, frame, swapchain.GetImageFormat(), GuestExtent(framebuffers));
|
||||
frame_gen.Process(device, frame, present_manager.SwapchainImageFormat(),
|
||||
GuestExtent(framebuffers));
|
||||
|
||||
const size_t generated_frames = frame_gen.GeneratedFrameCount();
|
||||
for (size_t generation = 0; generation < generated_frames; ++generation) {
|
||||
|
||||
@@ -362,7 +362,7 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
const auto subgroup_stage_bit{[subgroup_stages](VkShaderStageFlags flag, Shader::Stage stage) {
|
||||
return (subgroup_stages & flag) != 0 ? (1u << static_cast<u32>(stage)) : 0u;
|
||||
}};
|
||||
u32 supported_subgroup_stages{
|
||||
const u32 supported_subgroup_stages{
|
||||
subgroup_stage_bit(VK_SHADER_STAGE_VERTEX_BIT, Shader::Stage::VertexA) |
|
||||
subgroup_stage_bit(VK_SHADER_STAGE_VERTEX_BIT, Shader::Stage::VertexB) |
|
||||
subgroup_stage_bit(VK_SHADER_STAGE_TESSELLATION_CONTROL_BIT,
|
||||
@@ -372,9 +372,6 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
subgroup_stage_bit(VK_SHADER_STAGE_GEOMETRY_BIT, Shader::Stage::Geometry) |
|
||||
subgroup_stage_bit(VK_SHADER_STAGE_FRAGMENT_BIT, Shader::Stage::Fragment) |
|
||||
subgroup_stage_bit(VK_SHADER_STAGE_COMPUTE_BIT, Shader::Stage::Compute)};
|
||||
if (driver_id == VK_DRIVER_ID_MESA_TURNIP) {
|
||||
supported_subgroup_stages &= ~(1u << static_cast<u32>(Shader::Stage::Geometry));
|
||||
}
|
||||
profile = Shader::Profile{
|
||||
.supported_spirv = device.SupportedSpirvVersion(),
|
||||
.unified_descriptor_binding = true,
|
||||
@@ -412,8 +409,6 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
.support_shader_quad_control = device.IsKhrShaderQuadControlSupported(),
|
||||
.support_quad_shuffles = device.IsSubgroupFeatureSupported(VK_SUBGROUP_FEATURE_QUAD_BIT),
|
||||
.support_vote = device.IsSubgroupFeatureSupported(VK_SUBGROUP_FEATURE_VOTE_BIT),
|
||||
.support_shuffle_relative =
|
||||
device.IsSubgroupFeatureSupported(VK_SUBGROUP_FEATURE_SHUFFLE_RELATIVE_BIT),
|
||||
.supported_subgroup_stages = supported_subgroup_stages,
|
||||
.support_viewport_index_layer_non_geometry =
|
||||
device.IsExtShaderViewportIndexLayerSupported(),
|
||||
@@ -449,17 +444,13 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
driver_id == VK_DRIVER_ID_INTEL_OPEN_SOURCE_MESA,
|
||||
|
||||
.has_broken_spirv_clamp = driver_id == VK_DRIVER_ID_INTEL_PROPRIETARY_WINDOWS,
|
||||
.has_broken_spirv_position_input = driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY,
|
||||
.has_broken_spirv_position_input = driver_id == false,
|
||||
.has_broken_unsigned_image_offsets = false,
|
||||
.has_broken_signed_operations = false,
|
||||
.has_broken_fp16_float_controls = driver_id == VK_DRIVER_ID_NVIDIA_PROPRIETARY,
|
||||
.has_broken_fp32_denorm_flush = driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY,
|
||||
.ignore_nan_fp_comparisons = false,
|
||||
.has_broken_spirv_subgroup_mask_vector_extract_dynamic =
|
||||
driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY &&
|
||||
device.GetDriverVersion() < VK_MAKE_VERSION(512, 672, 0),
|
||||
.has_broken_spirv_subgroup_shuffle = driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY,
|
||||
.max_subgroup_size = device.GetMaxSubgroupSize(),
|
||||
.has_broken_spirv_subgroup_mask_vector_extract_dynamic = false,
|
||||
.has_broken_robust =
|
||||
device.IsNvidia() && device.GetNvidiaArch() <= NvidiaArchitecture::Arch_Pascal,
|
||||
.min_ssbo_alignment = device.GetStorageBufferAlignment(),
|
||||
@@ -486,10 +477,8 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
driver_id == VK_DRIVER_ID_SAMSUNG_PROPRIETARY,
|
||||
.support_snorm_render_buffer = true,
|
||||
.support_viewport_index_layer = device.IsExtShaderViewportIndexLayerSupported(),
|
||||
.support_viewport_mask = device.IsNvViewportArray2Supported(),
|
||||
.support_geometry_shader_passthrough = device.IsNvGeometryShaderPassthroughSupported(),
|
||||
.support_conditional_barrier = device.SupportsConditionalBarriers(),
|
||||
.single_lane_geometry_subgroups = !profile.SupportsSubgroupStage(Shader::Stage::Geometry),
|
||||
};
|
||||
host_info.ApplyDescriptorLimitPolicy();
|
||||
|
||||
|
||||
@@ -23,6 +23,7 @@ namespace Vulkan {
|
||||
namespace {
|
||||
|
||||
constexpr size_t MAX_FRAMES_IN_FLIGHT = 7;
|
||||
constexpr u32 MAX_PRESENT_ATTEMPTS = 3;
|
||||
#ifdef HAS_LSFG
|
||||
static_assert(MAX_FRAMES_IN_FLIGHT <= LSFG_MAX_TARGETS);
|
||||
#endif
|
||||
@@ -372,12 +373,33 @@ void PresentManager::SetImageCount() {
|
||||
#else
|
||||
image_count = std::min<size_t>(swapchain.GetImageCount(), MAX_FRAMES_IN_FLIGHT);
|
||||
#endif
|
||||
swapchain_image_count = swapchain.GetImageCount();
|
||||
swapchain_image_format = swapchain.GetImageFormat();
|
||||
}
|
||||
|
||||
void PresentManager::DiscardFrame(Frame* frame) {
|
||||
static constexpr VkPipelineStageFlags wait_stage = VK_PIPELINE_STAGE_TOP_OF_PIPE_BIT;
|
||||
const VkSemaphore render_ready = *frame->render_ready;
|
||||
const VkSubmitInfo submit_info{
|
||||
.sType = VK_STRUCTURE_TYPE_SUBMIT_INFO,
|
||||
.pNext = nullptr,
|
||||
.waitSemaphoreCount = 1U,
|
||||
.pWaitSemaphores = &render_ready,
|
||||
.pWaitDstStageMask = &wait_stage,
|
||||
.commandBufferCount = 0U,
|
||||
.pCommandBuffers = nullptr,
|
||||
.signalSemaphoreCount = 0U,
|
||||
.pSignalSemaphores = nullptr,
|
||||
};
|
||||
|
||||
std::scoped_lock submit_lock{scheduler.submit_mutex};
|
||||
void(device.GetGraphicsQueue().Submit(submit_info, *frame->present_done));
|
||||
}
|
||||
|
||||
void PresentManager::CopyToSwapchain(Frame* frame) {
|
||||
bool requires_recreation = false;
|
||||
|
||||
while (true) {
|
||||
for (u32 attempt = 0; attempt < MAX_PRESENT_ATTEMPTS; ++attempt) {
|
||||
try {
|
||||
// Recreate surface and swapchain if needed.
|
||||
if (requires_recreation) {
|
||||
@@ -397,6 +419,8 @@ void PresentManager::CopyToSwapchain(Frame* frame) {
|
||||
requires_recreation = true;
|
||||
}
|
||||
}
|
||||
|
||||
DiscardFrame(frame);
|
||||
}
|
||||
|
||||
void PresentManager::CopyToSwapchainImpl(Frame* frame) {
|
||||
|
||||
@@ -6,6 +6,7 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <atomic>
|
||||
#include <condition_variable>
|
||||
#include <mutex>
|
||||
#include <boost/container/deque.hpp>
|
||||
@@ -68,6 +69,9 @@ public:
|
||||
/// How many additional frames can be queued without stalling the render thread
|
||||
[[nodiscard]] size_t MaxExtraFrames() const;
|
||||
|
||||
[[nodiscard]] std::size_t SwapchainImageCount() const { return swapchain_image_count; }
|
||||
[[nodiscard]] VkFormat SwapchainImageFormat() const { return swapchain_image_format; }
|
||||
|
||||
private:
|
||||
void PresentThread(std::stop_token token);
|
||||
|
||||
@@ -77,6 +81,8 @@ private:
|
||||
|
||||
void RecreateSwapchain(Frame* frame);
|
||||
|
||||
void DiscardFrame(Frame* frame);
|
||||
|
||||
void SetImageCount();
|
||||
|
||||
private:
|
||||
@@ -100,7 +106,9 @@ private:
|
||||
bool blit_supported;
|
||||
bool storage_supported;
|
||||
bool use_present_thread;
|
||||
std::size_t image_count{};
|
||||
std::atomic<std::size_t> image_count{};
|
||||
std::atomic<std::size_t> swapchain_image_count{};
|
||||
std::atomic<VkFormat> swapchain_image_format{};
|
||||
};
|
||||
|
||||
} // namespace Vulkan
|
||||
|
||||
@@ -483,11 +483,6 @@ FN_MAX_LIMIT_LIST
|
||||
return properties.subgroup_properties.supportedStages;
|
||||
}
|
||||
|
||||
u32 GetMaxSubgroupSize() const {
|
||||
return (std::max)(properties.subgroup_properties.subgroupSize,
|
||||
properties.subgroup_size_control.maxSubgroupSize);
|
||||
}
|
||||
|
||||
/// Returns the maximum number of push descriptors.
|
||||
u32 MaxPushDescriptors() const {
|
||||
return properties.push_descriptor.maxPushDescriptors;
|
||||
|
||||
Reference in New Issue
Block a user