mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-10-06 06:00:12 +00:00
Compare commits
3 Commits
| Author | SHA1 | Date | |
|---|---|---|---|
| cf1d905786 | |||
| 60003ea642 | |||
| 43607d0b7e |
+2
-2
@@ -82,8 +82,8 @@ cmake_dependent_option(YUZU_USE_BUNDLED_QT "Download bundled Qt binaries" "${MSV
|
||||
option(ENABLE_DEBUG_TOOLS "Enable debugging tools (maxwell disassembler, SPIRV translator, etc)" OFF)
|
||||
option(ENABLE_WERROR "Enable -Werror diagnostics" ON)
|
||||
|
||||
# Lossless Scaling frame generation.
|
||||
option(ENABLE_LSFG "Enable Lossless Scaling frame generation" ON)
|
||||
# Lossless Scaling frame generation. Only Android.
|
||||
cmake_dependent_option(ENABLE_LSFG "Enable Lossless Scaling frame generation" ON "ANDROID" OFF)
|
||||
|
||||
option(ENABLE_RESHADE "Enable ReShade FX post-processing effects" ON)
|
||||
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package org.yuzu.yuzu_emu.adapters
|
||||
|
||||
+2
-4
@@ -20,21 +20,18 @@ enum class BooleanSetting(override val key: String) : AbstractBooleanSetting {
|
||||
EMULATE_BGR565("emulate_bgr565"),
|
||||
RESCALE_HACK("rescale_hack"),
|
||||
CPUOPT_UNSAFE_HOST_MMU("cpuopt_unsafe_host_mmu"),
|
||||
RELOCATE_36BIT_ADDRESS_SPACE("relocate_36bit_address_space"),
|
||||
USE_DOCKED_MODE("use_docked_mode"),
|
||||
USE_AUTO_STUB("use_auto_stub"),
|
||||
RENDERER_USE_DISK_SHADER_CACHE("use_disk_shader_cache"),
|
||||
RENDERER_FORCE_MAX_CLOCK("force_max_clock"),
|
||||
RENDERER_ASYNCHRONOUS_GPU_EMULATION("use_asynchronous_gpu_emulation"),
|
||||
RENDERER_ASYNC_PRESENTATION("async_presentation"),
|
||||
RENDERER_ASYNCHRONOUS_SHADERS("use_asynchronous_shaders"),
|
||||
RENDERER_REACTIVE_FLUSHING("use_reactive_flushing"),
|
||||
RENDERER_BARRIER_FEEDBACK_LOOPS("barrier_feedback_loops"),
|
||||
ENABLE_BUFFER_HISTORY("enable_buffer_history"),
|
||||
USE_OPTIMIZED_VERTEX_BUFFERS("use_optimized_vertex_buffers"),
|
||||
ENABLE_GPU_BUFFER_READBACK("enable_gpu_buffer_readback"),
|
||||
USE_UNIFIED_MEMORY("use_unified_memory"),
|
||||
SYNC_MEMORY_OPERATIONS("sync_memory_operations"),
|
||||
STALL_ON_GPU_FENCE_WAIT("stall_on_gpu_fence_wait"),
|
||||
BUFFER_REORDER_DISABLE("disable_buffer_reorder"),
|
||||
RENDERER_DEBUG("debug"),
|
||||
RENDERER_PATCH_OLD_QCOM_DRIVERS("patch_old_qcom_drivers"),
|
||||
@@ -42,6 +39,7 @@ enum class BooleanSetting(override val key: String) : AbstractBooleanSetting {
|
||||
RENDERER_SAMPLE_SHADING("sample_shading"),
|
||||
RENDERER_FRAME_GEN("frame_gen"),
|
||||
RENDERER_FRAME_GEN_FLOW_SCALE_AUTO("frame_gen_flow_scale_auto"),
|
||||
GPU_UNSWIZZLE_ENABLED("gpu_unswizzle_enabled"),
|
||||
PICTURE_IN_PICTURE("picture_in_picture"),
|
||||
USE_CUSTOM_RTC("custom_rtc_enabled"),
|
||||
BLACK_BACKGROUNDS("black_backgrounds"),
|
||||
|
||||
@@ -32,6 +32,7 @@ enum class IntSetting(override val key: String) : AbstractIntSetting {
|
||||
RENDERER_DYNA_STATE("dyna_state"),
|
||||
DMA_ACCURACY("dma_accuracy"),
|
||||
GPU_FENCE_BEHAVIOR("gpu_fence_behavior"),
|
||||
FRAME_PACING_MODE("frame_pacing_mode"),
|
||||
AUDIO_OUTPUT_ENGINE("output_engine"),
|
||||
MAX_ANISOTROPY("max_anisotropy"),
|
||||
THEME("theme"),
|
||||
@@ -50,6 +51,9 @@ enum class IntSetting(override val key: String) : AbstractIntSetting {
|
||||
FAST_CPU_TIME("fast_cpu_time"),
|
||||
CPU_TICKS("cpu_ticks"),
|
||||
FAST_GPU_TIME("fast_gpu_time"),
|
||||
GPU_UNSWIZZLE_TEXTURE_SIZE("gpu_unswizzle_texture_size"),
|
||||
GPU_UNSWIZZLE_STREAM_SIZE("gpu_unswizzle_stream_size"),
|
||||
GPU_UNSWIZZLE_CHUNK_SIZE("gpu_unswizzle_chunk_size"),
|
||||
BAT_TEMPERATURE_UNIT("bat_temperature_unit"),
|
||||
CABINET_APPLET("cabinet_applet_mode"),
|
||||
CONTROLLER_APPLET("controller_applet_mode"),
|
||||
|
||||
+89
@@ -0,0 +1,89 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package org.yuzu.yuzu_emu.features.settings.model.view
|
||||
|
||||
import androidx.annotation.ArrayRes
|
||||
import androidx.annotation.StringRes
|
||||
import org.yuzu.yuzu_emu.features.settings.model.AbstractSetting
|
||||
import org.yuzu.yuzu_emu.features.settings.model.BooleanSetting
|
||||
import org.yuzu.yuzu_emu.features.settings.model.IntSetting
|
||||
|
||||
class GpuUnswizzleSetting(
|
||||
@StringRes titleId: Int = 0,
|
||||
titleString: String = "",
|
||||
@StringRes descriptionId: Int = 0,
|
||||
descriptionString: String = "",
|
||||
@ArrayRes val textureSizeChoicesId: Int,
|
||||
@ArrayRes val textureSizeValuesId: Int,
|
||||
@ArrayRes val streamSizeChoicesId: Int,
|
||||
@ArrayRes val streamSizeValuesId: Int,
|
||||
@ArrayRes val chunkSizeChoicesId: Int,
|
||||
@ArrayRes val chunkSizeValuesId: Int
|
||||
) : SettingsItem(
|
||||
object : AbstractSetting {
|
||||
override val key: String = SettingsItem.GPU_UNSWIZZLE_COMBINED
|
||||
override val defaultValue: Any = false
|
||||
override val isSaveable = true
|
||||
override val isRuntimeModifiable = true
|
||||
override val isSwitchable = true
|
||||
override val pairedSettingKey: String = ""
|
||||
override var global: Boolean
|
||||
get() {
|
||||
return BooleanSetting.GPU_UNSWIZZLE_ENABLED.global &&
|
||||
IntSetting.GPU_UNSWIZZLE_TEXTURE_SIZE.global &&
|
||||
IntSetting.GPU_UNSWIZZLE_STREAM_SIZE.global &&
|
||||
IntSetting.GPU_UNSWIZZLE_CHUNK_SIZE.global
|
||||
}
|
||||
set(value) {
|
||||
BooleanSetting.GPU_UNSWIZZLE_ENABLED.global = value
|
||||
IntSetting.GPU_UNSWIZZLE_TEXTURE_SIZE.global = value
|
||||
IntSetting.GPU_UNSWIZZLE_STREAM_SIZE.global = value
|
||||
IntSetting.GPU_UNSWIZZLE_CHUNK_SIZE.global = value
|
||||
}
|
||||
override fun getValueAsString(needsGlobal: Boolean): String = "combined"
|
||||
override fun reset() {
|
||||
BooleanSetting.GPU_UNSWIZZLE_ENABLED.reset()
|
||||
IntSetting.GPU_UNSWIZZLE_TEXTURE_SIZE.reset()
|
||||
IntSetting.GPU_UNSWIZZLE_STREAM_SIZE.reset()
|
||||
IntSetting.GPU_UNSWIZZLE_CHUNK_SIZE.reset()
|
||||
}
|
||||
},
|
||||
titleId,
|
||||
titleString,
|
||||
descriptionId,
|
||||
descriptionString
|
||||
) {
|
||||
override val type = SettingsItem.TYPE_GPU_UNSWIZZLE
|
||||
|
||||
// Check if GPU unswizzle is enabled via the dedicated boolean setting
|
||||
fun isEnabled(needsGlobal: Boolean = false): Boolean =
|
||||
BooleanSetting.GPU_UNSWIZZLE_ENABLED.getBoolean(needsGlobal)
|
||||
|
||||
fun setEnabled(value: Boolean) =
|
||||
BooleanSetting.GPU_UNSWIZZLE_ENABLED.setBoolean(value)
|
||||
|
||||
fun enable() = setEnabled(true)
|
||||
|
||||
fun disable() = setEnabled(false)
|
||||
|
||||
fun getTextureSize(needsGlobal: Boolean = false): Int =
|
||||
IntSetting.GPU_UNSWIZZLE_TEXTURE_SIZE.getInt(needsGlobal)
|
||||
|
||||
fun setTextureSize(value: Int) =
|
||||
IntSetting.GPU_UNSWIZZLE_TEXTURE_SIZE.setInt(value)
|
||||
|
||||
fun getStreamSize(needsGlobal: Boolean = false): Int =
|
||||
IntSetting.GPU_UNSWIZZLE_STREAM_SIZE.getInt(needsGlobal)
|
||||
|
||||
fun setStreamSize(value: Int) =
|
||||
IntSetting.GPU_UNSWIZZLE_STREAM_SIZE.setInt(value)
|
||||
|
||||
fun getChunkSize(needsGlobal: Boolean = false): Int =
|
||||
IntSetting.GPU_UNSWIZZLE_CHUNK_SIZE.getInt(needsGlobal)
|
||||
|
||||
fun setChunkSize(value: Int) =
|
||||
IntSetting.GPU_UNSWIZZLE_CHUNK_SIZE.setInt(value)
|
||||
|
||||
fun reset() = setting.reset()
|
||||
}
|
||||
+48
-28
@@ -141,12 +141,14 @@ abstract class SettingsItem(
|
||||
const val TYPE_SPINBOX = 12
|
||||
const val TYPE_LAUNCHABLE = 13
|
||||
const val TYPE_PATH = 14
|
||||
const val TYPE_GPU_UNSWIZZLE = 15
|
||||
const val TYPE_FX_TOOLBAR = 16
|
||||
const val TYPE_FX_PRESET = 17
|
||||
const val TYPE_FX_SHADER = 18
|
||||
const val TYPE_FX_BUTTON = 19
|
||||
|
||||
const val FASTMEM_COMBINED = "fastmem_combined"
|
||||
const val GPU_UNSWIZZLE_COMBINED = "gpu_unswizzle_combined"
|
||||
|
||||
val emptySetting = object : AbstractSetting {
|
||||
override val key: String = ""
|
||||
@@ -751,6 +753,13 @@ abstract class SettingsItem(
|
||||
descriptionId = R.string.renderer_asynchronous_gpu_emulation_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.RENDERER_ASYNC_PRESENTATION,
|
||||
titleId = R.string.renderer_async_presentation,
|
||||
descriptionId = R.string.renderer_async_presentation_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SingleChoiceSetting(
|
||||
IntSetting.DMA_ACCURACY,
|
||||
@@ -785,6 +794,45 @@ abstract class SettingsItem(
|
||||
valuesId = R.array.gpuValues
|
||||
)
|
||||
)
|
||||
put(
|
||||
SingleChoiceSetting(
|
||||
IntSetting.GPU_UNSWIZZLE_TEXTURE_SIZE,
|
||||
titleId = R.string.gpu_unswizzle_texture_size,
|
||||
descriptionId = R.string.gpu_unswizzle_texture_size_description,
|
||||
choicesId = R.array.gpuTextureSizeSwizzleEntries,
|
||||
valuesId = R.array.gpuTextureSizeSwizzleValues
|
||||
)
|
||||
)
|
||||
put(
|
||||
SingleChoiceSetting(
|
||||
IntSetting.GPU_UNSWIZZLE_STREAM_SIZE,
|
||||
titleId = R.string.gpu_unswizzle_stream_size,
|
||||
descriptionId = R.string.gpu_unswizzle_stream_size_description,
|
||||
choicesId = R.array.gpuSwizzleEntries,
|
||||
valuesId = R.array.gpuSwizzleValues
|
||||
)
|
||||
)
|
||||
put(
|
||||
SingleChoiceSetting(
|
||||
IntSetting.GPU_UNSWIZZLE_CHUNK_SIZE,
|
||||
titleId = R.string.gpu_unswizzle_chunk_size,
|
||||
descriptionId = R.string.gpu_unswizzle_chunk_size_description,
|
||||
choicesId = R.array.gpuSwizzleChunkEntries,
|
||||
valuesId = R.array.gpuSwizzleChunkValues
|
||||
)
|
||||
)
|
||||
put(
|
||||
GpuUnswizzleSetting(
|
||||
titleId = R.string.gpu_unswizzle_settings,
|
||||
descriptionId = R.string.gpu_unswizzle_settings_description,
|
||||
textureSizeChoicesId = R.array.gpuTextureSizeSwizzleEntries,
|
||||
textureSizeValuesId = R.array.gpuTextureSizeSwizzleValues,
|
||||
streamSizeChoicesId = R.array.gpuSwizzleEntries,
|
||||
streamSizeValuesId = R.array.gpuSwizzleValues,
|
||||
chunkSizeChoicesId = R.array.gpuSwizzleChunkEntries,
|
||||
chunkSizeValuesId = R.array.gpuSwizzleChunkValues
|
||||
)
|
||||
)
|
||||
put(
|
||||
SingleChoiceSetting(
|
||||
IntSetting.FAST_CPU_TIME,
|
||||
@@ -846,13 +894,6 @@ abstract class SettingsItem(
|
||||
descriptionId = R.string.cpuopt_unsafe_host_mmu_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.RELOCATE_36BIT_ADDRESS_SPACE,
|
||||
titleId = R.string.relocate_36bit_address_space,
|
||||
descriptionId = R.string.relocate_36bit_address_space_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.RENDERER_REACTIVE_FLUSHING,
|
||||
@@ -860,13 +901,6 @@ abstract class SettingsItem(
|
||||
descriptionId = R.string.renderer_reactive_flushing_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.RENDERER_BARRIER_FEEDBACK_LOOPS,
|
||||
titleId = R.string.renderer_barrier_feedback_loops,
|
||||
descriptionId = R.string.renderer_barrier_feedback_loops_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.ENABLE_BUFFER_HISTORY,
|
||||
@@ -881,13 +915,6 @@ abstract class SettingsItem(
|
||||
descriptionId = R.string.enable_gpu_buffer_readback_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.USE_UNIFIED_MEMORY,
|
||||
titleId = R.string.use_unified_memory,
|
||||
descriptionId = R.string.use_unified_memory_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.USE_OPTIMIZED_VERTEX_BUFFERS,
|
||||
@@ -902,13 +929,6 @@ abstract class SettingsItem(
|
||||
descriptionId = R.string.sync_memory_operations_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.STALL_ON_GPU_FENCE_WAIT,
|
||||
titleId = R.string.stall_on_gpu_fence_wait,
|
||||
descriptionId = R.string.stall_on_gpu_fence_wait_description
|
||||
)
|
||||
)
|
||||
put(
|
||||
SwitchSetting(
|
||||
BooleanSetting.BUFFER_REORDER_DISABLE,
|
||||
|
||||
+206
@@ -0,0 +1,206 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package org.yuzu.yuzu_emu.features.settings.ui
|
||||
|
||||
import android.app.Dialog
|
||||
import android.content.DialogInterface
|
||||
import android.os.Bundle
|
||||
import android.view.LayoutInflater
|
||||
import android.widget.ArrayAdapter
|
||||
import androidx.fragment.app.DialogFragment
|
||||
import androidx.fragment.app.activityViewModels
|
||||
import com.google.android.material.dialog.MaterialAlertDialogBuilder
|
||||
import org.yuzu.yuzu_emu.R
|
||||
import org.yuzu.yuzu_emu.databinding.DialogGpuUnswizzleBinding
|
||||
import org.yuzu.yuzu_emu.features.settings.model.view.GpuUnswizzleSetting
|
||||
|
||||
class GpuUnswizzleDialogFragment : DialogFragment() {
|
||||
private var position = 0
|
||||
private val settingsViewModel: SettingsViewModel by activityViewModels()
|
||||
private lateinit var binding: DialogGpuUnswizzleBinding
|
||||
|
||||
override fun onCreate(savedInstanceState: Bundle?) {
|
||||
super.onCreate(savedInstanceState)
|
||||
position = requireArguments().getInt(POSITION)
|
||||
|
||||
if (settingsViewModel.clickedItem == null) dismiss()
|
||||
}
|
||||
|
||||
override fun onCreateDialog(savedInstanceState: Bundle?): Dialog {
|
||||
binding = DialogGpuUnswizzleBinding.inflate(LayoutInflater.from(requireContext()))
|
||||
val item = settingsViewModel.clickedItem as GpuUnswizzleSetting
|
||||
|
||||
// Setup texture size dropdown
|
||||
val textureSizeEntries = resources.getStringArray(item.textureSizeChoicesId)
|
||||
val textureSizeValues = resources.getIntArray(item.textureSizeValuesId)
|
||||
val textureSizeAdapter = ArrayAdapter(
|
||||
requireContext(),
|
||||
android.R.layout.simple_dropdown_item_1line,
|
||||
textureSizeEntries.toMutableList()
|
||||
)
|
||||
binding.dropdownTextureSize.setAdapter(textureSizeAdapter)
|
||||
|
||||
// Setup stream size dropdown
|
||||
val streamSizeEntries = resources.getStringArray(item.streamSizeChoicesId)
|
||||
val streamSizeValues = resources.getIntArray(item.streamSizeValuesId)
|
||||
val streamSizeAdapter = ArrayAdapter(
|
||||
requireContext(),
|
||||
android.R.layout.simple_dropdown_item_1line,
|
||||
streamSizeEntries.toMutableList()
|
||||
)
|
||||
binding.dropdownStreamSize.setAdapter(streamSizeAdapter)
|
||||
|
||||
// Setup chunk size dropdown
|
||||
val chunkSizeEntries = resources.getStringArray(item.chunkSizeChoicesId)
|
||||
val chunkSizeValues = resources.getIntArray(item.chunkSizeValuesId)
|
||||
val chunkSizeAdapter = ArrayAdapter(
|
||||
requireContext(),
|
||||
android.R.layout.simple_dropdown_item_1line,
|
||||
chunkSizeEntries.toMutableList()
|
||||
)
|
||||
binding.dropdownChunkSize.setAdapter(chunkSizeAdapter)
|
||||
|
||||
// Load current values
|
||||
val isEnabled = item.isEnabled()
|
||||
binding.switchEnable.isChecked = isEnabled
|
||||
|
||||
if (isEnabled) {
|
||||
val textureSizeIndex = textureSizeValues.indexOf(item.getTextureSize())
|
||||
if (textureSizeIndex >= 0) {
|
||||
binding.dropdownTextureSize.setText(textureSizeEntries[textureSizeIndex], false)
|
||||
}
|
||||
|
||||
val streamSizeIndex = streamSizeValues.indexOf(item.getStreamSize())
|
||||
if (streamSizeIndex >= 0) {
|
||||
binding.dropdownStreamSize.setText(streamSizeEntries[streamSizeIndex], false)
|
||||
}
|
||||
|
||||
val chunkSizeIndex = chunkSizeValues.indexOf(item.getChunkSize())
|
||||
if (chunkSizeIndex >= 0) {
|
||||
binding.dropdownChunkSize.setText(chunkSizeEntries[chunkSizeIndex], false)
|
||||
}
|
||||
} else {
|
||||
// Set default/recommended values when disabling
|
||||
binding.dropdownTextureSize.setText(textureSizeEntries[3], false)
|
||||
binding.dropdownStreamSize.setText(streamSizeEntries[3], false)
|
||||
binding.dropdownChunkSize.setText(chunkSizeEntries[3], false)
|
||||
}
|
||||
|
||||
// Clear adapter filters after setText to fix rotation bug
|
||||
textureSizeAdapter.filter.filter(null)
|
||||
streamSizeAdapter.filter.filter(null)
|
||||
chunkSizeAdapter.filter.filter(null)
|
||||
|
||||
// Enable/disable dropdowns based on switch state
|
||||
updateDropdownsState(isEnabled)
|
||||
binding.switchEnable.setOnCheckedChangeListener { _, checked ->
|
||||
updateDropdownsState(checked)
|
||||
}
|
||||
|
||||
val dialog = MaterialAlertDialogBuilder(requireContext())
|
||||
.setTitle(item.title)
|
||||
.setView(binding.root)
|
||||
.create()
|
||||
|
||||
// Setup button listeners
|
||||
binding.btnDefault.setOnClickListener {
|
||||
// Reset to defaults
|
||||
item.reset()
|
||||
// Refresh values with adapters reset
|
||||
val textureSizeIndex = textureSizeValues.indexOf(item.getTextureSize())
|
||||
if (textureSizeIndex >= 0) {
|
||||
binding.dropdownTextureSize.setText(textureSizeEntries[textureSizeIndex], false)
|
||||
}
|
||||
val streamSizeIndex = streamSizeValues.indexOf(item.getStreamSize())
|
||||
if (streamSizeIndex >= 0) {
|
||||
binding.dropdownStreamSize.setText(streamSizeEntries[streamSizeIndex], false)
|
||||
}
|
||||
val chunkSizeIndex = chunkSizeValues.indexOf(item.getChunkSize())
|
||||
if (chunkSizeIndex >= 0) {
|
||||
binding.dropdownChunkSize.setText(chunkSizeEntries[chunkSizeIndex], false)
|
||||
}
|
||||
// Clear filters
|
||||
textureSizeAdapter.filter.filter(null)
|
||||
streamSizeAdapter.filter.filter(null)
|
||||
chunkSizeAdapter.filter.filter(null)
|
||||
|
||||
settingsViewModel.setAdapterItemChanged(position)
|
||||
settingsViewModel.setShouldReloadSettingsList(true)
|
||||
}
|
||||
|
||||
binding.btnCancel.setOnClickListener {
|
||||
dialog.dismiss()
|
||||
}
|
||||
|
||||
binding.btnOk.setOnClickListener {
|
||||
if (binding.switchEnable.isChecked) {
|
||||
item.enable()
|
||||
// Save the selected values
|
||||
val selectedTextureIndex = textureSizeEntries.indexOf(
|
||||
binding.dropdownTextureSize.text.toString()
|
||||
)
|
||||
if (selectedTextureIndex >= 0) {
|
||||
item.setTextureSize(textureSizeValues[selectedTextureIndex])
|
||||
}
|
||||
|
||||
val selectedStreamIndex = streamSizeEntries.indexOf(
|
||||
binding.dropdownStreamSize.text.toString()
|
||||
)
|
||||
if (selectedStreamIndex >= 0) {
|
||||
item.setStreamSize(streamSizeValues[selectedStreamIndex])
|
||||
}
|
||||
|
||||
val selectedChunkIndex = chunkSizeEntries.indexOf(
|
||||
binding.dropdownChunkSize.text.toString()
|
||||
)
|
||||
if (selectedChunkIndex >= 0) {
|
||||
item.setChunkSize(chunkSizeValues[selectedChunkIndex])
|
||||
}
|
||||
} else {
|
||||
// Disable GPU unswizzle
|
||||
item.disable()
|
||||
}
|
||||
|
||||
settingsViewModel.setAdapterItemChanged(position)
|
||||
settingsViewModel.setShouldReloadSettingsList(true)
|
||||
dialog.dismiss()
|
||||
}
|
||||
|
||||
// Ensure filters are cleared after dialog is shown
|
||||
binding.root.post {
|
||||
textureSizeAdapter.filter.filter(null)
|
||||
streamSizeAdapter.filter.filter(null)
|
||||
chunkSizeAdapter.filter.filter(null)
|
||||
}
|
||||
|
||||
return dialog
|
||||
}
|
||||
|
||||
private fun updateDropdownsState(enabled: Boolean) {
|
||||
binding.layoutTextureSize.isEnabled = enabled
|
||||
binding.dropdownTextureSize.isEnabled = enabled
|
||||
binding.layoutStreamSize.isEnabled = enabled
|
||||
binding.dropdownStreamSize.isEnabled = enabled
|
||||
binding.layoutChunkSize.isEnabled = enabled
|
||||
binding.dropdownChunkSize.isEnabled = enabled
|
||||
}
|
||||
|
||||
companion object {
|
||||
const val TAG = "GpuUnswizzleDialogFragment"
|
||||
const val POSITION = "Position"
|
||||
|
||||
fun newInstance(
|
||||
settingsViewModel: SettingsViewModel,
|
||||
item: GpuUnswizzleSetting,
|
||||
position: Int
|
||||
): GpuUnswizzleDialogFragment {
|
||||
val dialog = GpuUnswizzleDialogFragment()
|
||||
val args = Bundle()
|
||||
args.putInt(POSITION, position)
|
||||
dialog.arguments = args
|
||||
settingsViewModel.clickedItem = item
|
||||
return dialog
|
||||
}
|
||||
}
|
||||
}
|
||||
+12
@@ -106,6 +106,10 @@ class SettingsAdapter(
|
||||
PathViewHolder(ListItemSettingBinding.inflate(inflater), this)
|
||||
}
|
||||
|
||||
SettingsItem.TYPE_GPU_UNSWIZZLE -> {
|
||||
GpuUnswizzleViewHolder(ListItemSettingBinding.inflate(inflater), this)
|
||||
}
|
||||
|
||||
SettingsItem.TYPE_FX_TOOLBAR -> {
|
||||
FxToolbarViewHolder(
|
||||
ListItemSettingFxToolbarBinding.inflate(inflater, parent, false),
|
||||
@@ -507,6 +511,14 @@ class SettingsAdapter(
|
||||
settingsViewModel.setShouldShowPathResetDialog(true)
|
||||
}
|
||||
|
||||
fun onGpuUnswizzleClick(item: GpuUnswizzleSetting, position: Int) {
|
||||
GpuUnswizzleDialogFragment.newInstance(
|
||||
settingsViewModel,
|
||||
item,
|
||||
position
|
||||
).show(fragment.childFragmentManager, GpuUnswizzleDialogFragment.TAG)
|
||||
}
|
||||
|
||||
private class DiffCallback : DiffUtil.ItemCallback<SettingsItem>() {
|
||||
override fun areItemsTheSame(oldItem: SettingsItem, newItem: SettingsItem): Boolean {
|
||||
return oldItem.setting.key == newItem.setting.key
|
||||
|
||||
+2
-4
@@ -545,14 +545,11 @@ class SettingsFragmentPresenter(
|
||||
add(IntSetting.RENDERER_NVDEC_EMULATION.key)
|
||||
|
||||
add(BooleanSetting.SYNC_MEMORY_OPERATIONS.key)
|
||||
add(BooleanSetting.STALL_ON_GPU_FENCE_WAIT.key)
|
||||
add(BooleanSetting.RENDERER_USE_DISK_SHADER_CACHE.key)
|
||||
add(BooleanSetting.RENDERER_FORCE_MAX_CLOCK.key)
|
||||
add(BooleanSetting.RENDERER_REACTIVE_FLUSHING.key)
|
||||
add(BooleanSetting.RENDERER_BARRIER_FEEDBACK_LOOPS.key)
|
||||
add(BooleanSetting.ENABLE_BUFFER_HISTORY.key)
|
||||
add(BooleanSetting.ENABLE_GPU_BUFFER_READBACK.key)
|
||||
add(BooleanSetting.USE_UNIFIED_MEMORY.key)
|
||||
add(BooleanSetting.USE_OPTIMIZED_VERTEX_BUFFERS.key)
|
||||
|
||||
add(HeaderSetting(R.string.hacks))
|
||||
@@ -563,6 +560,8 @@ class SettingsFragmentPresenter(
|
||||
add(BooleanSetting.RENDERER_ASYNCHRONOUS_SHADERS.key)
|
||||
add(IntSetting.ANDROID_PIPELINE_WORKERS.key)
|
||||
add(BooleanSetting.RENDERER_ASYNCHRONOUS_GPU_EMULATION.key)
|
||||
add(BooleanSetting.RENDERER_ASYNC_PRESENTATION.key)
|
||||
add(SettingsItem.GPU_UNSWIZZLE_COMBINED)
|
||||
|
||||
add(HeaderSetting(R.string.extensions))
|
||||
|
||||
@@ -1528,7 +1527,6 @@ class SettingsFragmentPresenter(
|
||||
add(HeaderSetting(R.string.cpu))
|
||||
|
||||
add(IntSetting.CPU_BACKEND.key)
|
||||
add(BooleanSetting.RELOCATE_36BIT_ADDRESS_SPACE.key)
|
||||
add(IntSetting.CPU_ACCURACY.key)
|
||||
add(BooleanSetting.USE_AUTO_STUB.key)
|
||||
add(SettingsItem.FASTMEM_COMBINED)
|
||||
|
||||
+71
@@ -0,0 +1,71 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
package org.yuzu.yuzu_emu.features.settings.ui.viewholder
|
||||
|
||||
import android.view.View
|
||||
import org.yuzu.yuzu_emu.R
|
||||
import org.yuzu.yuzu_emu.databinding.ListItemSettingBinding
|
||||
import org.yuzu.yuzu_emu.features.settings.model.view.GpuUnswizzleSetting
|
||||
import org.yuzu.yuzu_emu.features.settings.model.view.SettingsItem
|
||||
import org.yuzu.yuzu_emu.features.settings.ui.SettingsAdapter
|
||||
import org.yuzu.yuzu_emu.utils.ViewUtils.setVisible
|
||||
|
||||
class GpuUnswizzleViewHolder(val binding: ListItemSettingBinding, adapter: SettingsAdapter) :
|
||||
SettingViewHolder(binding.root, adapter) {
|
||||
private lateinit var setting: GpuUnswizzleSetting
|
||||
|
||||
override fun bind(item: SettingsItem) {
|
||||
setting = item as GpuUnswizzleSetting
|
||||
binding.textSettingName.text = setting.title
|
||||
binding.textSettingDescription.setVisible(item.description.isNotEmpty())
|
||||
binding.textSettingDescription.text = item.description
|
||||
|
||||
binding.textSettingValue.setVisible(true)
|
||||
val resMgr = binding.root.context.resources
|
||||
|
||||
if (setting.isEnabled()) {
|
||||
// Show a summary of current settings
|
||||
val textureSizeEntries = resMgr.getStringArray(setting.textureSizeChoicesId)
|
||||
val textureSizeValues = resMgr.getIntArray(setting.textureSizeValuesId)
|
||||
val textureSizeIndex = textureSizeValues.indexOf(setting.getTextureSize())
|
||||
val textureSizeLabel = if (textureSizeIndex >= 0) textureSizeEntries[textureSizeIndex] else "?"
|
||||
|
||||
val streamSizeEntries = resMgr.getStringArray(setting.streamSizeChoicesId)
|
||||
val streamSizeValues = resMgr.getIntArray(setting.streamSizeValuesId)
|
||||
val streamSizeIndex = streamSizeValues.indexOf(setting.getStreamSize())
|
||||
val streamSizeLabel = if (streamSizeIndex >= 0) streamSizeEntries[streamSizeIndex] else "?"
|
||||
|
||||
val chunkSizeEntries = resMgr.getStringArray(setting.chunkSizeChoicesId)
|
||||
val chunkSizeValues = resMgr.getIntArray(setting.chunkSizeValuesId)
|
||||
val chunkSizeIndex = chunkSizeValues.indexOf(setting.getChunkSize())
|
||||
val chunkSizeLabel = if (chunkSizeIndex >= 0) chunkSizeEntries[chunkSizeIndex] else "?"
|
||||
|
||||
binding.textSettingValue.text = "$textureSizeLabel ⋅ $streamSizeLabel ⋅ $chunkSizeLabel"
|
||||
} else {
|
||||
binding.textSettingValue.text = resMgr.getString(R.string.gpu_unswizzle_disabled)
|
||||
}
|
||||
|
||||
binding.buttonClear.setVisible(setting.clearable)
|
||||
binding.buttonClear.setOnClickListener {
|
||||
adapter.onClearClick(setting, bindingAdapterPosition)
|
||||
}
|
||||
|
||||
setStyle(setting.isEditable, binding)
|
||||
}
|
||||
|
||||
override fun onClick(clicked: View) {
|
||||
if (!setting.isEditable) {
|
||||
return
|
||||
}
|
||||
|
||||
adapter.onGpuUnswizzleClick(setting, bindingAdapterPosition)
|
||||
}
|
||||
|
||||
override fun onLongClick(clicked: View): Boolean {
|
||||
if (setting.isEditable) {
|
||||
return adapter.onLongClick(setting, bindingAdapterPosition)
|
||||
}
|
||||
return false
|
||||
}
|
||||
}
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: 2023 yuzu Emulator Project
|
||||
|
||||
@@ -64,8 +64,8 @@ void EmuWindow_Android::OnTouchReleased(int id) {
|
||||
EmulationSession::GetInstance().GetInputSubsystem().GetTouchScreen()->TouchReleased(id);
|
||||
}
|
||||
|
||||
void EmuWindow_Android::OnFrameDisplayed(u32 presented_frames) {
|
||||
UpdateObservedFrameRate(presented_frames);
|
||||
void EmuWindow_Android::OnFrameDisplayed() {
|
||||
UpdateObservedFrameRate();
|
||||
UpdateFrameRateHint();
|
||||
|
||||
if (!m_first_frame) {
|
||||
@@ -75,16 +75,15 @@ void EmuWindow_Android::OnFrameDisplayed(u32 presented_frames) {
|
||||
}
|
||||
}
|
||||
|
||||
void EmuWindow_Android::UpdateObservedFrameRate(u32 presented_frames) {
|
||||
m_presented_frames = static_cast<float>((std::max)(presented_frames, 1u));
|
||||
void EmuWindow_Android::UpdateObservedFrameRate() {
|
||||
const auto now = Clock::now();
|
||||
if (m_last_frame_display_time.time_since_epoch().count() != 0) {
|
||||
const auto frame_time = std::chrono::duration<float>(now - m_last_frame_display_time);
|
||||
const float seconds = frame_time.count();
|
||||
if (seconds > 0.0f) {
|
||||
const float instantaneous_rate = m_presented_frames / seconds;
|
||||
const float instantaneous_rate = 1.0f / seconds;
|
||||
if (std::isfinite(instantaneous_rate) && instantaneous_rate >= 1.0f &&
|
||||
instantaneous_rate <= 240.0f * m_presented_frames) {
|
||||
instantaneous_rate <= 240.0f) {
|
||||
constexpr float SmoothingFactor = 0.15f;
|
||||
if (m_smoothed_present_rate <= 0.0f) {
|
||||
m_smoothed_present_rate = instantaneous_rate;
|
||||
@@ -125,9 +124,18 @@ float EmuWindow_Android::GetFrameTimeVerifiedHint() const {
|
||||
return QuantizeFrameRateHint(verified_rate);
|
||||
}
|
||||
|
||||
float EmuWindow_Android::GetPresentedFrameMultiplier() {
|
||||
if (!Settings::values.frame_gen.GetValue()) {
|
||||
return 1.0f;
|
||||
}
|
||||
return static_cast<float>(std::clamp<u32>(Settings::values.frame_gen_multiplier.GetValue(), 2, 4));
|
||||
}
|
||||
|
||||
float EmuWindow_Android::GetFrameRateHint() const {
|
||||
const float observed_rate = std::clamp(m_smoothed_present_rate, 0.0f, 240.0f);
|
||||
const float frame_time_verified_hint = GetFrameTimeVerifiedHint() * m_presented_frames;
|
||||
const float presented_multiplier = GetPresentedFrameMultiplier();
|
||||
const float observed_rate =
|
||||
std::clamp(m_smoothed_present_rate * presented_multiplier, 0.0f, 240.0f);
|
||||
const float frame_time_verified_hint = GetFrameTimeVerifiedHint() * presented_multiplier;
|
||||
|
||||
if (m_last_frame_rate_hint > 0.0f && observed_rate > 0.0f) {
|
||||
const float tolerance = std::max(m_last_frame_rate_hint * 0.12f, 4.0f);
|
||||
@@ -151,7 +159,7 @@ float EmuWindow_Android::GetFrameRateHint() const {
|
||||
return frame_time_verified_hint;
|
||||
}
|
||||
|
||||
const float nominal_rate = 60.0f * m_presented_frames;
|
||||
const float nominal_rate = 60.0f * presented_multiplier;
|
||||
if (!Settings::values.use_speed_limit.GetValue()) {
|
||||
return QuantizeFrameRateHint(nominal_rate);
|
||||
}
|
||||
|
||||
@@ -41,7 +41,7 @@ public:
|
||||
~EmuWindow_Android() = default;
|
||||
|
||||
void OnSurfaceChanged(ANativeWindow* surface);
|
||||
void OnFrameDisplayed(u32 presented_frames) override;
|
||||
void OnFrameDisplayed() override;
|
||||
|
||||
void OnTouchPressed(int id, float x, float y);
|
||||
void OnTouchMoved(int id, float x, float y);
|
||||
@@ -58,9 +58,10 @@ private:
|
||||
using Clock = std::chrono::steady_clock;
|
||||
|
||||
void UpdateFrameRateHint();
|
||||
void UpdateObservedFrameRate(u32 presented_frames);
|
||||
void UpdateObservedFrameRate();
|
||||
[[nodiscard]] float GetFrameRateHint() const;
|
||||
[[nodiscard]] float GetFrameTimeVerifiedHint() const;
|
||||
[[nodiscard]] static float GetPresentedFrameMultiplier();
|
||||
[[nodiscard]] static float QuantizeFrameRateHint(float frame_rate);
|
||||
|
||||
float m_window_width{};
|
||||
@@ -72,7 +73,6 @@ private:
|
||||
float m_last_frame_rate_hint = -1.0f;
|
||||
float m_pending_frame_rate_hint = -1.0f;
|
||||
float m_smoothed_present_rate = 0.0f;
|
||||
float m_presented_frames = 1.0f;
|
||||
Clock::time_point m_last_frame_display_time{};
|
||||
Clock::time_point m_pending_frame_rate_since{};
|
||||
std::uint32_t m_pending_frame_rate_hint_votes = 0;
|
||||
|
||||
@@ -738,35 +738,6 @@ const char* fallback_cpu_detection() {
|
||||
return s_result.c_str();
|
||||
}
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
struct SystemDriverInfo {
|
||||
VkPhysicalDeviceProperties properties{};
|
||||
VkPhysicalDeviceDriverProperties driver{};
|
||||
};
|
||||
|
||||
SystemDriverInfo QuerySystemDriverInfo(JNIEnv* env, jstring j_hook_lib_dir) {
|
||||
const std::string hook_lib_dir = Common::Android::GetJString(env, j_hook_lib_dir);
|
||||
const Common::DynamicLibrary library{adrenotools_open_libvulkan(
|
||||
RTLD_NOW, 0, nullptr, hook_lib_dir.c_str(), nullptr, nullptr, nullptr, nullptr)};
|
||||
Vulkan::vk::InstanceDispatch dld;
|
||||
const Vulkan::vk::Instance instance = Vulkan::CreateInstance(library, dld, VK_API_VERSION_1_1);
|
||||
const std::vector<VkPhysicalDevice> devices = instance.EnumeratePhysicalDevices();
|
||||
if (devices.empty()) {
|
||||
throw Vulkan::vk::Exception(VK_ERROR_INITIALIZATION_FAILED);
|
||||
}
|
||||
SystemDriverInfo info{};
|
||||
info.driver.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_DRIVER_PROPERTIES;
|
||||
VkPhysicalDeviceProperties2 properties2{
|
||||
.sType = VK_STRUCTURE_TYPE_PHYSICAL_DEVICE_PROPERTIES_2,
|
||||
.pNext = &info.driver,
|
||||
.properties = {},
|
||||
};
|
||||
Vulkan::vk::PhysicalDevice(devices[0], dld).GetProperties2(properties2);
|
||||
info.properties = properties2.properties;
|
||||
return info;
|
||||
}
|
||||
#endif
|
||||
|
||||
} // namespace
|
||||
|
||||
extern "C" {
|
||||
@@ -886,18 +857,34 @@ jboolean JNICALL Java_org_yuzu_yuzu_1emu_utils_GpuDriverHelper_supportsCustomDri
|
||||
|
||||
jobjectArray Java_org_yuzu_yuzu_1emu_utils_GpuDriverHelper_getSystemDriverInfo(
|
||||
JNIEnv* env, jobject j_obj, jobject j_surf, jstring j_hook_lib_dir) {
|
||||
std::string version_string{"1.1.0"};
|
||||
std::string driver_name{"generic"};
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
try {
|
||||
const SystemDriverInfo info = QuerySystemDriverInfo(env, j_hook_lib_dir);
|
||||
const u32 driver_version = info.properties.driverVersion;
|
||||
version_string =
|
||||
fmt::format("{}.{}.{}", VK_API_VERSION_MAJOR(driver_version),
|
||||
VK_API_VERSION_MINOR(driver_version), VK_API_VERSION_PATCH(driver_version));
|
||||
driver_name = Vulkan::vk::GetDriverName(info.driver);
|
||||
} catch (...) {
|
||||
}
|
||||
const char* file_redirect_dir_{};
|
||||
int featureFlags{};
|
||||
std::string hook_lib_dir = Common::Android::GetJString(env, j_hook_lib_dir);
|
||||
auto handle = adrenotools_open_libvulkan(RTLD_NOW, featureFlags, nullptr, hook_lib_dir.c_str(),
|
||||
nullptr, nullptr, file_redirect_dir_, nullptr);
|
||||
auto driver_library = std::make_shared<Common::DynamicLibrary>(handle);
|
||||
InputCommon::InputSubsystem input_subsystem;
|
||||
auto window =
|
||||
std::make_unique<EmuWindow_Android>(ANativeWindow_fromSurface(env, j_surf), driver_library);
|
||||
|
||||
Vulkan::vk::InstanceDispatch dld;
|
||||
Vulkan::vk::Instance vk_instance = Vulkan::CreateInstance(
|
||||
*driver_library, dld, VK_API_VERSION_1_1, Core::Frontend::WindowSystemType::Android);
|
||||
|
||||
auto surface = Vulkan::CreateSurface(vk_instance, window->GetWindowInfo());
|
||||
|
||||
auto device = Vulkan::CreateDevice(vk_instance, dld, *surface);
|
||||
|
||||
auto driver_version = device.GetDriverVersion();
|
||||
auto version_string =
|
||||
fmt::format("{}.{}.{}", VK_API_VERSION_MAJOR(driver_version),
|
||||
VK_API_VERSION_MINOR(driver_version), VK_API_VERSION_PATCH(driver_version));
|
||||
auto driver_name = device.GetDriverName();
|
||||
#else
|
||||
auto driver_version = "1.0.0";
|
||||
auto version_string = "1.1.0"; //Assume lowest Vulkan level
|
||||
auto driver_name = "generic";
|
||||
#endif
|
||||
jobjectArray j_driver_info = env->NewObjectArray(2, Common::Android::GetStringClass(), Common::Android::ToJString(env, version_string));
|
||||
env->SetObjectArrayElement(j_driver_info, 1, Common::Android::ToJString(env, driver_name));
|
||||
@@ -906,12 +893,32 @@ jobjectArray Java_org_yuzu_yuzu_1emu_utils_GpuDriverHelper_getSystemDriverInfo(
|
||||
|
||||
jstring Java_org_yuzu_yuzu_1emu_utils_GpuDriverHelper_getGpuModel(JNIEnv *env, jobject j_obj, jobject j_surf, jstring j_hook_lib_dir) {
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
try {
|
||||
return Common::Android::ToJString(env, QuerySystemDriverInfo(env, j_hook_lib_dir).properties.deviceName);
|
||||
} catch (...) {
|
||||
}
|
||||
#endif
|
||||
const char* file_redirect_dir_{};
|
||||
int featureFlags{};
|
||||
std::string hook_lib_dir = Common::Android::GetJString(env, j_hook_lib_dir);
|
||||
auto handle = adrenotools_open_libvulkan(RTLD_NOW, featureFlags, nullptr, hook_lib_dir.c_str(),
|
||||
nullptr, nullptr, file_redirect_dir_, nullptr);
|
||||
auto driver_library = std::make_shared<Common::DynamicLibrary>(handle);
|
||||
InputCommon::InputSubsystem input_subsystem;
|
||||
auto window =
|
||||
std::make_unique<EmuWindow_Android>(ANativeWindow_fromSurface(env, j_surf), driver_library);
|
||||
|
||||
Vulkan::vk::InstanceDispatch dld;
|
||||
Vulkan::vk::Instance vk_instance = Vulkan::CreateInstance(
|
||||
*driver_library, dld, VK_API_VERSION_1_1, Core::Frontend::WindowSystemType::Android);
|
||||
|
||||
auto surface = Vulkan::CreateSurface(vk_instance, window->GetWindowInfo());
|
||||
|
||||
auto device = Vulkan::CreateDevice(vk_instance, dld, *surface);
|
||||
|
||||
const std::string model_name{device.GetModelName()};
|
||||
|
||||
window.release();
|
||||
|
||||
return Common::Android::ToJString(env, model_name);
|
||||
#else
|
||||
return Common::Android::ToJString(env, "no-info");
|
||||
#endif
|
||||
}
|
||||
|
||||
jboolean Java_org_yuzu_yuzu_1emu_NativeLibrary_reloadKeys(JNIEnv* env, jclass clazz) {
|
||||
|
||||
@@ -0,0 +1,96 @@
|
||||
<?xml version="1.0" encoding="utf-8"?>
|
||||
<ScrollView xmlns:android="http://schemas.android.com/apk/res/android"
|
||||
xmlns:app="http://schemas.android.com/apk/res-auto"
|
||||
android:layout_width="match_parent"
|
||||
android:layout_height="wrap_content"
|
||||
android:scrollbars="vertical">
|
||||
|
||||
<LinearLayout
|
||||
android:layout_width="match_parent"
|
||||
android:layout_height="wrap_content"
|
||||
android:orientation="vertical"
|
||||
android:padding="24dp">
|
||||
|
||||
<com.google.android.material.switchmaterial.SwitchMaterial
|
||||
android:id="@+id/switch_enable"
|
||||
android:layout_width="match_parent"
|
||||
android:layout_height="wrap_content"
|
||||
android:layout_marginBottom="16dp"
|
||||
android:text="@string/gpu_unswizzle_enable" />
|
||||
|
||||
<com.google.android.material.textfield.TextInputLayout
|
||||
android:id="@+id/layout_texture_size"
|
||||
style="@style/Widget.Material3.TextInputLayout.OutlinedBox.ExposedDropdownMenu"
|
||||
android:layout_width="match_parent"
|
||||
android:layout_height="wrap_content"
|
||||
android:layout_marginBottom="12dp"
|
||||
android:hint="@string/gpu_unswizzle_texture_size">
|
||||
|
||||
<com.google.android.material.textfield.MaterialAutoCompleteTextView
|
||||
android:id="@+id/dropdown_texture_size"
|
||||
android:layout_width="match_parent"
|
||||
android:layout_height="wrap_content"
|
||||
android:inputType="none" />
|
||||
</com.google.android.material.textfield.TextInputLayout>
|
||||
|
||||
<com.google.android.material.textfield.TextInputLayout
|
||||
android:id="@+id/layout_stream_size"
|
||||
style="@style/Widget.Material3.TextInputLayout.OutlinedBox.ExposedDropdownMenu"
|
||||
android:layout_width="match_parent"
|
||||
android:layout_height="wrap_content"
|
||||
android:layout_marginBottom="12dp"
|
||||
android:hint="@string/gpu_unswizzle_stream_size">
|
||||
|
||||
<com.google.android.material.textfield.MaterialAutoCompleteTextView
|
||||
android:id="@+id/dropdown_stream_size"
|
||||
android:layout_width="match_parent"
|
||||
android:layout_height="wrap_content"
|
||||
android:inputType="none" />
|
||||
</com.google.android.material.textfield.TextInputLayout>
|
||||
|
||||
<com.google.android.material.textfield.TextInputLayout
|
||||
android:id="@+id/layout_chunk_size"
|
||||
style="@style/Widget.Material3.TextInputLayout.OutlinedBox.ExposedDropdownMenu"
|
||||
android:layout_width="match_parent"
|
||||
android:layout_height="wrap_content"
|
||||
android:hint="@string/gpu_unswizzle_chunk_size">
|
||||
|
||||
<com.google.android.material.textfield.MaterialAutoCompleteTextView
|
||||
android:id="@+id/dropdown_chunk_size"
|
||||
android:layout_width="match_parent"
|
||||
android:layout_height="wrap_content"
|
||||
android:inputType="none" />
|
||||
</com.google.android.material.textfield.TextInputLayout>
|
||||
|
||||
<LinearLayout
|
||||
android:layout_width="match_parent"
|
||||
android:layout_height="wrap_content"
|
||||
android:layout_marginTop="24dp"
|
||||
android:orientation="horizontal"
|
||||
android:gravity="end"
|
||||
android:spacing="8dp">
|
||||
|
||||
<com.google.android.material.button.MaterialButton
|
||||
android:id="@+id/btn_default"
|
||||
style="@style/Widget.Material3.Button.TextButton"
|
||||
android:layout_width="wrap_content"
|
||||
android:layout_height="wrap_content"
|
||||
android:text="@string/gpu_unswizzle_default_button" />
|
||||
|
||||
<com.google.android.material.button.MaterialButton
|
||||
android:id="@+id/btn_cancel"
|
||||
style="@style/Widget.Material3.Button.TextButton"
|
||||
android:layout_width="wrap_content"
|
||||
android:layout_height="wrap_content"
|
||||
android:text="@android:string/cancel" />
|
||||
|
||||
<com.google.android.material.button.MaterialButton
|
||||
android:id="@+id/btn_ok"
|
||||
style="@style/Widget.Material3.Button.TextButton"
|
||||
android:layout_width="wrap_content"
|
||||
android:layout_height="wrap_content"
|
||||
android:text="@android:string/ok" />
|
||||
</LinearLayout>
|
||||
|
||||
</LinearLayout>
|
||||
</ScrollView>
|
||||
@@ -574,6 +574,8 @@
|
||||
<string name="renderer_force_max_clock_description">يجبر وحدة معالجة الرسومات على العمل بأقصى سرعة ممكنة (سيظل يتم تطبيق القيود الحرارية).</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">محاكاة غير متزامنة لوحدة معالجة الرسومات</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">يمكن لهذه الحيلة أن تزيد الأداء عن طريق تشغيل محاكاة وحدة معالجة الرسومات بشكل غير متزامن على حساب مشاكل الرسومات وزيادة معدلات الأعطال بسبب العمليات المتعلقة بالتوقيت.</string>
|
||||
<string name="renderer_async_presentation">عرض غير متزامن</string>
|
||||
<string name="renderer_async_presentation_description">يمكن لهذه الحيلة أن تزيد من الأداء عن طريق نقل عملية العرض إلى خيط معالجة منفصل على حساب مشاكل الرسوميات.</string>
|
||||
<string name="renderer_reactive_flushing">استخدم التنظيف التفاعلي</string>
|
||||
<string name="renderer_reactive_flushing_description">يحسن دقة العرض في بعض الألعاب على حساب الأداء.</string>
|
||||
<string name="enable_buffer_history">تمكين سجل التخزين المؤقت</string>
|
||||
@@ -597,6 +599,17 @@
|
||||
<string name="rescale_hack_description">يُمكّن هذا الخيار من التعامل مع عملية إعادة تحجيم الألعاب بطريقة تقليدية باستخدام مسار إعادة التحجيم السريع</string>
|
||||
<string name="renderer_asynchronous_shaders">استخدم تظليل غير متزامن</string>
|
||||
<string name="renderer_asynchronous_shaders_description">يقوم بتجميع التظليل بشكل غير متزامن. قد يقلل ذلك من التقطعات ولكنه قد يؤدي أيضًا إلى حدوث أخطاء.</string>
|
||||
<string name="gpu_unswizzle_settings">إعدادات إلغاء ترتيب بيانات وحدة معالجة الرسومات</string>
|
||||
<string name="gpu_unswizzle_settings_description">قم بضبط معلمات فكّ تشابك النسيج المستندة إلى وحدة معالجة الرسومات أو تعطيلها تمامًا. اضبط هذه الإعدادات لتحقيق التوازن بين الأداء وجودة تحميل النسيج.</string>
|
||||
<string name="gpu_unswizzle_enable">تفعيل إلغاء ترتيب بيانات وحدة معالجة الرسومات</string>
|
||||
<string name="gpu_unswizzle_disabled">تعطيل</string>
|
||||
<string name="gpu_unswizzle_texture_size">الحد الأقصى لحجم النسيج في وحدة معالجة الرسومات بعد إعادة ترتيب البيانات</string>
|
||||
<string name="gpu_unswizzle_texture_size_description">يُحدد هذا الخيار الحد الأقصى لحجم (ميغابايت) معالجة الصور باستخدام وحدة معالجة الرسومات. مع أن وحدة معالجة الرسومات أسرع في معالجة الصور المتوسطة والكبيرة، إلا أن وحدة المعالجة المركزية قد تكون أكثر كفاءة في معالجة الصور الصغيرة جدًا. اضبط هذا الخيار لتحقيق التوازن الأمثل بين سرعة معالجة الرسومات واستهلاك وحدة المعالجة المركزية.</string>
|
||||
<string name="gpu_unswizzle_stream_size">حجم تدفق إلغاء ترتيب بيانات وحدة معالجة الرسومات</string>
|
||||
<string name="gpu_unswizzle_stream_size_description">يحدد هذا الخيار حد البيانات لكل إطار لمعالجة الصور الكبيرة. القيم الأعلى تُسرّع تحميل الصور على حساب زيادة زمن استجابة الإطارات؛ أما القيم الأقل فتُقلل من الحمل الزائد على وحدة معالجة الرسومات، ولكنها قد تتسبب في ظهور الصور بشكل مفاجئ.</string>
|
||||
<string name="gpu_unswizzle_chunk_size">حجم كتلة إلغاء ترتيب بيانات وحدة معالجة الرسومات</string>
|
||||
<string name="gpu_unswizzle_chunk_size_description">يُحدد هذا الخيار عدد شرائح العمق التي تتم معالجتها لكل دفعة من الصور ثلاثية الأبعاد (3D). زيادة هذا العدد تُحسّن كفاءة الإنتاجية على وحدات معالجة الرسومات القوية، ولكنها قد تُسبب تقطعًا أو انقطاعًا في عمل برنامج التشغيل على الأجهزة ذات المواصفات الأقل قوة.</string>
|
||||
<string name="gpu_unswizzle_default_button">افتراضي</string>
|
||||
|
||||
|
||||
<string name="extensions">إضافات</string>
|
||||
@@ -1037,10 +1050,25 @@
|
||||
<string name="fast_gpu_high">كسر السرعة</string>
|
||||
|
||||
<!-- GPU swizzle texture size -->
|
||||
<string name="gpu_texturesizeswizzle_verysmall">صغير جدًا (16 ميغابايت)</string>
|
||||
<string name="gpu_texturesizeswizzle_small">صغير (32 ميغابايت)</string>
|
||||
<string name="gpu_texturesizeswizzle_normal">قياسي (128 ميغابايت)</string>
|
||||
<string name="gpu_texturesizeswizzle_large">كبير (256 ميغابايت)</string>
|
||||
<string name="gpu_texturesizeswizzle_verylarge">كبير جدًا (512 ميغابايت)</string>
|
||||
|
||||
<!-- GPU swizzle streams -->
|
||||
<string name="gpu_swizzle_verylow">منخفض جدًا (4 ميغابايت)</string>
|
||||
<string name="gpu_swizzle_low">منخفض (8 ميغابايت)</string>
|
||||
<string name="gpu_swizzle_normal">قياسي (16 ميغابايت)</string>
|
||||
<string name="gpu_swizzle_medium">متوسط (32 ميغابايت)</string>
|
||||
<string name="gpu_swizzle_high">عالي (64 ميغابايت)</string>
|
||||
|
||||
<!-- GPU swizzle chunks -->
|
||||
<string name="gpu_swizzlechunk_verylow">منخفض جدًا (32)</string>
|
||||
<string name="gpu_swizzlechunk_low">منخفض (64)</string>
|
||||
<string name="gpu_swizzlechunk_normal">قياسي (128)</string>
|
||||
<string name="gpu_swizzlechunk_medium">متوسط (256)</string>
|
||||
<string name="gpu_swizzlechunk_high">عالي (512)</string>
|
||||
|
||||
<!-- Temperature Units -->
|
||||
<string name="temperature_celsius">مئوية</string>
|
||||
|
||||
@@ -508,6 +508,7 @@ Wird der Handheld-Modus verwendet, verringert es die Auflösung und erhöht die
|
||||
<string name="skip_cpu_inner_invalidation_description">Überspringt bestimmte Cache-Invalidierungen auf CPU-Seite während Speicherupdates, reduziert die CPU-Auslastung und verbessert die Leistung. Kann in einigen Spielen zu Fehlern oder Abstürzen führen.</string>
|
||||
<string name="renderer_asynchronous_shaders">Asynchrone Shader</string>
|
||||
<string name="renderer_asynchronous_shaders_description">Kompiliert Shader asynchron. Dies kann Ruckler reduzieren, aber auch Grafikfehler verursachen.</string>
|
||||
<string name="gpu_unswizzle_default_button">Standard</string>
|
||||
|
||||
|
||||
<string name="extensions">Erweiterungen</string>
|
||||
@@ -917,10 +918,25 @@ Wirklich fortfahren?</string>
|
||||
<string name="fast_gpu_high">Übertaktung</string>
|
||||
|
||||
<!-- GPU swizzle texture size -->
|
||||
<string name="gpu_texturesizeswizzle_verysmall">Sehr klein (16 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_small">Klein (32 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_normal">Normal (128 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_large">Groß (256 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_verylarge">Sehr groß (512 MB)</string>
|
||||
|
||||
<!-- GPU swizzle streams -->
|
||||
<string name="gpu_swizzle_verylow">Sehr niedrig (4 MB)</string>
|
||||
<string name="gpu_swizzle_low">Niedrig (8 MB)</string>
|
||||
<string name="gpu_swizzle_normal">Normal (16 MB)</string>
|
||||
<string name="gpu_swizzle_medium">Mittel (32 MB)</string>
|
||||
<string name="gpu_swizzle_high">Hoch (64 MB)</string>
|
||||
|
||||
<!-- GPU swizzle chunks -->
|
||||
<string name="gpu_swizzlechunk_verylow">Sehr niedrig (32)</string>
|
||||
<string name="gpu_swizzlechunk_low">Niedrig (64)</string>
|
||||
<string name="gpu_swizzlechunk_normal">Normal (128)</string>
|
||||
<string name="gpu_swizzlechunk_medium">Mittel (256)</string>
|
||||
<string name="gpu_swizzlechunk_high">Hoch (512)</string>
|
||||
|
||||
<!-- Temperature Units -->
|
||||
<string name="temperature_celsius">Celsius</string>
|
||||
|
||||
@@ -517,6 +517,8 @@
|
||||
<string name="renderer_force_max_clock_description">Fuerza a la GPU a ejecutarse a la velocidad máxima de reloj posible (se seguirán aplicando restricciones térmicas).</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">Emulación de GPU asíncrona</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">Este hack puede aumentar el rendimiento ejecutando la emulación de la GPU de forma asíncrona, a costa de problemas gráficos y un aumento en la tasa de fallos debido a operaciones relacionadas con la sincronización.</string>
|
||||
<string name="renderer_async_presentation">Presentación asíncrona</string>
|
||||
<string name="renderer_async_presentation_description">Este hack puede aumentar el rendimiento al mover la presentación a un hilo independiente de la CPU a costa de problemas gráficos.</string>
|
||||
<string name="renderer_reactive_flushing">Usar limpieza reactiva</string>
|
||||
<string name="renderer_reactive_flushing_description">Mejora la precisión de renderizado en algunos juegos, pero reduce el rendimiento.</string>
|
||||
<string name="enable_buffer_history">Activar el historial del búfer</string>
|
||||
@@ -539,6 +541,17 @@
|
||||
<string name="rescale_hack_description">Permite el manejo de versiones anteriores para el paso de configuración de reescalado para juegos mediante el uso de una ruta de reescalado rápida.</string>
|
||||
<string name="renderer_asynchronous_shaders">Usar sombreadores asíncronos</string>
|
||||
<string name="renderer_asynchronous_shaders_description">Compila los sombreadores de forma asíncrona. Esto puede reducir los tirones, pero también puede introducir errores gráficos.</string>
|
||||
<string name="gpu_unswizzle_settings">Ajustes de desentrelazado de la GPU</string>
|
||||
<string name="gpu_unswizzle_settings_description">Configura los parámetros de desentrelazado de texturas basadas en la GPU o desactívelos por completo. Modifique estos ajustes para equilibrar el rendimiento y la calidad de las texturas cargadas.</string>
|
||||
<string name="gpu_unswizzle_enable">Activar desentrelazado de la GPU</string>
|
||||
<string name="gpu_unswizzle_disabled">Desactivado</string>
|
||||
<string name="gpu_unswizzle_texture_size">Tamaño máximo de textura de desentrelazado de la GPU</string>
|
||||
<string name="gpu_unswizzle_texture_size_description">Establece el tamaño máximo (en MB) para el desentrelazado de texturas basada en GPU. Aunque la GPU es más rápida para texturas medianas y grandes, la CPU puede ser más eficiente para texturas muy pequeñas. Ajuste este valor para encontrar el equilibrio entre la aceleración de la GPU y la sobrecarga de la CPU.</string>
|
||||
<string name="gpu_unswizzle_stream_size">Tamaño del flujo de desentrelazado de la GPU</string>
|
||||
<string name="gpu_unswizzle_stream_size_description">Establece el límite de datos por fotograma para desentrelazar texturas grandes. Los valores altos aceleran la carga de texturas, a coste de una mayor latencia por fotograma; los valores bajos reducen la carga de la GPU, pero pueden causar parpadeos visibles en las texturas.</string>
|
||||
<string name="gpu_unswizzle_chunk_size">Tamaño del trozo de desentrelazado de la GPU</string>
|
||||
<string name="gpu_unswizzle_chunk_size_description">Determina la cantidad de cortes de profundidad procesados en un solo envío de texturas 3D. Aumentar este valor puede mejorar el rendimiento en una GPU de gama alta, pero puede causar tirones y problemas en los tiempos de respuesta en hardware más modesto.</string>
|
||||
<string name="gpu_unswizzle_default_button">Por defecto</string>
|
||||
|
||||
|
||||
<string name="extensions">Extensiones</string>
|
||||
@@ -975,10 +988,25 @@
|
||||
<string name="fast_gpu_high">Overclock</string>
|
||||
|
||||
<!-- GPU swizzle texture size -->
|
||||
<string name="gpu_texturesizeswizzle_verysmall">Muy pequeño (16 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_small">Pequeño (32 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_normal">Normal (128 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_large">Grande (256 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_verylarge">Muy grande (512 MB)</string>
|
||||
|
||||
<!-- GPU swizzle streams -->
|
||||
<string name="gpu_swizzle_verylow">Muy bajo (4 MB)</string>
|
||||
<string name="gpu_swizzle_low">Bajo (8 MB)</string>
|
||||
<string name="gpu_swizzle_normal">Normal (16 MB)</string>
|
||||
<string name="gpu_swizzle_medium">Medio (32 MB)</string>
|
||||
<string name="gpu_swizzle_high">Alto (64 MB)</string>
|
||||
|
||||
<!-- GPU swizzle chunks -->
|
||||
<string name="gpu_swizzlechunk_verylow">Muy bajo (32)</string>
|
||||
<string name="gpu_swizzlechunk_low">Bajo (64)</string>
|
||||
<string name="gpu_swizzlechunk_normal">Normal (128)</string>
|
||||
<string name="gpu_swizzlechunk_medium">Medio (256)</string>
|
||||
<string name="gpu_swizzlechunk_high">Alto (512)</string>
|
||||
|
||||
<!-- Temperature Units -->
|
||||
<string name="temperature_celsius">Celsius</string>
|
||||
|
||||
@@ -482,6 +482,8 @@
|
||||
<string name="emulate_bgr565">Emuler BGR565</string>
|
||||
<string name="renderer_asynchronous_shaders">Utiliser les shaders asynchrones</string>
|
||||
<string name="renderer_asynchronous_shaders_description">Compile les shaders de manière asynchrone. Cela peut réduire les saccades mais peut aussi provoquer des problèmes graphiques.</string>
|
||||
<string name="gpu_unswizzle_disabled">Désactivé</string>
|
||||
<string name="gpu_unswizzle_default_button">Par défaut</string>
|
||||
|
||||
|
||||
<string name="extensions">Extensions</string>
|
||||
@@ -882,10 +884,25 @@
|
||||
<string name="memory_8gb">8 Go (Dangereux)</string>
|
||||
|
||||
<!-- GPU swizzle texture size -->
|
||||
<string name="gpu_texturesizeswizzle_verysmall">Très petit (16 Mo)</string>
|
||||
<string name="gpu_texturesizeswizzle_small">Petit (32 Mo)</string>
|
||||
<string name="gpu_texturesizeswizzle_normal">Normal (128 Mo)</string>
|
||||
<string name="gpu_texturesizeswizzle_large">Large (256 Mo)</string>
|
||||
<string name="gpu_texturesizeswizzle_verylarge">Très large (512 Mo)</string>
|
||||
|
||||
<!-- GPU swizzle streams -->
|
||||
<string name="gpu_swizzle_verylow">Très faible (4 Mo)</string>
|
||||
<string name="gpu_swizzle_low">Faible (8 Mo)</string>
|
||||
<string name="gpu_swizzle_normal">Normal (16 Mo)</string>
|
||||
<string name="gpu_swizzle_medium">Moyen (32 Mo)</string>
|
||||
<string name="gpu_swizzle_high">Élevé (64 Mo)</string>
|
||||
|
||||
<!-- GPU swizzle chunks -->
|
||||
<string name="gpu_swizzlechunk_verylow">Très faible (32)</string>
|
||||
<string name="gpu_swizzlechunk_low">Faible (64)</string>
|
||||
<string name="gpu_swizzlechunk_normal">Normal (128)</string>
|
||||
<string name="gpu_swizzlechunk_medium">Moyen (256)</string>
|
||||
<string name="gpu_swizzlechunk_high">Élevé (512)</string>
|
||||
|
||||
<!-- Temperature Units -->
|
||||
<string name="temperature_celsius">Celsius</string>
|
||||
|
||||
@@ -554,6 +554,8 @@
|
||||
<string name="renderer_force_max_clock_description">Заставляет ГПУ работать на максимально возможных тактовых частотах (тепловые ограничения все равно будут применяться).</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">Асинхронная эмуляция ГПУ</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">Может повысить производительность за счёт асинхронного запуска эмуляции ГПУ, но ценой появления графических ошибок и увеличения частоты вылетов из-за операций, зависящих от синхронизации.</string>
|
||||
<string name="renderer_async_presentation">Асинхронная презентация</string>
|
||||
<string name="renderer_async_presentation_description">Может повысить производительность за счёт перемещения вывода кадров в отдельный поток ЦП, но ценой возникновения графических проблем.</string>
|
||||
<string name="renderer_reactive_flushing">Реактивная очистка</string>
|
||||
<string name="renderer_reactive_flushing_description">Повышение точности рендеринга в некоторых играх за счет снижения производительности.</string>
|
||||
<string name="enable_buffer_history">Включить историю буфера</string>
|
||||
@@ -577,6 +579,17 @@
|
||||
<string name="rescale_hack_description">Включает старый метод обработки этапа перенастройки масштабирования для игр за счёт использования быстрого алгоритма перемасштабирования.</string>
|
||||
<string name="renderer_asynchronous_shaders">Использовать асинхронные шейдеры</string>
|
||||
<string name="renderer_asynchronous_shaders_description">Компилирует шейдеры асинхронно. Это может уменьшить подтормаживания, но также может вызвать графические артефакты.</string>
|
||||
<string name="gpu_unswizzle_settings">Настройки распаковки текстур (Unswizzle)</string>
|
||||
<string name="gpu_unswizzle_settings_description">Настройте параметры распаковки текстур на стороне ГПУ либо полностью отключите эту функцию. Изменение этих параметров позволяет найти баланс между производительностью и качеством загрузки текстур.</string>
|
||||
<string name="gpu_unswizzle_enable">Включить распаковку текстур (Unswizzle)</string>
|
||||
<string name="gpu_unswizzle_disabled">Отключено</string>
|
||||
<string name="gpu_unswizzle_texture_size">Макс. размер текстуры Unswizzle</string>
|
||||
<string name="gpu_unswizzle_texture_size_description">Задает максимальный размер (в МБ) текстур для преобразования формата (unswizzle) на ГПУ. Хотя ГПУ быстрее работает со средними и большими текстурами, ЦП может быть эффективнее для очень маленьких. Настройте это значение, чтобы найти баланс между ускорением на ГПУ и нагрузкой на ЦП.</string>
|
||||
<string name="gpu_unswizzle_stream_size">Размер потока Unswizzle</string>
|
||||
<string name="gpu_unswizzle_stream_size_description">Задает лимит данных на кадр для преобразования крупных текстур (unswizzle). Высокие значения ускоряют загрузку текстур ценой увеличения задержки кадра; низкие значения снижают нагрузку на ГПУ, но могут вызывать заметную постепенную подгрузку текстур.</string>
|
||||
<string name="gpu_unswizzle_chunk_size">Размер блока Unswizzle</string>
|
||||
<string name="gpu_unswizzle_chunk_size_description">Задает количество слоёв глубины, обрабатываемых за одну пачку для 3D-текстур. Увеличение этого значения улучшает пропускную способность на мощных ГПУ, но может вызывать подтормаживания или таймауты драйвера на слабом железе.</string>
|
||||
<string name="gpu_unswizzle_default_button">По умолчанию</string>
|
||||
|
||||
|
||||
<string name="extensions">Расширения</string>
|
||||
@@ -1017,10 +1030,25 @@
|
||||
<string name="fast_gpu_high">Разгон</string>
|
||||
|
||||
<!-- GPU swizzle texture size -->
|
||||
<string name="gpu_texturesizeswizzle_verysmall">Очень малый (16 МБ)</string>
|
||||
<string name="gpu_texturesizeswizzle_small">Малый (32 МБ)</string>
|
||||
<string name="gpu_texturesizeswizzle_normal">Обычный (128 МБ)</string>
|
||||
<string name="gpu_texturesizeswizzle_large">Большой (256 МБ)</string>
|
||||
<string name="gpu_texturesizeswizzle_verylarge">Очень большой (512 МБ)</string>
|
||||
|
||||
<!-- GPU swizzle streams -->
|
||||
<string name="gpu_swizzle_verylow">Очень низкий (4 МБ)</string>
|
||||
<string name="gpu_swizzle_low">Низкий (8 МБ)</string>
|
||||
<string name="gpu_swizzle_normal">Обычный (16 МБ)</string>
|
||||
<string name="gpu_swizzle_medium">Средний (32 МБ)</string>
|
||||
<string name="gpu_swizzle_high">Высокий (64 МБ)</string>
|
||||
|
||||
<!-- GPU swizzle chunks -->
|
||||
<string name="gpu_swizzlechunk_verylow">Очень малый (32)</string>
|
||||
<string name="gpu_swizzlechunk_low">Малый (64)</string>
|
||||
<string name="gpu_swizzlechunk_normal">Обычный (128)</string>
|
||||
<string name="gpu_swizzlechunk_medium">Средний (256)</string>
|
||||
<string name="gpu_swizzlechunk_high">Большой (512)</string>
|
||||
|
||||
<!-- Temperature Units -->
|
||||
<string name="temperature_celsius">Цельсий</string>
|
||||
|
||||
@@ -481,6 +481,8 @@
|
||||
<string name="renderer_force_max_clock_description">Змушує GPU працювати на максимальній тактовій частоті.</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">Асинхронна емуляція ГП</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">Це обхідне рішення може покращити продуктивність завдяки асинхронному виконанню емуляції ГП, але спричинить проблеми з графікою та збільшить частоту збоїв чутливих до таймінгів операцій.</string>
|
||||
<string name="renderer_async_presentation">Асинхронне подання</string>
|
||||
<string name="renderer_async_presentation_description">Це обхідне рішення може покращити продуктивність завдяки переміщенню подання на окремий потік ЦП, але спричинить проблеми з графікою.</string>
|
||||
<string name="renderer_reactive_flushing">Реактивне очищення</string>
|
||||
<string name="renderer_reactive_flushing_description">Покращує точність рендерингу в деяких іграх.</string>
|
||||
<string name="enable_buffer_history">Увімкнути історію буфера</string>
|
||||
@@ -500,6 +502,17 @@
|
||||
<string name="rescale_hack_description">Вмикає застарілу обробку масштабування для ігор, використовуючи швидкий шлях масштабування</string>
|
||||
<string name="renderer_asynchronous_shaders">Асинхронні шейдери</string>
|
||||
<string name="renderer_asynchronous_shaders_description">Компілює шейдери асинхронно. Це може зменшити затримки, але також може спричинити графічні баги.</string>
|
||||
<string name="gpu_unswizzle_settings">Налаштування розпакування за допомогою ГП</string>
|
||||
<string name="gpu_unswizzle_settings_description">Налаштуйте розпакування текстур за допомогою ГП або повністю вимкнути його. Відкоригуйте ці налаштування, щоб урівноважити продуктивність і якість завантаження текстур.</string>
|
||||
<string name="gpu_unswizzle_enable">Увімкнути розпакування за допомогою ГП</string>
|
||||
<string name="gpu_unswizzle_disabled">Вимкнено</string>
|
||||
<string name="gpu_unswizzle_texture_size">Максимальний розмір текстур для розпакування ГП за допомогою ГП</string>
|
||||
<string name="gpu_unswizzle_texture_size_description">Встановлює максимальний розмір (МБ) для розпакування текстур за допомогою ГП. ГП швидше справляється з текстурами середніх і великих розмірів, а ЦП ефективніший для дуже маленьких. Налаштуйте, щоб збалансувати ГП-прискоренням і навантаженням на ЦП.</string>
|
||||
<string name="gpu_unswizzle_stream_size">Розмір потоку розпакування за допомогою ГП</string>
|
||||
<string name="gpu_unswizzle_stream_size_description">Встановлює обмеження даних на кадр для розпакування великих текстур. Вищі значення пришвидшують завантаження текстур за рахунок більших кадрових затримок; менші значення зменшують перевантаження ГП але може спричинити помітні появи текстур.</string>
|
||||
<string name="gpu_unswizzle_chunk_size">Розмір блоків розпакування за допомогою ГП</string>
|
||||
<string name="gpu_unswizzle_chunk_size_description">Визначає кількість зрізів глибини, оброблених за партію 3D-текстур. Збільшення здатне покращити пропускну здатність на потужних ГП, але може призвести до затримок або затримок драйвера зі слабшим устаткуванням.</string>
|
||||
<string name="gpu_unswizzle_default_button">Стандартно</string>
|
||||
|
||||
|
||||
<string name="extensions">Розширення</string>
|
||||
@@ -916,10 +929,25 @@
|
||||
<string name="memory_8gb">8 ГБ (Небезпечно)</string>
|
||||
|
||||
<!-- GPU swizzle texture size -->
|
||||
<string name="gpu_texturesizeswizzle_verysmall">Дуже малий (16 МБ)</string>
|
||||
<string name="gpu_texturesizeswizzle_small">Малий (32 МБ)</string>
|
||||
<string name="gpu_texturesizeswizzle_normal">Нормальний (128 МБ)</string>
|
||||
<string name="gpu_texturesizeswizzle_large">Великий (256 МБ)</string>
|
||||
<string name="gpu_texturesizeswizzle_verylarge">Дуже великий (512 МБ)</string>
|
||||
|
||||
<!-- GPU swizzle streams -->
|
||||
<string name="gpu_swizzle_verylow">Дуже низький (4 МБ)</string>
|
||||
<string name="gpu_swizzle_low">Низький (8 МБ)</string>
|
||||
<string name="gpu_swizzle_normal">Нормальний (16 МБ)</string>
|
||||
<string name="gpu_swizzle_medium">Середній (32 МБ)</string>
|
||||
<string name="gpu_swizzle_high">Високий (64 МБ)</string>
|
||||
|
||||
<!-- GPU swizzle chunks -->
|
||||
<string name="gpu_swizzlechunk_verylow">Дуже низький (32)</string>
|
||||
<string name="gpu_swizzlechunk_low">Низький (64)</string>
|
||||
<string name="gpu_swizzlechunk_normal">Нормальний (128)</string>
|
||||
<string name="gpu_swizzlechunk_medium">Середній (256)</string>
|
||||
<string name="gpu_swizzlechunk_high">Високий (512)</string>
|
||||
|
||||
<!-- Temperature Units -->
|
||||
<string name="temperature_celsius">Цельсій</string>
|
||||
|
||||
@@ -535,6 +535,8 @@
|
||||
<string name="renderer_force_max_clock_description">强制 GPU 以最大时钟运行 (温控依然生效)。</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">GPU 异步模拟</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">此技巧可通过异步运行 GPU 模拟来提升性能,但在执行与时序相关的操作时,可能带来图形显示问题以及增加崩溃概率。</string>
|
||||
<string name="renderer_async_presentation">异步呈现</string>
|
||||
<string name="renderer_async_presentation_description">此技巧通过将图形呈现移至独立的 CPU 线程来提升性能,但可能会带来图形显示问题。</string>
|
||||
<string name="renderer_reactive_flushing">启用反应性刷新</string>
|
||||
<string name="renderer_reactive_flushing_description">通过牺牲性能来提升某些游戏的渲染精度。</string>
|
||||
<string name="enable_buffer_history">启用缓冲区历史</string>
|
||||
@@ -558,6 +560,17 @@
|
||||
<string name="rescale_hack_description">启用通过使用快速缩放路径,来为游戏提供缩放配置处理的传统处理方式</string>
|
||||
<string name="renderer_asynchronous_shaders">使用异步着色器</string>
|
||||
<string name="renderer_asynchronous_shaders_description">以异步方式编译着色器。采用此方式或可减少卡顿,但也可能引入故障点。</string>
|
||||
<string name="gpu_unswizzle_settings">GPU Unswizzle 设置</string>
|
||||
<string name="gpu_unswizzle_settings_description">配置基于 GPU 的纹理 unswizzling 参数,或完全禁用该功能。通过调整这些设置,以尝试在性能与纹理加载质量之间取得平衡。</string>
|
||||
<string name="gpu_unswizzle_enable">启用 GPU Unswizzle</string>
|
||||
<string name="gpu_unswizzle_disabled">禁用</string>
|
||||
<string name="gpu_unswizzle_texture_size">GPU Unswizzle 最大纹理尺寸</string>
|
||||
<string name="gpu_unswizzle_texture_size_description">设置基于 GPU 的纹理 unswizzling 的最大尺寸(MB)。虽然 GPU 处理中等和大型纹理的速度更快,但对于非常小的纹理,CPU 可能更为高效。通过调节此项设置,以平衡GPU 加速与 CPU 开销。</string>
|
||||
<string name="gpu_unswizzle_stream_size">GPU Unswizzle 流大小</string>
|
||||
<string name="gpu_unswizzle_stream_size_description">设置用于 unswizzling 大型纹理时的每帧数据限制。较高的数值可以加速纹理的加载过程,但会带来更高的帧延迟。而较低的数值则可以降低 GPU 的开销,但也可能会导致可见的纹理闪现。</string>
|
||||
<string name="gpu_unswizzle_chunk_size">GPU Unswizzle 块大小</string>
|
||||
<string name="gpu_unswizzle_chunk_size_description">定义了 3D 纹理每批次处理的深度切片数量。增加此数值可在高性能 GPU 上提升吞吐效率,但在性能较弱的硬件上可能会导致卡顿或驱动超时。</string>
|
||||
<string name="gpu_unswizzle_default_button">默认</string>
|
||||
|
||||
|
||||
<string name="extensions">扩展</string>
|
||||
@@ -991,10 +1004,25 @@
|
||||
<string name="fast_gpu_high">超频</string>
|
||||
|
||||
<!-- GPU swizzle texture size -->
|
||||
<string name="gpu_texturesizeswizzle_verysmall">极小 (16 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_small">较小 (32 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_normal">正常 (128 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_large">较大 (256 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_verylarge">极大 (512 MB)</string>
|
||||
|
||||
<!-- GPU swizzle streams -->
|
||||
<string name="gpu_swizzle_verylow">极低 (4 MB)</string>
|
||||
<string name="gpu_swizzle_low">低 (8 MB)</string>
|
||||
<string name="gpu_swizzle_normal">正常 (16 MB)</string>
|
||||
<string name="gpu_swizzle_medium">中 (32 MB)</string>
|
||||
<string name="gpu_swizzle_high">高 (64 MB)</string>
|
||||
|
||||
<!-- GPU swizzle chunks -->
|
||||
<string name="gpu_swizzlechunk_verylow">极低 (32)</string>
|
||||
<string name="gpu_swizzlechunk_low">低 (64)</string>
|
||||
<string name="gpu_swizzlechunk_normal">正常 (128)</string>
|
||||
<string name="gpu_swizzlechunk_medium">中 (256)</string>
|
||||
<string name="gpu_swizzlechunk_high">高 (512)</string>
|
||||
|
||||
<!-- Temperature Units -->
|
||||
<string name="temperature_celsius">摄氏度</string>
|
||||
|
||||
@@ -555,6 +555,8 @@
|
||||
<string name="renderer_force_max_clock_description">強制 GPU 以可能的最大時脈執行 (熱溫限制仍會被套用)</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">GPU 非同步模擬</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">此修改可透過使用 GPU 非同步模擬來提升性能,不過可能會導致圖形問題且會提高在進行時序相關操作時當機的機率</string>
|
||||
<string name="renderer_async_presentation">非同步呈現</string>
|
||||
<string name="renderer_async_presentation_description">此修改可以透過將渲染移至單獨的 CPU 執行緒來提升性能,但可能會導致圖形問題</string>
|
||||
<string name="renderer_reactive_flushing">使用重新啟用排清</string>
|
||||
<string name="renderer_reactive_flushing_description">犧牲效能,以改善部分遊戲的轉譯準確度</string>
|
||||
<string name="enable_buffer_history">啟用緩衝區歷史</string>
|
||||
@@ -578,6 +580,17 @@
|
||||
<string name="rescale_hack_description">透過快速重新縮放處理來啟用遊戲重新縮放設定階段的舊版處理方式</string>
|
||||
<string name="renderer_asynchronous_shaders">使用非同步著色器</string>
|
||||
<string name="renderer_asynchronous_shaders_description">使用非同步編譯著色器。這可能會減少卡頓,但也可能導致圖形錯誤。</string>
|
||||
<string name="gpu_unswizzle_settings">GPU Unswizzle 設定</string>
|
||||
<string name="gpu_unswizzle_settings_description">設定基於 GPU 的 Unswizzling 參數或完全停用此功能。調整此設定以在性能與紋理載入品質中取得平衡</string>
|
||||
<string name="gpu_unswizzle_enable">啟用 GPU Unswizzle</string>
|
||||
<string name="gpu_unswizzle_disabled">停用</string>
|
||||
<string name="gpu_unswizzle_texture_size">GPU Unswizzle 最大紋理尺寸</string>
|
||||
<string name="gpu_unswizzle_texture_size_description">設定 GPU 紋理 Unswizzling 的最大值 (MB)。雖然 GPU 處理中型和大型紋理的速度較快,但 CPU 處理非常小的紋理可能更有效率。整此設定以在 GPU 加速與 CPU 負載之間取得平衡</string>
|
||||
<string name="gpu_unswizzle_stream_size">GPU Unswizzle 流大小</string>
|
||||
<string name="gpu_unswizzle_stream_size_description">設定每個影格處理大型紋理 Unswizzling 的資料上限。數值越高載入紋理的速度越快,但會增加畫面延遲;數值較低則能降低 GPU 負載,不過有可能導致至紋理載入不完全甚至突然出現</string>
|
||||
<string name="gpu_unswizzle_chunk_size">GPU Unswizzle 塊大小</string>
|
||||
<string name="gpu_unswizzle_chunk_size_description">設定 3D 紋理每批次處理的深度切片數量。增加此數值可提升規格較佳 GPU 的處理效率,但在較低規格的硬體上可能導致卡頓或驅動程式逾時</string>
|
||||
<string name="gpu_unswizzle_default_button">預設</string>
|
||||
|
||||
|
||||
<string name="extensions">擴充功能</string>
|
||||
|
||||
@@ -591,6 +591,54 @@
|
||||
<item>2</item>
|
||||
</integer-array>
|
||||
|
||||
<string-array name="gpuTextureSizeSwizzleEntries">
|
||||
<item>@string/gpu_texturesizeswizzle_verysmall</item>
|
||||
<item>@string/gpu_texturesizeswizzle_small</item>
|
||||
<item>@string/gpu_texturesizeswizzle_normal</item>
|
||||
<item>@string/gpu_texturesizeswizzle_large</item>
|
||||
<item>@string/gpu_texturesizeswizzle_verylarge</item>
|
||||
</string-array>
|
||||
|
||||
<integer-array name="gpuTextureSizeSwizzleValues">
|
||||
<item>0</item>
|
||||
<item>1</item>
|
||||
<item>2</item>
|
||||
<item>3</item>
|
||||
<item>4</item>
|
||||
</integer-array>
|
||||
|
||||
<string-array name="gpuSwizzleEntries">
|
||||
<item>@string/gpu_swizzle_verylow</item>
|
||||
<item>@string/gpu_swizzle_low</item>
|
||||
<item>@string/gpu_swizzle_normal</item>
|
||||
<item>@string/gpu_swizzle_medium</item>
|
||||
<item>@string/gpu_swizzle_high</item>
|
||||
</string-array>
|
||||
|
||||
<integer-array name="gpuSwizzleValues">
|
||||
<item>0</item>
|
||||
<item>1</item>
|
||||
<item>2</item>
|
||||
<item>3</item>
|
||||
<item>4</item>
|
||||
</integer-array>
|
||||
|
||||
<string-array name="gpuSwizzleChunkEntries">
|
||||
<item>@string/gpu_swizzlechunk_verylow</item>
|
||||
<item>@string/gpu_swizzlechunk_low</item>
|
||||
<item>@string/gpu_swizzlechunk_normal</item>
|
||||
<item>@string/gpu_swizzlechunk_medium</item>
|
||||
<item>@string/gpu_swizzlechunk_high</item>
|
||||
</string-array>
|
||||
|
||||
<integer-array name="gpuSwizzleChunkValues">
|
||||
<item>0</item>
|
||||
<item>1</item>
|
||||
<item>2</item>
|
||||
<item>3</item>
|
||||
<item>4</item>
|
||||
</integer-array>
|
||||
|
||||
<string-array name="temperatureUnitEntries">
|
||||
<item>@string/temperature_celsius</item>
|
||||
<item>@string/temperature_fahrenheit</item>
|
||||
|
||||
@@ -105,8 +105,6 @@
|
||||
<string name="use_sync_core_description">Synchronize the core tick speed to the maximum speed percentage to improve performance without altering the game\'s actual speed.</string>
|
||||
<string name="cpuopt_unsafe_host_mmu">Enable Host MMU Emulation</string>
|
||||
<string name="cpuopt_unsafe_host_mmu_description">This optimization speeds up memory accesses by the guest program. Enabling it causes guest memory reads/writes to be done directly into memory and make use of Host\'s MMU. Disabling this forces all memory accesses to use Software MMU Emulation.</string>
|
||||
<string name="relocate_36bit_address_space">Run 36-bit games natively</string>
|
||||
<string name="relocate_36bit_address_space_description">Gives games built for the 36-bit address space the 39-bit layout so they can run with NCE instead of the JIT. Experimental: a game that relies on 36-bit addresses can crash.</string>
|
||||
<string name="debug_knobs">Debug knobs</string>
|
||||
<string name="debug_knobs_description">For development use only.</string>
|
||||
<string name="debug_knobs_hint">0 to 65535</string>
|
||||
@@ -348,7 +346,7 @@
|
||||
<string name="frame_gen_flow_scale_auto">Match motion estimation to the game</string>
|
||||
<string name="frame_gen_flow_scale_auto_description">Estimate motion at the resolution the game actually renders instead of the upscaled output. Costs nothing in accuracy, since upscaling adds no motion detail.</string>
|
||||
<string name="frame_gen_flow_scale">Motion estimation resolution</string>
|
||||
<string name="frame_gen_flow_scale_description">Resolution of the optical flow pass, as a fraction of the game\'s render resolution. Lowering it is the cheapest way to reclaim performance.</string>
|
||||
<string name="frame_gen_flow_scale_description">Resolution of the optical flow pass, as a fraction of the output. Lowering it is the cheapest way to reclaim performance.</string>
|
||||
<string name="frame_gen_unsupported">Frame generation unavailable</string>
|
||||
<string name="frame_gen_unsupported_description">This GPU driver lacks the Vulkan memory model or half precision (float16) support that the Lossless Scaling shaders require.</string>
|
||||
<string name="lossless_scaling_setup_description">Optional. Provide your own Lossless.dll to enable frame generation later</string>
|
||||
@@ -575,8 +573,6 @@
|
||||
<string name="accelerate_astc_description">Pick how ASTC-compressed textures are decoded for rendering: CPU (slow, safe), GPU (fast, recommended), or CPU Async (no stutters, may cause issues)</string>
|
||||
|
||||
<string name="sync_memory_operations">Sync Memory Operations</string>
|
||||
<string name="stall_on_gpu_fence_wait">Pause CPU on GPU fence timeouts</string>
|
||||
<string name="stall_on_gpu_fence_wait_description">When a game keeps timing out while waiting for the GPU, pauses every CPU thread until the GPU catches up. Disabling it keeps the other threads running, which can reduce stutter but may break games that rely on it.</string>
|
||||
<string name="sync_memory_operations_description">Ensures data consistency between compute and memory operations. This option should fix issues in some games, but may also reduce performance in some cases. Unreal Engine 4 games often see the most significant changes thereof.</string>
|
||||
<string name="use_disk_shader_cache">Disk shader cache</string>
|
||||
<string name="use_disk_shader_cache_description">Reduces stuttering by locally storing and loading generated shaders.</string>
|
||||
@@ -584,15 +580,13 @@
|
||||
<string name="renderer_force_max_clock_description">Forces the GPU to run at the maximum possible clocks (thermal constraints will still be applied).</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation">GPU async emulation</string>
|
||||
<string name="renderer_asynchronous_gpu_emulation_description">This hack can increase performance by running GPU emulation asynchronously at the cost of graphical issues and increased crash rates by timing-related operations.</string>
|
||||
<string name="renderer_async_presentation">Asynchronous presentation</string>
|
||||
<string name="renderer_async_presentation_description">This hack can increase performance by moving presentation to a separate CPU thread at the cost of graphical issues.</string>
|
||||
<string name="renderer_reactive_flushing">Use reactive flushing</string>
|
||||
<string name="renderer_reactive_flushing_description">Improves rendering accuracy in some games at the cost of performance.</string>
|
||||
<string name="renderer_barrier_feedback_loops">Barrier feedback loops</string>
|
||||
<string name="renderer_barrier_feedback_loops_description">Improves rendering of transparency effects in specific games at the cost of performance.</string>
|
||||
<string name="enable_buffer_history">Enable buffer history</string>
|
||||
<string name="enable_buffer_history_description">Enables access to previous buffer states. This option may improve rendering quality and performance consistency in some games.</string>
|
||||
<string name="enable_gpu_buffer_readback">Enable GPU Buffer Readback</string>
|
||||
<string name="use_unified_memory">Unified Memory</string>
|
||||
<string name="use_unified_memory_description">Backs part of the emulated memory with GPU-shared buffers so textures are unswizzled directly from it, skipping a CPU copy. Needs 12 GB of RAM or more and a GPU with I/O coherency; it turns itself off otherwise. Takes effect on the next game launch.</string>
|
||||
<string name="enable_gpu_buffer_readback_description">Preserves GPU-modified buffer data by reading it back before uploads. Some games require this to render certain effects properly. May cause issues if the hardware cannot handle the additional workload.</string>
|
||||
<string name="use_optimized_vertex_buffers">Optimized Vertex Buffers</string>
|
||||
<string name="use_optimized_vertex_buffers_description">Enables optimized vertex buffer binding for improved performance. Requires Mesa 26.0+ Turnip drivers/ QCOM drivers. Will crash on older Turnip drivers (25.3 and below).</string>
|
||||
@@ -611,6 +605,17 @@
|
||||
<string name="rescale_hack_description">Enables a legacy handling for the rescale configuration pass for games by using a quick rescale path</string>
|
||||
<string name="renderer_asynchronous_shaders">Use asynchronous shaders</string>
|
||||
<string name="renderer_asynchronous_shaders_description">Compiles shaders asynchronously. This may reduce stutters but may also introduce glitches.</string>
|
||||
<string name="gpu_unswizzle_settings">GPU Unswizzle Settings</string>
|
||||
<string name="gpu_unswizzle_settings_description">Configure GPU-based texture unswizzling parameters or disable it entirely. Adjust these settings to balance performance and texture loading quality.</string>
|
||||
<string name="gpu_unswizzle_enable">Enable GPU Unswizzle</string>
|
||||
<string name="gpu_unswizzle_disabled">Disabled</string>
|
||||
<string name="gpu_unswizzle_texture_size">GPU Unswizzle Max Texture Size</string>
|
||||
<string name="gpu_unswizzle_texture_size_description">Sets the maximum size (MB) for GPU-based texture unswizzling. While the GPU is faster for medium and large textures, the CPU may be more efficient for very small ones. Adjust this to find the balance between GPU acceleration and CPU overhead.</string>
|
||||
<string name="gpu_unswizzle_stream_size">GPU Unswizzle Stream Size</string>
|
||||
<string name="gpu_unswizzle_stream_size_description">Sets the data limit per frame for unswizzling large textures. Higher values speed up texture loading at the cost of higher frame latency; lower values reduce GPU overhead but may cause visible texture pop-in.</string>
|
||||
<string name="gpu_unswizzle_chunk_size">GPU Unswizzle Chunk Size</string>
|
||||
<string name="gpu_unswizzle_chunk_size_description">Defines the number of depth slices processed per batch for 3D textures. Increasing this improves throughput efficiency on powerful GPUs but may cause stuttering or driver timeouts on weaker hardware.</string>
|
||||
<string name="gpu_unswizzle_default_button">Default</string>
|
||||
|
||||
|
||||
<string name="extensions">Extensions</string>
|
||||
@@ -1060,10 +1065,25 @@
|
||||
<string name="fast_gpu_high">Overclock</string>
|
||||
|
||||
<!-- GPU swizzle texture size -->
|
||||
<string name="gpu_texturesizeswizzle_verysmall">Very Small (16 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_small">Small (32 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_normal">Normal (128 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_large">Large (256 MB)</string>
|
||||
<string name="gpu_texturesizeswizzle_verylarge">Very Large (512 MB)</string>
|
||||
|
||||
<!-- GPU swizzle streams -->
|
||||
<string name="gpu_swizzle_verylow">Very Low (4 MB)</string>
|
||||
<string name="gpu_swizzle_low">Low (8 MB)</string>
|
||||
<string name="gpu_swizzle_normal">Normal (16 MB)</string>
|
||||
<string name="gpu_swizzle_medium">Medium (32 MB)</string>
|
||||
<string name="gpu_swizzle_high">High (64 MB)</string>
|
||||
|
||||
<!-- GPU swizzle chunks -->
|
||||
<string name="gpu_swizzlechunk_verylow">Very Low (32)</string>
|
||||
<string name="gpu_swizzlechunk_low">Low (64)</string>
|
||||
<string name="gpu_swizzlechunk_normal">Normal (128)</string>
|
||||
<string name="gpu_swizzlechunk_medium">Medium (256)</string>
|
||||
<string name="gpu_swizzlechunk_high">High (512)</string>
|
||||
|
||||
<!-- Temperature Units -->
|
||||
<string name="temperature_celsius">Celsius</string>
|
||||
|
||||
@@ -108,6 +108,8 @@ add_library(
|
||||
settings_setting.h
|
||||
slot_vector.h
|
||||
socket_types.h
|
||||
sparse_large_vector.cpp
|
||||
sparse_large_vector.h
|
||||
spin_lock.h
|
||||
stb.cpp
|
||||
stb.h
|
||||
@@ -137,8 +139,6 @@ add_library(
|
||||
uuid.cpp
|
||||
uuid.h
|
||||
vector_math.h
|
||||
virtual_buffer.cpp
|
||||
virtual_buffer.h
|
||||
zstd_compression.cpp
|
||||
zstd_compression.h
|
||||
fs/ryujinx_compat.h fs/ryujinx_compat.cpp
|
||||
|
||||
@@ -9,7 +9,6 @@
|
||||
|
||||
#include "common/assert.h"
|
||||
#include "common/fiber.h"
|
||||
#include "common/virtual_buffer.h"
|
||||
|
||||
#include <boost/context/detail/fcontext.hpp>
|
||||
|
||||
|
||||
+40
-12
@@ -178,6 +178,14 @@ public:
|
||||
Release();
|
||||
}
|
||||
|
||||
void* Allocate(size_t size) {
|
||||
auto* ptr = VirtualAlloc(nullptr, size, MEM_RESERVE | MEM_COMMIT, PAGE_READWRITE);
|
||||
if (ptr == nullptr) {
|
||||
LOG_CRITICAL(HW_Memory, "Failed to allocate fallback buffer with size {:#x}, error {}", size, GetLastError());
|
||||
}
|
||||
return ptr;
|
||||
}
|
||||
|
||||
void Map(size_t virtual_offset, size_t host_offset, size_t length, MemoryPermission perms) {
|
||||
std::unique_lock lock{placeholder_mutex};
|
||||
if (!IsNiechePlaceholder(virtual_offset, length)) {
|
||||
@@ -398,6 +406,10 @@ private:
|
||||
// For managarm: see https://github.com/managarm/managarm/issues/1370
|
||||
#else // ^^^ Windows ^^^ vvv POSIX vvv
|
||||
|
||||
#ifndef MAP_NOCORE
|
||||
#define MAP_NOCORE 0
|
||||
#endif
|
||||
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
|
||||
static void* ChooseVirtualBase(size_t virtual_size) {
|
||||
@@ -422,7 +434,7 @@ static void* ChooseVirtualBase(size_t virtual_size) {
|
||||
// Note: we may be able to take advantage of MAP_FIXED_NOREPLACE here.
|
||||
void* map_pointer =
|
||||
mmap(reinterpret_cast<void*>(hint_address), virtual_size, PROT_READ | PROT_WRITE,
|
||||
MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1, 0);
|
||||
MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE | MAP_NOCORE, -1, 0);
|
||||
|
||||
// If we successfully mapped, we're done.
|
||||
if (reinterpret_cast<uintptr_t>(map_pointer) == hint_address) {
|
||||
@@ -442,11 +454,11 @@ static void* ChooseVirtualBase(size_t virtual_size) {
|
||||
|
||||
static void* ChooseVirtualBase(size_t virtual_size) {
|
||||
#if defined(__FreeBSD__) || defined(__DragonFly__) || defined(__OpenBSD__) || defined(__sun__) || defined(__HAIKU__) || defined(__managarm__) || defined(__AIX__)
|
||||
void* virtual_base = mmap(nullptr, virtual_size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE | MAP_ALIGNED_SUPER, -1, 0);
|
||||
void* virtual_base = mmap(nullptr, virtual_size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE | MAP_ALIGNED_SUPER | MAP_NOCORE, -1, 0);
|
||||
if (virtual_base != MAP_FAILED)
|
||||
return virtual_base;
|
||||
#endif
|
||||
return mmap(nullptr, virtual_size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1, 0);
|
||||
return mmap(nullptr, virtual_size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE | MAP_NOCORE, -1, 0);
|
||||
}
|
||||
|
||||
#endif
|
||||
@@ -540,13 +552,13 @@ public:
|
||||
}
|
||||
if (use_anon) {
|
||||
LOG_WARNING(Common_Memory, "Using private mappings instead of shared ones");
|
||||
backing_base = static_cast<u8*>(mmap(nullptr, backing_size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE, -1, 0));
|
||||
backing_base = static_cast<u8*>(mmap(nullptr, backing_size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE | MAP_NOCORE, -1, 0));
|
||||
if (fd > 0) {
|
||||
fd = -1;
|
||||
close(fd);
|
||||
}
|
||||
} else {
|
||||
backing_base = static_cast<u8*>(mmap(nullptr, backing_size, PROT_READ | PROT_WRITE, MAP_SHARED, fd, 0));
|
||||
backing_base = static_cast<u8*>(mmap(nullptr, backing_size, PROT_READ | PROT_WRITE, MAP_SHARED | MAP_NOCORE, fd, 0));
|
||||
}
|
||||
if (backing_base == MAP_FAILED) {
|
||||
LOG_CRITICAL(HW_Memory, "mmap failed: {}", strerror(errno));
|
||||
@@ -570,6 +582,14 @@ public:
|
||||
Release();
|
||||
}
|
||||
|
||||
void* Allocate(size_t size) {
|
||||
auto* ptr = mmap(nullptr, size, PROT_READ | PROT_WRITE, MAP_ANONYMOUS | MAP_PRIVATE, -1, 0);
|
||||
if (ptr == MAP_FAILED) {
|
||||
LOG_CRITICAL(HW_Memory, "Failed to allocate fallback buffer with size {:#x}, {}", size, strerror(errno));
|
||||
}
|
||||
return ptr;
|
||||
}
|
||||
|
||||
void Map(size_t virtual_offset, size_t host_offset, size_t length, MemoryPermission perms) {
|
||||
// Intersect the range with our address space.
|
||||
AdjustMap(&virtual_offset, &length);
|
||||
@@ -690,12 +710,10 @@ HostMemory::HostMemory(size_t backing_size_, size_t virtual_size_)
|
||||
{
|
||||
#if defined(__OPENORBIS__) || defined(__managarm__)
|
||||
LOG_WARNING(HW_Memory, "Platform doesn't support fastmem");
|
||||
fallback_buffer.emplace(backing_size);
|
||||
backing_base = fallback_buffer->data();
|
||||
backing_base = static_cast<u8*>(mmap(nullptr, backing_size, PROT_READ | PROT_WRITE, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
|
||||
virtual_base = nullptr;
|
||||
#else
|
||||
// Try to allocate a fastmem arena.
|
||||
// The implementation will fail with std::bad_alloc on errors.
|
||||
impl = std::make_unique<HostMemory::Impl>(AlignUp(backing_size, PageAlignment), AlignUp(virtual_size, PageAlignment) + HugePageSize);
|
||||
if (impl->Init()) {
|
||||
backing_base = impl->backing_base;
|
||||
@@ -706,16 +724,26 @@ HostMemory::HostMemory(size_t backing_size_, size_t virtual_size_)
|
||||
virtual_base_offset = virtual_base - impl->virtual_base;
|
||||
}
|
||||
} else {
|
||||
impl.reset();
|
||||
LOG_WARNING(HW_Memory, "Platform can support fastmem, but can't create it");
|
||||
fallback_buffer.emplace(backing_size);
|
||||
backing_base = fallback_buffer->data();
|
||||
fallback_buffer = true;
|
||||
backing_base = static_cast<u8*>(impl->Allocate(backing_size));
|
||||
virtual_base = nullptr;
|
||||
impl.reset();
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
HostMemory::~HostMemory() = default;
|
||||
HostMemory::~HostMemory() {
|
||||
#ifdef _WIN32
|
||||
if (fallback_buffer) {
|
||||
VirtualFree(backing_base, backing_size, MEM_RELEASE);
|
||||
}
|
||||
#else
|
||||
if (fallback_buffer) {
|
||||
munmap(backing_base, backing_size);
|
||||
}
|
||||
#endif
|
||||
}
|
||||
|
||||
HostMemory::HostMemory(HostMemory&&) noexcept = default;
|
||||
|
||||
|
||||
@@ -10,7 +10,6 @@
|
||||
#include <optional>
|
||||
#include "common/common_funcs.h"
|
||||
#include "common/common_types.h"
|
||||
#include "common/virtual_buffer.h"
|
||||
|
||||
namespace Common {
|
||||
|
||||
@@ -86,7 +85,7 @@ private:
|
||||
u8* virtual_base{};
|
||||
size_t virtual_base_offset{};
|
||||
// Windows requires it for kernels whom lack proper support for some functions!
|
||||
std::optional<Common::VirtualBuffer<u8>> fallback_buffer;
|
||||
bool fallback_buffer{false};
|
||||
};
|
||||
|
||||
} // namespace Common
|
||||
|
||||
@@ -13,39 +13,11 @@ PageTable::PageTable() = default;
|
||||
|
||||
PageTable::~PageTable() noexcept = default;
|
||||
|
||||
bool PageTable::BeginTraversal(TraversalEntry* out_entry, TraversalContext* out_context,
|
||||
Common::ProcessAddress address) const {
|
||||
out_context->next_offset = GetInteger(address);
|
||||
out_context->next_page = address / page_size;
|
||||
|
||||
return this->ContinueTraversal(out_entry, out_context);
|
||||
}
|
||||
|
||||
bool PageTable::ContinueTraversal(TraversalEntry* out_entry, TraversalContext* context) const {
|
||||
// Setup invalid defaults.
|
||||
out_entry->phys_addr = 0;
|
||||
out_entry->block_size = page_size;
|
||||
// Validate that we can read the actual entry.
|
||||
if (auto const page = context->next_page; page < entries.size()) {
|
||||
// Validate that the entry is mapped.
|
||||
if (auto const paddr = entries[page].addr; paddr != 0) {
|
||||
// Populate the results.
|
||||
out_entry->phys_addr = paddr + context->next_offset;
|
||||
context->next_page += 1;
|
||||
context->next_offset += page_size;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
context->next_page += 1;
|
||||
context->next_offset += page_size;
|
||||
return false;
|
||||
}
|
||||
|
||||
void PageTable::Resize(std::size_t address_space_width_in_bits, std::size_t page_size_in_bits) {
|
||||
auto const num_page_table_entries = 1ULL << (address_space_width_in_bits - page_size_in_bits);
|
||||
entries.resize(num_page_table_entries);
|
||||
void PageTable::Resize(std::size_t address_space_width_in_bits, std::size_t page_bits) {
|
||||
auto const num_page_table_entries = 1ULL << (address_space_width_in_bits - page_bits);
|
||||
entries.ResizeAndClear(num_page_table_entries);
|
||||
current_address_space_width_in_bits = address_space_width_in_bits;
|
||||
page_size = 1ULL << page_size_in_bits;
|
||||
current_page_bits = page_bits;
|
||||
}
|
||||
|
||||
} // namespace Common
|
||||
|
||||
+64
-56
@@ -9,22 +9,22 @@
|
||||
#include <atomic>
|
||||
|
||||
#include "common/common_types.h"
|
||||
#include "common/sparse_large_vector.h"
|
||||
#include "common/typed_address.h"
|
||||
#include "common/virtual_buffer.h"
|
||||
|
||||
namespace Common {
|
||||
|
||||
enum class PageType : u8 {
|
||||
/// Page is unmapped and should cause an access error.
|
||||
Unmapped,
|
||||
Unmapped = 0b00,
|
||||
/// Page is mapped to regular memory. This is the only type you can get pointers to.
|
||||
Memory,
|
||||
Memory = 0b01,
|
||||
/// Page is mapped to regular memory, but inaccessible from CPU fastmem and must use
|
||||
/// the callbacks.
|
||||
DebugMemory,
|
||||
DebugMemory = 0b10,
|
||||
/// Page is mapped to regular memory, but also needs to check for rasterizer cache flushing and
|
||||
/// invalidation
|
||||
RasterizerCachedMemory,
|
||||
RasterizerCachedMemory = 0b11,
|
||||
};
|
||||
|
||||
/**
|
||||
@@ -42,57 +42,86 @@ struct PageTable {
|
||||
u64 next_offset{};
|
||||
};
|
||||
|
||||
/// Number of bits reserved for attribute tagging.
|
||||
/// This can be at most the guaranteed alignment of the pointers in the page table.
|
||||
static constexpr int ATTRIBUTE_BITS = 2;
|
||||
/// Masks out bits reserved for attribute tagging.
|
||||
static constexpr u64 ATTRIBUTE_MASK = ((1ULL << 44) - 1) << 12;
|
||||
|
||||
/// Specifies sign bit for page table entries.
|
||||
static constexpr u64 SIGN_BIT = 45 + 12; // 44 bits of data + page offset
|
||||
|
||||
/**
|
||||
* Pair of host pointer and page type attribute.
|
||||
* This uses the lower bits of a given pointer to store the attribute tag.
|
||||
* Atomic tuple of host pointer, page type, and block id.
|
||||
* This uses the lower bits of a given pointer to store the attributes.
|
||||
* Writing and reading the pointer attribute pair is guaranteed to be atomic for the same method
|
||||
* call. In other words, they are guaranteed to be synchronized at all times.
|
||||
*/
|
||||
class PageInfo {
|
||||
class PageEntryData {
|
||||
public:
|
||||
struct Data {
|
||||
Data(bool marked_, PageType type_, u16 block_, u64 page_)
|
||||
: marked(static_cast<u64>(marked_) & 0b1)
|
||||
, type(static_cast<u64>(type_) & ((1ULL << 2) - 1))
|
||||
, block(static_cast<u64>(block_) & ((1ULL << 9) - 1))
|
||||
, page((page_ >> 12) & ((1ULL << 45) - 1))
|
||||
, block2((static_cast<u64>(block_) >> 9) & ((1ULL << 7) - 1)) {}
|
||||
u64 marked : 1;
|
||||
u64 type : 2;
|
||||
u64 block : 9;
|
||||
u64 page : 45; // 44 bits of actual data (64 - page offset (12) - reserved (8)) + a sign bit
|
||||
u64 block2 : 7;
|
||||
};
|
||||
|
||||
[[nodiscard]] Data Raw() const noexcept {
|
||||
return std::bit_cast<Data>(data_raw.load(std::memory_order_relaxed));
|
||||
}
|
||||
|
||||
/// Returns the page pointer
|
||||
[[nodiscard]] uintptr_t Pointer() const noexcept {
|
||||
return ExtractPointer(raw.load(std::memory_order_relaxed));
|
||||
[[nodiscard]] uintptr_t Pointer(bool ignored_marked = false) const noexcept {
|
||||
return ExtractPointer(std::bit_cast<Data>(data_raw.load(std::memory_order_relaxed)), ignored_marked);
|
||||
}
|
||||
|
||||
/// Returns the page type attribute
|
||||
[[nodiscard]] PageType Type() const noexcept {
|
||||
return ExtractType(raw.load(std::memory_order_relaxed));
|
||||
return static_cast<PageType>(std::bit_cast<Data>(data_raw.load(std::memory_order_relaxed)).type);
|
||||
}
|
||||
|
||||
/// Returns the block identifier.
|
||||
[[nodiscard]] u16 Block() const noexcept {
|
||||
return ExtractBlock(std::bit_cast<Data>(data_raw.load(std::memory_order_relaxed)));
|
||||
}
|
||||
|
||||
/// Returns the page pointer and attribute pair, extracted from the same atomic read
|
||||
[[nodiscard]] std::pair<uintptr_t, PageType> PointerType() const noexcept {
|
||||
const uintptr_t non_atomic_raw = raw.load(std::memory_order_relaxed);
|
||||
return {ExtractPointer(non_atomic_raw), ExtractType(non_atomic_raw)};
|
||||
[[nodiscard]] std::tuple<uintptr_t, PageType, u16> PointerTypeBlock(bool ignore_marked = false) const noexcept {
|
||||
const auto non_atomic_raw = std::bit_cast<Data>(data_raw.load(std::memory_order_relaxed));
|
||||
return {ExtractPointer(non_atomic_raw, ignore_marked), static_cast<PageType>(non_atomic_raw.type), ExtractBlock(non_atomic_raw)};
|
||||
}
|
||||
|
||||
/// Returns the raw representation of the page information.
|
||||
/// Use ExtractPointer and ExtractType to unpack the value.
|
||||
[[nodiscard]] uintptr_t Raw() const noexcept {
|
||||
return raw.load(std::memory_order_relaxed);
|
||||
/// Write page info atomically
|
||||
constexpr void Store(bool marked, PageType type, u16 block, uintptr_t pointer) noexcept {
|
||||
data_raw.store(std::bit_cast<u64>(Data{marked, type, block, pointer}));
|
||||
}
|
||||
|
||||
/// Write a page pointer and type pair atomically
|
||||
void Store(uintptr_t pointer, PageType type) noexcept {
|
||||
raw.store(pointer | uintptr_t(type));
|
||||
constexpr void MarkRasterizerCached() noexcept {
|
||||
data_raw.fetch_or(0b111);
|
||||
}
|
||||
|
||||
constexpr void MarkDebug(u64 ptr, u16 block) noexcept {
|
||||
Store(true, PageType::DebugMemory, block, ptr);
|
||||
}
|
||||
|
||||
/// Unpack a pointer from a page info raw representation
|
||||
[[nodiscard]] static uintptr_t ExtractPointer(uintptr_t raw) noexcept {
|
||||
return raw & (~uintptr_t{0} << ATTRIBUTE_BITS);
|
||||
[[nodiscard]] static uintptr_t ExtractPointer(Data raw, bool ignore_marked = false) noexcept {
|
||||
return raw.marked && !ignore_marked ? 0
|
||||
// shift raw.page's fake sign bit to the actual sign bit, then sign extend
|
||||
: ((s64)(raw.page << (64 - 44))) >> (64 - 44 - 12);
|
||||
}
|
||||
|
||||
/// Unpack a page type from a page info raw representation
|
||||
[[nodiscard]] static PageType ExtractType(uintptr_t raw) noexcept {
|
||||
return static_cast<PageType>(raw & ((uintptr_t{1} << ATTRIBUTE_BITS) - 1));
|
||||
[[nodiscard]] static u16 ExtractBlock(Data raw) noexcept {
|
||||
return static_cast<u16>(raw.block | (raw.block2 << 9));
|
||||
}
|
||||
|
||||
private:
|
||||
std::atomic<uintptr_t> raw;
|
||||
std::atomic<u64> data_raw;
|
||||
static_assert(sizeof(Data) == sizeof(std::atomic<u64>));
|
||||
};
|
||||
|
||||
PageTable();
|
||||
@@ -100,13 +129,8 @@ struct PageTable {
|
||||
|
||||
PageTable(const PageTable&) = delete;
|
||||
PageTable& operator=(const PageTable&) = delete;
|
||||
|
||||
PageTable(PageTable&&) noexcept = default;
|
||||
PageTable& operator=(PageTable&&) noexcept = default;
|
||||
|
||||
bool BeginTraversal(TraversalEntry* out_entry, TraversalContext* out_context,
|
||||
Common::ProcessAddress address) const;
|
||||
bool ContinueTraversal(TraversalEntry* out_entry, TraversalContext* context) const;
|
||||
PageTable(PageTable&&) noexcept = delete;
|
||||
PageTable& operator=(PageTable&&) noexcept = delete;
|
||||
|
||||
/**
|
||||
* Resizes the page table to be able to accommodate enough pages within
|
||||
@@ -121,30 +145,14 @@ struct PageTable {
|
||||
return current_address_space_width_in_bits;
|
||||
}
|
||||
|
||||
bool GetPhysicalAddress(Common::PhysicalAddress* out_phys_addr,
|
||||
Common::ProcessAddress virt_addr) const {
|
||||
if (virt_addr > (1ULL << this->GetAddressSpaceBits())) {
|
||||
return false;
|
||||
}
|
||||
|
||||
*out_phys_addr = entries[virt_addr / page_size].addr + GetInteger(virt_addr);
|
||||
return true;
|
||||
}
|
||||
|
||||
/// Vector of memory pointers backing each page. An entry can only be non-null if the
|
||||
/// corresponding attribute element is of type `Memory`.
|
||||
struct PageEntryData {
|
||||
PageInfo ptr;
|
||||
u64 block;
|
||||
u64 addr;
|
||||
u64 padding;
|
||||
};
|
||||
VirtualBuffer<PageEntryData> entries;
|
||||
static_assert(sizeof(PageEntryData) == 32);
|
||||
SparseLargeVector<PageEntryData> entries;
|
||||
static_assert(sizeof(PageEntryData) == 8);
|
||||
|
||||
u8* fastmem_arena{};
|
||||
std::size_t current_address_space_width_in_bits{};
|
||||
std::size_t page_size{};
|
||||
std::size_t current_page_bits{};
|
||||
};
|
||||
|
||||
} // namespace Common
|
||||
|
||||
@@ -48,6 +48,7 @@ SWITCHABLE(AnisotropyMode, true);
|
||||
SWITCHABLE(AntiAliasing, false);
|
||||
SWITCHABLE(AspectRatio, true);
|
||||
SWITCHABLE(AstcDecodeMode, true);
|
||||
SWITCHABLE(AstcRecompression, true);
|
||||
SWITCHABLE(AudioMode, true);
|
||||
SWITCHABLE(CpuBackend, true);
|
||||
SWITCHABLE(CpuAccuracy, true);
|
||||
|
||||
+40
-14
@@ -65,6 +65,7 @@ SWITCHABLE(AnisotropyMode, true);
|
||||
SWITCHABLE(AntiAliasing, false);
|
||||
SWITCHABLE(AspectRatio, true);
|
||||
SWITCHABLE(AstcDecodeMode, true);
|
||||
SWITCHABLE(AstcRecompression, true);
|
||||
SWITCHABLE(AudioMode, true);
|
||||
SWITCHABLE(CpuBackend, true);
|
||||
SWITCHABLE(CpuAccuracy, true);
|
||||
@@ -259,8 +260,6 @@ struct Values {
|
||||
Category::Cpu};
|
||||
SwitchableSetting<CpuAccuracy, true> cpu_accuracy{linkage, CpuAccuracy::Auto,
|
||||
"cpu_accuracy", Category::Cpu};
|
||||
SwitchableSetting<bool> relocate_36bit_address_space{
|
||||
linkage, false, "relocate_36bit_address_space", Category::Cpu};
|
||||
SwitchableSetting<CpuClock> cpu_clock{linkage,
|
||||
CpuClock::Normal,
|
||||
"fast_cpu_time",
|
||||
@@ -554,6 +553,22 @@ struct Values {
|
||||
"accelerate_astc",
|
||||
Category::RendererAdvanced};
|
||||
|
||||
SwitchableSetting<FramePacingMode, true> frame_pacing_mode{linkage,
|
||||
FramePacingMode::Target_Auto,
|
||||
FramePacingMode::Target_Auto,
|
||||
FramePacingMode::Target_120,
|
||||
"frame_pacing_mode",
|
||||
Category::RendererAdvanced,
|
||||
Specialization::Default,
|
||||
true,
|
||||
true};
|
||||
|
||||
SwitchableSetting<AstcRecompression, true> astc_recompression{linkage,
|
||||
AstcRecompression::Uncompressed,
|
||||
"astc_recompression",
|
||||
Category::RendererAdvanced};
|
||||
|
||||
|
||||
SwitchableSetting<bool> sync_memory_operations{linkage,
|
||||
false,
|
||||
"sync_memory_operations",
|
||||
@@ -561,8 +576,6 @@ struct Values {
|
||||
Specialization::Default,
|
||||
true,
|
||||
true};
|
||||
SwitchableSetting<bool> stall_on_gpu_fence_wait{linkage, true, "stall_on_gpu_fence_wait",
|
||||
Category::RendererAdvanced};
|
||||
|
||||
SwitchableSetting<bool> renderer_force_max_clock{linkage, false, "force_max_clock",
|
||||
Category::RendererAdvanced};
|
||||
@@ -589,13 +602,8 @@ struct Values {
|
||||
"use_reactive_flushing",
|
||||
Category::RendererAdvanced};
|
||||
|
||||
SwitchableSetting<bool> barrier_feedback_loops{linkage,
|
||||
true,
|
||||
"barrier_feedback_loops",
|
||||
Category::RendererAdvanced,
|
||||
Specialization::Default,
|
||||
true,
|
||||
true};
|
||||
SwitchableSetting<bool> barrier_feedback_loops{linkage, true, "barrier_feedback_loops",
|
||||
Category::RendererAdvanced};
|
||||
|
||||
SwitchableSetting<bool> enable_buffer_history{linkage,
|
||||
false,
|
||||
@@ -633,7 +641,7 @@ struct Values {
|
||||
true};
|
||||
SwitchableSetting<bool> async_presentation{linkage,
|
||||
#ifdef __ANDROID__
|
||||
true,
|
||||
false,
|
||||
#else
|
||||
false,
|
||||
#endif
|
||||
@@ -658,8 +666,26 @@ struct Values {
|
||||
SwitchableSetting<bool> use_asynchronous_shaders{linkage, false, "use_asynchronous_shaders",
|
||||
Category::RendererHacks};
|
||||
|
||||
SwitchableSetting<bool> use_unified_memory{linkage, false, "use_unified_memory",
|
||||
Category::RendererHacks};
|
||||
SwitchableSetting<GpuUnswizzleSize> gpu_unswizzle_texture_size{linkage,
|
||||
GpuUnswizzleSize::Large,
|
||||
"gpu_unswizzle_texture_size",
|
||||
Category::RendererHacks,
|
||||
Specialization::Default};
|
||||
|
||||
SwitchableSetting<GpuUnswizzle> gpu_unswizzle_stream_size{linkage,
|
||||
GpuUnswizzle::Medium,
|
||||
"gpu_unswizzle_stream_size",
|
||||
Category::RendererHacks,
|
||||
Specialization::Default};
|
||||
|
||||
SwitchableSetting<GpuUnswizzleChunk> gpu_unswizzle_chunk_size{linkage,
|
||||
GpuUnswizzleChunk::Medium,
|
||||
"gpu_unswizzle_chunk_size",
|
||||
Category::RendererHacks,
|
||||
Specialization::Default};
|
||||
|
||||
SwitchableSetting<bool> gpu_unswizzle_enabled{linkage, false, "gpu_unswizzle_enabled",
|
||||
Category::RendererHacks};
|
||||
|
||||
SwitchableSetting<ExtendedDynamicState> dyna_state{linkage,
|
||||
#if defined(__ANDROID__)
|
||||
|
||||
@@ -130,6 +130,8 @@ ENUM(TimeZone, Auto, Default, Cet, Cst6Cdt, Cuba, Eet, Egypt, Eire, Est, Est5Edt
|
||||
Roc, Rok, Singapore, Turkey, Uct, Universal, Utc, WSu, Wet, Zulu);
|
||||
ENUM(AnisotropyMode, Automatic, Default, X2, X4, X8, X16);
|
||||
ENUM(AstcDecodeMode, Cpu, Gpu, CpuAsynchronous);
|
||||
ENUM(AstcRecompression, Uncompressed, Bc1, Bc3);
|
||||
ENUM(FramePacingMode, Target_Auto, Target_30, Target_60, Target_90, Target_120);
|
||||
ENUM(VSyncMode, Immediate, Mailbox, Fifo, FifoRelaxed);
|
||||
ENUM(VramUsageMode, Conservative, Aggressive);
|
||||
ENUM(RendererBackend, OpenGL_GLSL, Vulkan, Null, OpenGL_GLASM, OpenGL_SPIRV);
|
||||
@@ -151,6 +153,9 @@ ENUM(ConsoleMode, Handheld, Docked);
|
||||
ENUM(AppletMode, HLE, LLE);
|
||||
ENUM(SpirvOptimizeMode, Never, OnLoad, Always);
|
||||
ENUM(GpuClock, Normal, Boost, Overclock)
|
||||
ENUM(GpuUnswizzleSize, VerySmall, Small, Normal, Large, VeryLarge)
|
||||
ENUM(GpuUnswizzle, VeryLow, Low, Normal, Medium, High)
|
||||
ENUM(GpuUnswizzleChunk, VeryLow, Low, Normal, Medium, High)
|
||||
ENUM(TemperatureUnits, Celsius, Fahrenheit)
|
||||
ENUM(ExtendedDynamicState, Disabled, EDS1, EDS2, EDS3);
|
||||
ENUM(GpuLogLevel, Off, Errors, Standard, Verbose, All)
|
||||
|
||||
@@ -0,0 +1,143 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
/* virtual_buffer.cpp */
|
||||
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#ifdef _WIN32
|
||||
#include <windows.h>
|
||||
#include <mutex>
|
||||
#else
|
||||
#include <sys/mman.h>
|
||||
#endif
|
||||
|
||||
#include "common/alignment.h"
|
||||
#include "common/assert.h"
|
||||
#include "common/sparse_large_vector.h"
|
||||
|
||||
namespace Common {
|
||||
|
||||
#ifdef _WIN32
|
||||
static std::vector<std::pair<u64, u64>> vector_regions {};
|
||||
|
||||
// Workaround for handling non-commited memory accessed by Dynarmic; usually result of an error
|
||||
static LONG WINAPI FakePageFaultHandler(PEXCEPTION_POINTERS info) {
|
||||
DWORD code = info->ExceptionRecord->ExceptionCode;
|
||||
u64 exception_addr = reinterpret_cast<u64>(info->ExceptionRecord->ExceptionAddress);
|
||||
|
||||
if (code != EXCEPTION_ACCESS_VIOLATION) {
|
||||
// Not our problem
|
||||
return EXCEPTION_CONTINUE_SEARCH;
|
||||
}
|
||||
|
||||
u64 addr = 0, addr2 = 0;
|
||||
|
||||
for (auto region: vector_regions) {
|
||||
auto addr_shifted = exception_addr >> HostPageBits;
|
||||
if (region.first <= addr_shifted && addr_shifted <= region.second) {
|
||||
addr = addr_shifted;
|
||||
}
|
||||
|
||||
// Page-boundary accesses
|
||||
if (auto addr_ = (exception_addr + 0x40) >> HostPageBits; addr_ != addr_shifted && region.first <= addr_ && addr_ <= region.second) {
|
||||
addr2 = addr_;
|
||||
}
|
||||
|
||||
if (addr != 0 || addr2 != 0) {
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
if (addr == 0 && addr2 == 0) {
|
||||
// Not our problem
|
||||
return EXCEPTION_CONTINUE_SEARCH;
|
||||
}
|
||||
|
||||
LOG_ERROR(HW_Memory, "Accessing an unallocated region of a SparseLargeVector at {:#x}; this shouldn't happen and is likely a Dynarmic error!", exception_addr);
|
||||
|
||||
// Commit this region
|
||||
if (addr != 0) {
|
||||
if (!CommitVectorPage(addr << HostPageBits, false)) {
|
||||
return EXCEPTION_CONTINUE_SEARCH;
|
||||
}
|
||||
}
|
||||
// Commit next region if needed
|
||||
if (addr2 != 0) {
|
||||
if (!CommitVectorPage(addr2 << HostPageBits, false)) {
|
||||
return EXCEPTION_CONTINUE_SEARCH;
|
||||
}
|
||||
}
|
||||
|
||||
return EXCEPTION_CONTINUE_EXECUTION;
|
||||
}
|
||||
|
||||
bool CommitVectorPage(uintptr_t addr, bool write) noexcept {
|
||||
MEMORY_BASIC_INFORMATION info {};
|
||||
auto res = VirtualQuery(reinterpret_cast<void*>(addr), &info, sizeof(info));
|
||||
if (res == 0) {
|
||||
LOG_CRITICAL(HW_Memory, "Failed to query large buffer region at {:#x} with error {}, will try committing anyway", addr, GetLastError());
|
||||
} else if (info.State != MEM_RESERVE) {
|
||||
LOG_ERROR(HW_Memory, "Tried to commit an unreserved large buffer region at {:#x} that is not mapped or is already committed (state {:#x})", addr, info.State);
|
||||
return false;
|
||||
}
|
||||
|
||||
auto perm = write ? PAGE_READWRITE : PAGE_READONLY;
|
||||
void* res2 = VirtualAlloc(reinterpret_cast<LPVOID>(addr), HostPageSize, MEM_COMMIT, perm);
|
||||
if (res2 == nullptr) {
|
||||
LOG_ERROR(HW_Memory, "Failed to commit large buffer region at {:#x}, error {}", addr, GetLastError());
|
||||
return false;
|
||||
}
|
||||
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
|
||||
#ifndef MAP_NOCORE
|
||||
#define MAP_NOCORE 0
|
||||
#endif
|
||||
|
||||
void* AllocateMemoryPages(std::size_t size) noexcept {
|
||||
if (auto page = HostPageSize; size % page != 0) {
|
||||
LOG_WARNING(HW_Memory, "Allocating unaligned large vector with size {:#x}; aligning to {} page size", size, page);
|
||||
size = AlignUp(size, page);
|
||||
}
|
||||
|
||||
#ifdef _WIN32
|
||||
// We will never use this memory entirely so instead of committing it up front let's just reserve it and commit each page individually
|
||||
void* base = VirtualAlloc(nullptr, size, MEM_RESERVE, PAGE_READWRITE);
|
||||
|
||||
if (base != nullptr) {
|
||||
vector_regions.emplace_back(reinterpret_cast<u64>(base), reinterpret_cast<u64>(base) + size);
|
||||
|
||||
static std::once_flag flag;
|
||||
std::call_once(flag, []() { AddVectoredExceptionHandler(1, FakePageFaultHandler); });
|
||||
} else {
|
||||
// Try committing everything instead??
|
||||
LOG_WARNING(HW_Memory, "Failed to reserve large vector region with error {}, trying to commit instead..", GetLastError());
|
||||
base = VirtualAlloc(nullptr, size, MEM_COMMIT, PAGE_READWRITE);
|
||||
}
|
||||
ASSERT_MSG(base, "Failed to reserve {:#x} sized region with error {}", size, GetLastError());
|
||||
#else
|
||||
void* base = mmap(nullptr, size, PROT_READ, MAP_ANON | MAP_PRIVATE | MAP_NOCORE, -1, 0);
|
||||
if (base == MAP_FAILED)
|
||||
base = nullptr;
|
||||
ASSERT_MSG(base, "Failed to allocate {:#x} sized region with error {}", size, strerror(errno));
|
||||
#endif
|
||||
return base;
|
||||
}
|
||||
|
||||
void FreeMemoryPages(void* base, [[maybe_unused]] std::size_t size) noexcept {
|
||||
if (auto page = HostPageSize; size % page != 0) {
|
||||
size = AlignUp(size, page);
|
||||
}
|
||||
if (!base)
|
||||
return;
|
||||
#ifdef _WIN32
|
||||
ASSERT(VirtualFree(base, 0, MEM_RELEASE));
|
||||
#else
|
||||
ASSERT(munmap(base, size) == 0);
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace Common
|
||||
@@ -0,0 +1,194 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
/* virtual_buffer.h */
|
||||
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <atomic>
|
||||
#include <bit>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <unistd.h>
|
||||
#include <sys/mman.h>
|
||||
#endif
|
||||
|
||||
#include "common/alignment.h"
|
||||
#include "common/assert.h"
|
||||
|
||||
namespace Common {
|
||||
|
||||
#ifdef _WIN32
|
||||
constexpr u64 HostPageSize = 0x1000;
|
||||
constexpr u64 HostPageBits = 12;
|
||||
constexpr u64 HostPageMask = ~(HostPageSize - 1);
|
||||
bool CommitVectorPage(uintptr_t addr, bool write) noexcept;
|
||||
#else
|
||||
const u64 HostPageSize = sysconf(_SC_PAGESIZE);
|
||||
const u64 HostPageBits = std::countr_zero(HostPageSize);
|
||||
const u64 HostPageMask = ~(HostPageSize - 1);
|
||||
#endif
|
||||
|
||||
void* AllocateMemoryPages(std::size_t size) noexcept;
|
||||
void FreeMemoryPages(void* base, std::size_t size) noexcept;
|
||||
|
||||
/// A large page-aligned buffer that has optimized memory usage for zero-writes.
|
||||
template <typename T>
|
||||
requires std::is_trivially_copyable_v<T>
|
||||
class SparseLargeVector final {
|
||||
public:
|
||||
constexpr SparseLargeVector() = default;
|
||||
|
||||
explicit SparseLargeVector(std::size_t count) noexcept
|
||||
: alloc_size{count * sizeof(T)}
|
||||
{
|
||||
base_ptr = static_cast<T*>(AllocateMemoryPages(alloc_size));
|
||||
|
||||
// each item in vector holds information for 64 pages
|
||||
auto denom = HostPageSize * 64;
|
||||
committed_pages = std::vector<std::atomic<u64>>((alloc_size + denom - 1) / denom);
|
||||
}
|
||||
|
||||
~SparseLargeVector() noexcept {
|
||||
FreeMemoryPages(base_ptr, alloc_size);
|
||||
}
|
||||
|
||||
SparseLargeVector(const SparseLargeVector&) = delete;
|
||||
SparseLargeVector& operator=(const SparseLargeVector&) = delete;
|
||||
SparseLargeVector(SparseLargeVector&& other) = delete;
|
||||
SparseLargeVector& operator=(SparseLargeVector&& other) = delete;
|
||||
|
||||
void ResizeAndClear(std::size_t count) noexcept {
|
||||
if (auto const new_size = count * sizeof(T); new_size != alloc_size) {
|
||||
FreeMemoryPages(base_ptr, alloc_size);
|
||||
alloc_size = new_size;
|
||||
base_ptr = static_cast<T*>(AllocateMemoryPages(alloc_size));
|
||||
|
||||
auto denom = HostPageSize * 64;
|
||||
committed_pages = std::vector<std::atomic<u64>>((alloc_size + denom - 1) / denom);
|
||||
}
|
||||
}
|
||||
|
||||
/// Returns a reference to the value of the requested index and allocates memory if needed.
|
||||
T& GetAndFault(std::size_t index) noexcept {
|
||||
if (index > alloc_size / sizeof(T)) {
|
||||
UNREACHABLE_MSG("Out of bounds RW access on SparseLargeVector @ {}", index);
|
||||
}
|
||||
|
||||
if (!IsCommittedPage(index)) {
|
||||
CommitPage(index);
|
||||
}
|
||||
return base_ptr[index];
|
||||
}
|
||||
|
||||
/// Returns a reference to the value of the requested index if initialized, or will otherwise return a zero-initialized object.
|
||||
const T& GetOrDefault(std::size_t index) const {
|
||||
#ifdef _WIN32
|
||||
if (!IsCommittedPage(index)) {
|
||||
return *reinterpret_cast<const T*>(&default_val);
|
||||
}
|
||||
#endif
|
||||
// On non-Windows, OS page table should optimize this by pointing to a zero page if unallocated.
|
||||
return base_ptr[index];
|
||||
}
|
||||
|
||||
void Set(std::size_t index, const T& value) noexcept {
|
||||
if (index > alloc_size / sizeof(T)) {
|
||||
LOG_CRITICAL(Common_Memory, "Out of bounds write on SparseLargeVector @ {}", index);
|
||||
return;
|
||||
}
|
||||
if (!IsCommittedPage(index))
|
||||
CommitPage(index);
|
||||
base_ptr[index] = value;
|
||||
}
|
||||
|
||||
void ZeroRegion(std::size_t start, std::size_t end_) noexcept {
|
||||
u64 base = reinterpret_cast<u64>(&base_ptr[start]);
|
||||
const u64 end = reinterpret_cast<u64>(&base_ptr[end_]);
|
||||
|
||||
const u64 end_page = AlignUp(base, HostPageSize);
|
||||
const u64 first_size = (std::min)(end_page, end) - base;
|
||||
|
||||
if (IsCommittedPage(start / sizeof(T))) {
|
||||
std::memset(reinterpret_cast<void*>(base), 0, first_size);
|
||||
}
|
||||
|
||||
if (end <= end_page)
|
||||
return;
|
||||
|
||||
base = end_page;
|
||||
|
||||
for (u64 page = base; page < end; page += HostPageSize) {
|
||||
if (!IsCommittedPage((page - reinterpret_cast<u64>(base_ptr)) / sizeof(T))) {
|
||||
continue;
|
||||
}
|
||||
|
||||
std::memset(reinterpret_cast<void*>(page), 0, (std::min)( HostPageSize, end - page));
|
||||
}
|
||||
}
|
||||
|
||||
constexpr void CommitRegion(size_t index, size_t end_) {
|
||||
const u64 base = static_cast<u64>(index) * sizeof(T);
|
||||
const u64 end = static_cast<u64>(end_) * sizeof(T);
|
||||
|
||||
for (u64 page = AlignDown(base, HostPageSize); page < end; page += HostPageSize) {
|
||||
if (!IsCommittedPage(page / sizeof(T))) {
|
||||
CommitPage(page / sizeof(T));
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
constexpr T& GetUnchecked(size_t index) {
|
||||
return base_ptr[index];
|
||||
}
|
||||
|
||||
[[nodiscard]] constexpr const T& operator[](std::size_t index) const noexcept {
|
||||
return GetOrDefault(index);
|
||||
}
|
||||
|
||||
[[nodiscard]] constexpr const T* data() const noexcept {
|
||||
return base_ptr;
|
||||
}
|
||||
|
||||
[[nodiscard]] constexpr std::size_t size() const noexcept {
|
||||
return alloc_size / sizeof(T);
|
||||
}
|
||||
|
||||
private:
|
||||
[[nodiscard]] constexpr bool IsCommittedPage(std::size_t index) const noexcept {
|
||||
if (index > alloc_size / sizeof(T)) {
|
||||
LOG_CRITICAL(Common_Memory, "Out of bounds access on large vector @ {}", index);
|
||||
return false;
|
||||
}
|
||||
|
||||
auto page = (index * sizeof(T)) >> HostPageBits;
|
||||
auto val = committed_pages[page >> 6].load(std::memory_order_acquire);
|
||||
return (val >> (page & 63)) & 1;
|
||||
}
|
||||
|
||||
constexpr void CommitPage(std::size_t index) noexcept {
|
||||
auto page_index = (index * sizeof(T)) >> HostPageBits;
|
||||
auto page = reinterpret_cast<uintptr_t>(base_ptr + index) & HostPageMask;
|
||||
#if defined(_WIN32)
|
||||
CommitVectorPage(page, true);
|
||||
#else
|
||||
mprotect(reinterpret_cast<void*>(page), HostPageSize, PROT_READ | PROT_WRITE);
|
||||
#endif
|
||||
|
||||
committed_pages[page_index >> 6].fetch_or(1ULL << (page_index & 63), std::memory_order_release);
|
||||
}
|
||||
|
||||
std::size_t alloc_size{};
|
||||
T* base_ptr{};
|
||||
|
||||
std::vector<std::atomic<u64>> committed_pages{};
|
||||
#ifdef _WIN32
|
||||
const std::array<u8, sizeof(T)> default_val{};
|
||||
#endif
|
||||
};
|
||||
|
||||
} // namespace Common
|
||||
@@ -1,44 +0,0 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#ifdef _WIN32
|
||||
#include <windows.h>
|
||||
#else
|
||||
#include <sys/mman.h>
|
||||
#endif
|
||||
|
||||
#include "common/assert.h"
|
||||
#include "common/virtual_buffer.h"
|
||||
|
||||
namespace Common {
|
||||
|
||||
void* AllocateMemoryPages(std::size_t size) noexcept {
|
||||
#ifdef _WIN32
|
||||
void* base = VirtualAlloc(nullptr, size, MEM_COMMIT | MEM_RESERVE, PAGE_READWRITE);
|
||||
if (base == nullptr) {
|
||||
// Probably failing to reserve is less likely than failing to commit
|
||||
base = VirtualAlloc(nullptr, size, MEM_COMMIT, PAGE_READWRITE);
|
||||
}
|
||||
#else
|
||||
void* base = mmap(nullptr, size, PROT_READ | PROT_WRITE, MAP_ANON | MAP_PRIVATE, -1, 0);
|
||||
if (base == MAP_FAILED)
|
||||
base = nullptr;
|
||||
#endif
|
||||
ASSERT(base);
|
||||
return base;
|
||||
}
|
||||
|
||||
void FreeMemoryPages(void* base, [[maybe_unused]] std::size_t size) noexcept {
|
||||
if (!base)
|
||||
return;
|
||||
#ifdef _WIN32
|
||||
ASSERT(VirtualFree(base, 0, MEM_RELEASE));
|
||||
#else
|
||||
ASSERT(munmap(base, size) == 0);
|
||||
#endif
|
||||
}
|
||||
|
||||
} // namespace Common
|
||||
@@ -1,84 +0,0 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2020 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <utility>
|
||||
|
||||
namespace Common {
|
||||
|
||||
void* AllocateMemoryPages(std::size_t size) noexcept;
|
||||
void FreeMemoryPages(void* base, std::size_t size) noexcept;
|
||||
|
||||
template <typename T>
|
||||
class VirtualBuffer final {
|
||||
public:
|
||||
// TODO: Uncomment this and change Common::PageTable::PageInfo to be trivially constructible
|
||||
// using std::atomic_ref once libc++ has support for it
|
||||
// static_assert(
|
||||
// std::is_trivially_constructible_v<T>,
|
||||
// "T must be trivially constructible, as non-trivial constructors will not be executed "
|
||||
// "with the current allocator");
|
||||
|
||||
constexpr VirtualBuffer() = default;
|
||||
explicit VirtualBuffer(std::size_t count) noexcept
|
||||
: alloc_size{count * sizeof(T)}
|
||||
{
|
||||
base_ptr = reinterpret_cast<T*>(AllocateMemoryPages(alloc_size));
|
||||
}
|
||||
|
||||
~VirtualBuffer() noexcept {
|
||||
FreeMemoryPages(base_ptr, alloc_size);
|
||||
}
|
||||
|
||||
VirtualBuffer(const VirtualBuffer&) = delete;
|
||||
VirtualBuffer& operator=(const VirtualBuffer&) = delete;
|
||||
|
||||
VirtualBuffer(VirtualBuffer&& other) noexcept
|
||||
: alloc_size{std::exchange(other.alloc_size, 0)}
|
||||
, base_ptr{std::exchange(other.base_ptr, nullptr)}
|
||||
{}
|
||||
|
||||
VirtualBuffer& operator=(VirtualBuffer&& other) noexcept {
|
||||
alloc_size = std::exchange(other.alloc_size, 0);
|
||||
base_ptr = std::exchange(other.base_ptr, nullptr);
|
||||
return *this;
|
||||
}
|
||||
|
||||
void resize(std::size_t count) noexcept {
|
||||
if (auto const new_size = count * sizeof(T); new_size != alloc_size) {
|
||||
FreeMemoryPages(base_ptr, alloc_size);
|
||||
alloc_size = new_size;
|
||||
base_ptr = reinterpret_cast<T*>(AllocateMemoryPages(alloc_size));
|
||||
}
|
||||
}
|
||||
|
||||
[[nodiscard]] constexpr const T& operator[](std::size_t index) const noexcept {
|
||||
return base_ptr[index];
|
||||
}
|
||||
|
||||
[[nodiscard]] constexpr T& operator[](std::size_t index) noexcept {
|
||||
return base_ptr[index];
|
||||
}
|
||||
|
||||
[[nodiscard]] constexpr T* data() noexcept {
|
||||
return base_ptr;
|
||||
}
|
||||
|
||||
[[nodiscard]] constexpr const T* data() const noexcept {
|
||||
return base_ptr;
|
||||
}
|
||||
|
||||
[[nodiscard]] constexpr std::size_t size() const noexcept {
|
||||
return alloc_size / sizeof(T);
|
||||
}
|
||||
|
||||
private:
|
||||
std::size_t alloc_size{};
|
||||
T* base_ptr{};
|
||||
};
|
||||
|
||||
} // namespace Common
|
||||
@@ -1238,7 +1238,8 @@ if (HAS_NCE)
|
||||
arm/nce/interpreter_visitor.cpp
|
||||
arm/nce/interpreter_visitor.h
|
||||
arm/nce/patcher.cpp
|
||||
arm/nce/patcher.h)
|
||||
arm/nce/patcher.h
|
||||
arm/nce/visitor_base.h)
|
||||
target_link_libraries(core PRIVATE merry::oaknut)
|
||||
endif()
|
||||
|
||||
|
||||
@@ -172,12 +172,12 @@ void ArmDynarmic32::MakeJit(Common::PageTable* page_table) {
|
||||
if (page_table) {
|
||||
constexpr size_t PageBits = 12;
|
||||
constexpr size_t NumPageTableEntries = 1 << (32 - PageBits);
|
||||
constexpr size_t PageLog2Stride = 5;
|
||||
static_assert(1 << PageLog2Stride == sizeof(Common::PageTable::PageEntryData));
|
||||
|
||||
config.page_table = reinterpret_cast<std::array<std::uint8_t*, NumPageTableEntries>*>(page_table->entries.data());
|
||||
config.page_table_pointer_mask_bits = Common::PageTable::ATTRIBUTE_BITS;
|
||||
config.page_table_log2_stride = PageLog2Stride;
|
||||
// Dynarmic will not write to the page table, const_cast is safe here
|
||||
config.page_table = reinterpret_cast<std::array<std::uint8_t*, NumPageTableEntries>*>(
|
||||
const_cast<Common::PageTable::PageEntryData*>(page_table->entries.data()));
|
||||
config.page_table_pointer_mask = Common::PageTable::ATTRIBUTE_MASK;
|
||||
config.page_table_marked_bit = 0;
|
||||
config.absolute_offset_page_table = true;
|
||||
config.detect_misaligned_access_via_page_table = 16 | 32 | 64 | 128;
|
||||
config.only_detect_misalignment_via_page_table_on_page_boundary = true;
|
||||
@@ -188,6 +188,13 @@ void ArmDynarmic32::MakeJit(Common::PageTable* page_table) {
|
||||
|
||||
config.fastmem_exclusive_access = config.fastmem_pointer != std::nullopt;
|
||||
config.recompile_on_exclusive_fastmem_failure = true;
|
||||
|
||||
if (reinterpret_cast<u64>(m_system.DeviceMemory().buffer.BackingBasePointer() +
|
||||
Kernel::Board::Nintendo::Nx::KSystemControl::Init::GetIntendedMemorySize()) < (1ULL << 39)) {
|
||||
// Systems like FreeBSD allocate memory really low by default, and since we pack our page table entries,
|
||||
// we have to manually sign extend when our actual pointer is negative.
|
||||
config.page_table_sign_extension = Common::PageTable::SIGN_BIT;
|
||||
}
|
||||
}
|
||||
|
||||
// Multi-process state
|
||||
@@ -420,6 +427,7 @@ void ArmDynarmic32::SignalInterrupt(Kernel::KThread* thread) {
|
||||
}
|
||||
|
||||
void ArmDynarmic32::ClearInstructionCache() {
|
||||
m_cb->last_code_addr = u64(-1);
|
||||
m_jit->ClearCache();
|
||||
}
|
||||
|
||||
|
||||
@@ -211,13 +211,12 @@ void ArmDynarmic64::MakeJit(Common::PageTable* page_table, std::size_t address_s
|
||||
|
||||
// Memory
|
||||
if (page_table) {
|
||||
constexpr size_t PageLog2Stride = 5;
|
||||
static_assert(1 << PageLog2Stride == sizeof(Common::PageTable::PageEntryData));
|
||||
|
||||
config.page_table = reinterpret_cast<void**>(page_table->entries.data());
|
||||
// Dynarmic will not write to the page table, const_cast is safe here
|
||||
config.page_table = reinterpret_cast<void**>(
|
||||
const_cast<Common::PageTable::PageEntryData*>(page_table->entries.data()));
|
||||
config.page_table_address_space_bits = std::uint32_t(address_space_bits);
|
||||
config.page_table_pointer_mask_bits = Common::PageTable::ATTRIBUTE_BITS;
|
||||
config.page_table_log2_stride = PageLog2Stride;
|
||||
config.page_table_pointer_mask = Common::PageTable::ATTRIBUTE_MASK;
|
||||
config.page_table_marked_bit = 0;
|
||||
config.silently_mirror_page_table = false;
|
||||
config.absolute_offset_page_table = true;
|
||||
config.detect_misaligned_access_via_page_table = 16 | 32 | 64 | 128;
|
||||
@@ -231,6 +230,13 @@ void ArmDynarmic64::MakeJit(Common::PageTable* page_table, std::size_t address_s
|
||||
|
||||
config.fastmem_exclusive_access = config.fastmem_pointer != std::nullopt;
|
||||
config.recompile_on_exclusive_fastmem_failure = true;
|
||||
|
||||
if (reinterpret_cast<u64>(m_system.DeviceMemory().buffer.BackingBasePointer() +
|
||||
Kernel::Board::Nintendo::Nx::KSystemControl::Init::GetIntendedMemorySize()) < (1ULL << 39)) {
|
||||
// Systems like FreeBSD allocate memory really low by default, and since we pack our page table entries,
|
||||
// we have to manually sign extend when our actual pointer is negative.
|
||||
config.page_table_sign_extension = Common::PageTable::SIGN_BIT;
|
||||
}
|
||||
}
|
||||
|
||||
// Multi-process state
|
||||
@@ -447,6 +453,7 @@ void ArmDynarmic64::SignalInterrupt(Kernel::KThread* thread) {
|
||||
}
|
||||
|
||||
void ArmDynarmic64::ClearInstructionCache() {
|
||||
m_cb->last_code_addr = u64(-1);
|
||||
m_jit->ClearCache();
|
||||
}
|
||||
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -7,77 +7,97 @@
|
||||
|
||||
#pragma once
|
||||
|
||||
#include <optional>
|
||||
#include <span>
|
||||
#include <atomic>
|
||||
#include <signal.h>
|
||||
#include <unistd.h>
|
||||
#include <span>
|
||||
|
||||
#pragma GCC diagnostic push
|
||||
#pragma GCC diagnostic ignored "-Wshadow"
|
||||
#include <dynarmic/frontend/A64/a64_types.h>
|
||||
#include <dynarmic/frontend/imm.h>
|
||||
#pragma GCC diagnostic pop
|
||||
|
||||
#include "core/hle/kernel/k_thread.h"
|
||||
#include "core/memory.h"
|
||||
#include "common/logging.h"
|
||||
#include "core/arm/nce/visitor_base.h"
|
||||
|
||||
namespace Core {
|
||||
|
||||
class InterpreterVisitor {
|
||||
namespace Memory {
|
||||
class Memory;
|
||||
}
|
||||
|
||||
class InterpreterVisitor final : public VisitorBase {
|
||||
public:
|
||||
explicit InterpreterVisitor(Core::Memory::Memory& memory, std::span<u64, 31> regs,
|
||||
std::span<u128, 32> fpsimd_regs, u64& sp, const u64& pc)
|
||||
: m_memory(memory), m_regs(regs), m_fpsimd_regs(fpsimd_regs), m_sp(sp), m_pc(pc) {}
|
||||
~InterpreterVisitor() override = default;
|
||||
|
||||
bool Execute(u32 inst);
|
||||
enum class MemOp {
|
||||
Load,
|
||||
Store,
|
||||
Prefetch,
|
||||
};
|
||||
|
||||
u128 GetVec(Vec v);
|
||||
u64 GetReg(Reg r);
|
||||
u64 GetSp();
|
||||
u64 GetPc();
|
||||
|
||||
void SetVec(Vec v, u128 value);
|
||||
void SetReg(Reg r, u64 value);
|
||||
void SetSp(u64 value);
|
||||
|
||||
u64 ExtendReg(size_t bitsize, Reg reg, Imm<3> option, u8 shift);
|
||||
|
||||
// Loads and stores - Load/Store Exclusive
|
||||
bool Ordered(size_t size, bool L, bool o0, Reg Rn, Reg Rt);
|
||||
bool STLLR(Imm<2> size, Reg Rn, Reg Rt) override;
|
||||
bool STLR(Imm<2> size, Reg Rn, Reg Rt) override;
|
||||
bool LDLAR(Imm<2> size, Reg Rn, Reg Rt) override;
|
||||
bool LDAR(Imm<2> size, Reg Rn, Reg Rt) override;
|
||||
|
||||
// Loads and stores - Load register (literal)
|
||||
bool LDR_lit_gen(bool opc_0, Imm<19> imm19, Reg Rt) override;
|
||||
bool LDR_lit_fpsimd(Imm<2> opc, Imm<19> imm19, Vec Vt) override;
|
||||
|
||||
// Loads and stores - Load/Store register pair
|
||||
bool STP_LDP_gen(Imm<2> opc, bool not_postindex, bool wback, Imm<1> L, Imm<7> imm7, Reg Rt2,
|
||||
Reg Rn, Reg Rt) override;
|
||||
bool STP_LDP_fpsimd(Imm<2> opc, bool not_postindex, bool wback, Imm<1> L, Imm<7> imm7, Vec Vt2,
|
||||
Reg Rn, Vec Vt) override;
|
||||
|
||||
// Loads and stores - Load/Store register (immediate)
|
||||
bool RegisterImmediate(bool wback, bool postindex, size_t scale, u64 offset, Imm<2> size,
|
||||
Imm<2> opc, Reg Rn, Reg Rt);
|
||||
bool STRx_LDRx_imm_1(Imm<2> size, Imm<2> opc, Imm<9> imm9, bool not_postindex, Reg Rn,
|
||||
Reg Rt) override;
|
||||
bool STRx_LDRx_imm_2(Imm<2> size, Imm<2> opc, Imm<12> imm12, Reg Rn, Reg Rt) override;
|
||||
bool STURx_LDURx(Imm<2> size, Imm<2> opc, Imm<9> imm9, Reg Rn, Reg Rt) override;
|
||||
|
||||
bool SIMDImmediate(bool wback, bool postindex, size_t scale, u64 offset, MemOp memop, Reg Rn,
|
||||
Vec Vt);
|
||||
bool STR_imm_fpsimd_1(Imm<2> size, Imm<1> opc_1, Imm<9> imm9, bool not_postindex, Reg Rn,
|
||||
Vec Vt) override;
|
||||
bool STR_imm_fpsimd_2(Imm<2> size, Imm<1> opc_1, Imm<12> imm12, Reg Rn, Vec Vt) override;
|
||||
bool LDR_imm_fpsimd_1(Imm<2> size, Imm<1> opc_1, Imm<9> imm9, bool not_postindex, Reg Rn,
|
||||
Vec Vt) override;
|
||||
bool LDR_imm_fpsimd_2(Imm<2> size, Imm<1> opc_1, Imm<12> imm12, Reg Rn, Vec Vt) override;
|
||||
bool STUR_fpsimd(Imm<2> size, Imm<1> opc_1, Imm<9> imm9, Reg Rn, Vec Vt) override;
|
||||
bool LDUR_fpsimd(Imm<2> size, Imm<1> opc_1, Imm<9> imm9, Reg Rn, Vec Vt) override;
|
||||
|
||||
// Loads and stores - Load/Store register (register offset)
|
||||
bool RegisterOffset(size_t scale, u8 shift, Imm<2> size, Imm<1> opc_1, Imm<1> opc_0, Reg Rm,
|
||||
Imm<3> option, Reg Rn, Reg Rt);
|
||||
bool STRx_reg(Imm<2> size, Imm<1> opc_1, Reg Rm, Imm<3> option, bool S, Reg Rn,
|
||||
Reg Rt) override;
|
||||
bool LDRx_reg(Imm<2> size, Imm<1> opc_1, Reg Rm, Imm<3> option, bool S, Reg Rn,
|
||||
Reg Rt) override;
|
||||
|
||||
bool SIMDOffset(size_t scale, u8 shift, Imm<1> opc_0, Reg Rm, Imm<3> option, Reg Rn, Vec Vt);
|
||||
bool STR_reg_fpsimd(Imm<2> size, Imm<1> opc_1, Reg Rm, Imm<3> option, bool S, Reg Rn,
|
||||
Vec Vt) override;
|
||||
bool LDR_reg_fpsimd(Imm<2> size, Imm<1> opc_1, Reg Rm, Imm<3> option, bool S, Reg Rn,
|
||||
Vec Vt) override;
|
||||
|
||||
private:
|
||||
template <size_t BitSize>
|
||||
using Imm = Dynarmic::Imm<BitSize>;
|
||||
using Reg = Dynarmic::A64::Reg;
|
||||
using Vec = Dynarmic::A64::Vec;
|
||||
|
||||
u128 GetVec(Vec v) const {
|
||||
return m_fpsimd_regs[static_cast<u32>(v)];
|
||||
}
|
||||
void SetVec(Vec v, u128 value) {
|
||||
m_fpsimd_regs[static_cast<u32>(v)] = value;
|
||||
}
|
||||
u64 GetReg(Reg r) const {
|
||||
return m_regs[static_cast<u32>(r)];
|
||||
}
|
||||
void SetReg(Reg r, u64 value) {
|
||||
m_regs[static_cast<u32>(r)] = value;
|
||||
}
|
||||
u64 GetRegSp(Reg r) const {
|
||||
if (r == Reg::SP) {
|
||||
return m_sp;
|
||||
}
|
||||
return m_regs[static_cast<u32>(r)];
|
||||
}
|
||||
void SetRegSp(Reg r, u64 value) {
|
||||
if (r == Reg::SP) {
|
||||
m_sp = value;
|
||||
return;
|
||||
}
|
||||
m_regs[static_cast<u32>(r)] = value;
|
||||
}
|
||||
|
||||
u64 ExtendReg(Reg reg, Imm<3> option, u8 shift);
|
||||
|
||||
bool Ordered(size_t size, bool load, Reg Rn, Reg Rt);
|
||||
bool LoadLiteral(bool wide, Imm<19> imm19, Reg Rt);
|
||||
bool LoadLiteralSimd(Imm<2> opc, Imm<19> imm19, Vec Vt);
|
||||
bool Pair(Imm<2> opc, bool not_postindex, bool wback, bool load, Imm<7> imm7, Reg Rt2, Reg Rn,
|
||||
Reg Rt);
|
||||
bool PairSimd(Imm<2> opc, bool not_postindex, bool wback, bool load, Imm<7> imm7, Vec Vt2,
|
||||
Reg Rn, Vec Vt);
|
||||
bool RegisterImmediate(bool wback, bool postindex, u64 offset, Imm<2> size, Imm<2> opc, Reg Rn,
|
||||
Reg Rt);
|
||||
bool RegisterOffset(bool S, Imm<2> size, Imm<1> opc_1, Imm<1> opc_0, Reg Rm, Imm<3> option,
|
||||
Reg Rn, Reg Rt);
|
||||
bool SimdImmediate(bool wback, bool postindex, size_t scale, u64 offset, bool load, Reg Rn,
|
||||
Vec Vt);
|
||||
bool SimdOffset(size_t scale, bool S, bool load, Reg Rm, Imm<3> option, Reg Rn, Vec Vt);
|
||||
|
||||
Core::Memory::Memory& m_memory;
|
||||
std::span<u64, 31> m_regs;
|
||||
std::span<u128, 32> m_fpsimd_regs;
|
||||
|
||||
@@ -371,9 +371,14 @@ size_t Patcher::GetPreSectionSize() const noexcept {
|
||||
}
|
||||
|
||||
void Patcher::WriteLoadContext(oaknut::VectorCodeGenerator& cg) {
|
||||
// This function was called, which modifies X30, so use that as a scratch register.
|
||||
// SP contains the guest X30, so save our return X30 to SP + 8, since we have allocated 16 bytes
|
||||
// of stack.
|
||||
cg.STR(X30, SP, 8);
|
||||
cg.LDR(X30, SP, 0);
|
||||
cg.MRS(X30, oaknut::SystemReg::TPIDR_EL0);
|
||||
cg.LDR(X30, X30, offsetof(NativeExecutionParameters, native_context));
|
||||
|
||||
// Load system registers.
|
||||
cg.LDR(W0, X30, offsetof(GuestContext, fpsr));
|
||||
cg.MSR(oaknut::SystemReg::FPSR, X0);
|
||||
cg.LDR(W0, X30, offsetof(GuestContext, fpcr));
|
||||
@@ -381,9 +386,7 @@ void Patcher::WriteLoadContext(oaknut::VectorCodeGenerator& cg) {
|
||||
cg.LDR(W0, X30, offsetof(GuestContext, nzcv));
|
||||
cg.MSR(oaknut::SystemReg::NZCV, X0);
|
||||
|
||||
cg.LDR(X0, X30, 8 * 30);
|
||||
cg.STR(X0, SP, 0);
|
||||
|
||||
// Load all vector registers.
|
||||
static constexpr size_t VEC_OFF = offsetof(GuestContext, vector_registers);
|
||||
for (int i = 0; i <= 30; i += 2) {
|
||||
cg.LDP(oaknut::QReg{i}, oaknut::QReg{i + 1}, X30, VEC_OFF + 16 * i);
|
||||
@@ -433,7 +436,7 @@ void Patcher::WriteSaveContext(oaknut::VectorCodeGenerator& cg) {
|
||||
cg.STR(W0, X30, offsetof(GuestContext, nzcv));
|
||||
cg.LDR(X0, SP, POST_INDEXED, 16);
|
||||
|
||||
cg.MOV(X1, X30);
|
||||
// Reload our return X30 from the stack, and return.
|
||||
cg.LDR(X30, SP, 8);
|
||||
cg.RET();
|
||||
}
|
||||
@@ -454,6 +457,8 @@ void Patcher::WriteSvcTrampoline(ModuleDestLabel module_dest, u32 svc_id, oaknut
|
||||
// Now that we've saved all registers, we can use any registers as scratch.
|
||||
// Store PC + 4 to arm interface, since we know the instruction offset from the entry point.
|
||||
oaknut::Label pc_after_svc;
|
||||
cg.MRS(X1, oaknut::SystemReg::TPIDR_EL0);
|
||||
cg.LDR(X1, X1, offsetof(NativeExecutionParameters, native_context));
|
||||
cg.LDR(X2, pc_after_svc);
|
||||
cg.STR(X2, X1, offsetof(GuestContext, pc));
|
||||
|
||||
@@ -509,13 +514,24 @@ void Patcher::WriteSvcTrampoline(ModuleDestLabel module_dest, u32 svc_id, oaknut
|
||||
|
||||
// Host called this location. Save the return address so we can
|
||||
// unwind the stack properly when jumping back.
|
||||
cg.ADD(X0, X1, offsetof(GuestContext, host_ctx));
|
||||
cg.MRS(X2, oaknut::SystemReg::TPIDR_EL0);
|
||||
cg.LDR(X2, X2, offsetof(NativeExecutionParameters, native_context));
|
||||
cg.ADD(X0, X2, offsetof(GuestContext, host_ctx));
|
||||
cg.STR(X30, X0, offsetof(HostContext, host_saved_regs) + 11 * sizeof(u64));
|
||||
|
||||
cg.STR(X1, SP, PRE_INDEXED, -16);
|
||||
// Reload all guest registers except X30 and PC.
|
||||
// The function also expects 16 bytes of stack already allocated.
|
||||
cg.STR(X30, SP, PRE_INDEXED, -16);
|
||||
cg.BL(load_ctx);
|
||||
cg.LDR(X30, SP, POST_INDEXED, 16);
|
||||
|
||||
// Use X1 as a scratch register to restore X30.
|
||||
cg.STR(X1, SP, PRE_INDEXED, -16);
|
||||
cg.MRS(X1, oaknut::SystemReg::TPIDR_EL0);
|
||||
cg.LDR(X1, X1, offsetof(NativeExecutionParameters, native_context));
|
||||
cg.LDR(X30, X1, offsetof(GuestContext, cpu_registers) + sizeof(u64) * 30);
|
||||
cg.LDR(X1, SP, POST_INDEXED, 16);
|
||||
|
||||
// Unlock the context.
|
||||
this->UnlockContext(cg);
|
||||
|
||||
|
||||
File diff suppressed because it is too large
Load Diff
@@ -29,8 +29,11 @@ public:
|
||||
|
||||
template <typename T>
|
||||
Common::PhysicalAddress GetPhysicalAddr(const T* ptr) const {
|
||||
return (reinterpret_cast<uintptr_t>(ptr) -
|
||||
reinterpret_cast<uintptr_t>(buffer.BackingBasePointer())) +
|
||||
return GetPhysicalAddr(reinterpret_cast<uintptr_t>(ptr));
|
||||
}
|
||||
|
||||
Common::PhysicalAddress GetPhysicalAddr(uintptr_t ptr) const {
|
||||
return (ptr - reinterpret_cast<uintptr_t>(buffer.BackingBasePointer())) +
|
||||
DramMemoryMap::Base;
|
||||
}
|
||||
|
||||
|
||||
@@ -18,11 +18,7 @@
|
||||
#include "common/common_types.h"
|
||||
#include "common/range_mutex.h"
|
||||
#include "common/scratch_buffer.h"
|
||||
#include "common/virtual_buffer.h"
|
||||
|
||||
namespace Common {
|
||||
class HostMemory;
|
||||
}
|
||||
#include "common/sparse_large_vector.h"
|
||||
|
||||
namespace Core {
|
||||
|
||||
@@ -130,10 +126,6 @@ public:
|
||||
// New batch API to update multiple ranges with a single lock acquisition.
|
||||
void UpdatePagesCachedBatch(std::span<const std::pair<DAddr, size_t>> ranges, s32 delta);
|
||||
|
||||
const Common::HostMemory& GetHostMemory() const noexcept {
|
||||
return host_memory;
|
||||
}
|
||||
|
||||
private:
|
||||
struct TranslationEntry {
|
||||
DAddr guest_page{};
|
||||
@@ -179,7 +171,6 @@ private:
|
||||
std::unique_ptr<DeviceMemoryManagerAllocator<Traits>> impl;
|
||||
|
||||
const uintptr_t physical_base;
|
||||
const Common::HostMemory& host_memory;
|
||||
DeviceInterface* device_inter;
|
||||
|
||||
struct TrackedEntry {
|
||||
@@ -187,8 +178,8 @@ private:
|
||||
u32 continuity_tracker;
|
||||
u32 compressed_physical_ptr;
|
||||
};
|
||||
Common::VirtualBuffer<u32> compressed_device_addr;
|
||||
Common::VirtualBuffer<TrackedEntry> tracked_entries;
|
||||
Common::SparseLargeVector<u32> compressed_device_addr;
|
||||
Common::SparseLargeVector<TrackedEntry> tracked_entries;
|
||||
|
||||
// Process memory interfaces
|
||||
|
||||
@@ -209,8 +200,8 @@ private:
|
||||
return std::make_pair(asid, address);
|
||||
}
|
||||
|
||||
void InsertCPUBacking(size_t page_index, VAddr address, Asid asid) {
|
||||
tracked_entries[page_index].cpu_backing_address = address | (asid.id << asid_start_bit);
|
||||
constexpr void InsertCPUBacking(size_t page_index, VAddr address, Asid asid) {
|
||||
tracked_entries.GetUnchecked(page_index).cpu_backing_address = address | (asid.id << asid_start_bit);
|
||||
}
|
||||
|
||||
std::array<TranslationEntry, 4> t_slot{};
|
||||
|
||||
@@ -171,24 +171,12 @@ struct DeviceMemoryManagerAllocator {
|
||||
template <typename Traits>
|
||||
DeviceMemoryManager<Traits>::DeviceMemoryManager(const DeviceMemory& device_memory_)
|
||||
: physical_base{uintptr_t(device_memory_.buffer.BackingBasePointer())}
|
||||
, host_memory{device_memory_.buffer}
|
||||
, device_inter{nullptr}
|
||||
, compressed_device_addr(1ULL << ((Settings::values.memory_layout_mode.GetValue() == Settings::MemoryLayout::Memory_4Gb ? physical_min_bits : physical_max_bits) - Memory::YUZU_PAGEBITS))
|
||||
, tracked_entries(device_as_size >> Memory::YUZU_PAGEBITS)
|
||||
{
|
||||
impl = std::make_unique<DeviceMemoryManagerAllocator<Traits>>();
|
||||
cached_pages = std::make_unique<CachedPages>();
|
||||
|
||||
const size_t total_virtual = device_as_size >> Memory::YUZU_PAGEBITS;
|
||||
for (size_t i = 0; i < total_virtual; i++) {
|
||||
tracked_entries[i].compressed_physical_ptr = 0;
|
||||
tracked_entries[i].continuity_tracker = 1;
|
||||
tracked_entries[i].cpu_backing_address = 0;
|
||||
}
|
||||
const size_t total_phys = 1ULL << ((Settings::values.memory_layout_mode.GetValue() == Settings::MemoryLayout::Memory_4Gb ? physical_min_bits : physical_max_bits) - Memory::YUZU_PAGEBITS);
|
||||
for (size_t i = 0; i < total_phys; i++) {
|
||||
compressed_device_addr[i] = 0;
|
||||
}
|
||||
}
|
||||
|
||||
template <typename Traits>
|
||||
@@ -221,26 +209,28 @@ void DeviceMemoryManager<Traits>::Map(DAddr address, VAddr virtual_address, size
|
||||
size_t start_page_d = address >> Memory::YUZU_PAGEBITS;
|
||||
size_t num_pages = Common::AlignUp(size, Memory::YUZU_PAGESIZE) >> Memory::YUZU_PAGEBITS;
|
||||
std::scoped_lock lk(mapping_guard);
|
||||
|
||||
tracked_entries.CommitRegion(start_page_d, start_page_d + num_pages);
|
||||
for (size_t i = 0; i < num_pages; i++) {
|
||||
const VAddr new_vaddress = virtual_address + i * Memory::YUZU_PAGESIZE;
|
||||
auto* ptr = process_memory->GetPointerSilent(Common::ProcessAddress(new_vaddress));
|
||||
if (ptr == nullptr) [[unlikely]] {
|
||||
tracked_entries[start_page_d + i].compressed_physical_ptr = 0;
|
||||
tracked_entries.GetUnchecked(start_page_d + i).compressed_physical_ptr = 0;
|
||||
continue;
|
||||
}
|
||||
auto phys_addr = static_cast<u32>(GetRawPhysicalAddr(ptr) >> Memory::YUZU_PAGEBITS) + 1U;
|
||||
tracked_entries[start_page_d + i].compressed_physical_ptr = phys_addr;
|
||||
tracked_entries.GetUnchecked(start_page_d + i).compressed_physical_ptr = phys_addr;
|
||||
InsertCPUBacking(start_page_d + i, new_vaddress, asid);
|
||||
const u32 base_dev = compressed_device_addr[phys_addr - 1U];
|
||||
const u32 new_dev = static_cast<u32>(start_page_d + i);
|
||||
if (base_dev == 0) [[likely]] {
|
||||
compressed_device_addr[phys_addr - 1U] = new_dev;
|
||||
compressed_device_addr.GetAndFault(phys_addr - 1U) = new_dev;
|
||||
continue;
|
||||
}
|
||||
u32 start_id = base_dev & MULTI_MASK;
|
||||
if ((base_dev >> MULTI_FLAG_BITS) == 0) {
|
||||
start_id = impl->multi_dev_address.Register(base_dev);
|
||||
compressed_device_addr[phys_addr - 1U] = MULTI_FLAG | start_id;
|
||||
compressed_device_addr.GetAndFault(phys_addr - 1U) = MULTI_FLAG | start_id;
|
||||
}
|
||||
impl->multi_dev_address.Register(new_dev, start_id);
|
||||
}
|
||||
@@ -256,24 +246,26 @@ void DeviceMemoryManager<Traits>::Unmap(DAddr address, size_t size) {
|
||||
size_t num_pages = Common::AlignUp(size, Memory::YUZU_PAGESIZE) >> Memory::YUZU_PAGEBITS;
|
||||
device_inter->InvalidateRegion(address, size);
|
||||
std::scoped_lock lk(mapping_guard);
|
||||
|
||||
tracked_entries.CommitRegion(start_page_d, start_page_d + num_pages); // should already be committed, but just in case
|
||||
for (size_t i = 0; i < num_pages; i++) {
|
||||
auto phys_addr = tracked_entries[start_page_d + i].compressed_physical_ptr;
|
||||
tracked_entries[start_page_d + i].compressed_physical_ptr = 0;
|
||||
tracked_entries[start_page_d + i].cpu_backing_address = 0;
|
||||
auto& entry = tracked_entries.GetUnchecked(start_page_d + i);
|
||||
auto phys_addr = entry.compressed_physical_ptr;
|
||||
entry.compressed_physical_ptr = 0;
|
||||
entry.cpu_backing_address = 0;
|
||||
if (phys_addr != 0) [[likely]] {
|
||||
const u32 base_dev = compressed_device_addr[phys_addr - 1U];
|
||||
u32& base_dev = compressed_device_addr.GetAndFault(phys_addr - 1U);
|
||||
if ((base_dev >> MULTI_FLAG_BITS) == 0) [[likely]] {
|
||||
compressed_device_addr[phys_addr - 1] = 0;
|
||||
base_dev = 0;
|
||||
continue;
|
||||
}
|
||||
const auto [more_entries, new_start] = impl->multi_dev_address.Unregister(
|
||||
static_cast<u32>(start_page_d + i), base_dev & MULTI_MASK);
|
||||
if (!more_entries) {
|
||||
compressed_device_addr[phys_addr - 1] =
|
||||
impl->multi_dev_address.ReleaseEntry(new_start);
|
||||
base_dev = impl->multi_dev_address.ReleaseEntry(new_start);
|
||||
continue;
|
||||
}
|
||||
compressed_device_addr[phys_addr - 1] = new_start | MULTI_FLAG;
|
||||
base_dev = new_start | MULTI_FLAG;
|
||||
}
|
||||
}
|
||||
t_slot = {};
|
||||
@@ -286,6 +278,8 @@ void DeviceMemoryManager<Traits>::TrackContinuityImpl(DAddr address, VAddr virtu
|
||||
size_t num_pages = Common::AlignUp(size, Memory::YUZU_PAGESIZE) >> Memory::YUZU_PAGEBITS;
|
||||
uintptr_t last_ptr = 0;
|
||||
size_t page_count = 1;
|
||||
|
||||
tracked_entries.CommitRegion(start_page_d, start_page_d + num_pages);
|
||||
for (size_t i = num_pages; i > 0; i--) {
|
||||
size_t index = i - 1;
|
||||
const VAddr new_vaddress = virtual_address + index * Memory::YUZU_PAGESIZE;
|
||||
@@ -297,14 +291,14 @@ void DeviceMemoryManager<Traits>::TrackContinuityImpl(DAddr address, VAddr virtu
|
||||
page_count = 1;
|
||||
}
|
||||
last_ptr = new_ptr;
|
||||
tracked_entries[start_page_d + index].continuity_tracker = static_cast<u32>(page_count);
|
||||
tracked_entries.GetUnchecked(start_page_d + index).continuity_tracker = static_cast<u32>(page_count) - 1;
|
||||
}
|
||||
}
|
||||
template <typename Traits>
|
||||
u8* DeviceMemoryManager<Traits>::GetSpan(const DAddr src_addr, const std::size_t size) {
|
||||
size_t page_index = src_addr >> page_bits;
|
||||
size_t subbits = src_addr & page_mask;
|
||||
if ((static_cast<size_t>(tracked_entries[page_index].continuity_tracker) << page_bits) >= size + subbits) {
|
||||
if ((static_cast<size_t>(tracked_entries[page_index].continuity_tracker+1) << page_bits) >= size + subbits) {
|
||||
return GetPointer<u8>(src_addr);
|
||||
}
|
||||
return nullptr;
|
||||
@@ -314,7 +308,7 @@ template <typename Traits>
|
||||
const u8* DeviceMemoryManager<Traits>::GetSpan(const DAddr src_addr, const std::size_t size) const {
|
||||
size_t page_index = src_addr >> page_bits;
|
||||
size_t subbits = src_addr & page_mask;
|
||||
if ((static_cast<size_t>(tracked_entries[page_index].continuity_tracker) << page_bits) >= size + subbits) {
|
||||
if ((static_cast<size_t>(tracked_entries[page_index].continuity_tracker+1) << page_bits) >= size + subbits) {
|
||||
return GetPointer<u8>(src_addr);
|
||||
}
|
||||
return nullptr;
|
||||
@@ -384,7 +378,7 @@ void DeviceMemoryManager<Traits>::WalkBlock(DAddr addr, std::size_t size, auto o
|
||||
std::size_t page_index = addr >> Memory::YUZU_PAGEBITS;
|
||||
std::size_t page_offset = addr & Memory::YUZU_PAGEMASK;
|
||||
while (remaining_size) {
|
||||
const size_t next_pages = std::size_t(tracked_entries[page_index].continuity_tracker);
|
||||
const size_t next_pages = std::size_t(tracked_entries[page_index].continuity_tracker+1);
|
||||
const std::size_t copy_amount = (std::min)((next_pages << Memory::YUZU_PAGEBITS) - page_offset, remaining_size);
|
||||
const auto current_vaddr = u64((page_index << Memory::YUZU_PAGEBITS) + page_offset);
|
||||
SCOPE_EXIT{
|
||||
|
||||
@@ -147,10 +147,6 @@ ProgramAddressSpaceType ProgramMetadata::GetAddressSpaceType() const {
|
||||
return npdm_header.address_space_type;
|
||||
}
|
||||
|
||||
void ProgramMetadata::SetAddressSpaceType(ProgramAddressSpaceType address_space) {
|
||||
npdm_header.address_space_type.Assign(address_space);
|
||||
}
|
||||
|
||||
u8 ProgramMetadata::GetMainThreadPriority() const {
|
||||
return npdm_header.main_thread_priority;
|
||||
}
|
||||
|
||||
@@ -1,6 +1,3 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2018 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
@@ -76,7 +73,6 @@ public:
|
||||
|
||||
bool Is64BitProgram() const;
|
||||
ProgramAddressSpaceType GetAddressSpaceType() const;
|
||||
void SetAddressSpaceType(ProgramAddressSpaceType address_space);
|
||||
u8 GetMainThreadPriority() const;
|
||||
u8 GetMainThreadCore() const;
|
||||
u32 GetMainThreadStackSize() const;
|
||||
|
||||
@@ -12,7 +12,6 @@
|
||||
#include "hid_core/frontend/emulated_controller.h"
|
||||
#include "hid_core/hid_core.h"
|
||||
#include "hid_core/hid_types.h"
|
||||
#include <array>
|
||||
|
||||
namespace Core::Frontend {
|
||||
|
||||
@@ -30,39 +29,23 @@ void DefaultControllerApplet::ReconfigureControllers(ReconfigureCallback callbac
|
||||
|
||||
const std::size_t min_supported_players =
|
||||
parameters.enable_single_mode ? 1 : parameters.min_players;
|
||||
using Core::HID::NpadStyleIndex;
|
||||
const std::size_t max_supported_players = parameters.enable_single_mode ? 1 : parameters.max_players;
|
||||
std::size_t num_selected_players = 0;
|
||||
std::array<bool, HID::HIDCore::available_controllers> keep_connected{};
|
||||
|
||||
// reserve existing AND valid players before filling slots. include Handheld, but not other
|
||||
for (std::size_t index = 0; index < hid_core.available_controllers - 1; ++index) {
|
||||
const auto* controller = hid_core.GetEmulatedControllerByIndex(index);
|
||||
if (!parameters.keep_controllers_connected || !controller->IsConnected() || num_selected_players >= max_supported_players) continue;
|
||||
|
||||
const auto style = controller->GetNpadStyleIndex();
|
||||
keep_connected[index] =
|
||||
(style == NpadStyleIndex::Fullkey && parameters.allow_pro_controller) ||
|
||||
(style == NpadStyleIndex::JoyconDual && parameters.allow_dual_joycons) ||
|
||||
(style == NpadStyleIndex::JoyconLeft && parameters.allow_left_joycon) ||
|
||||
(style == NpadStyleIndex::JoyconRight && parameters.allow_right_joycon) ||
|
||||
(style == NpadStyleIndex::Handheld && parameters.enable_single_mode && parameters.allow_handheld && !Settings::IsDockedMode()) ||
|
||||
(style == NpadStyleIndex::GameCube && parameters.allow_gamecube_controller);
|
||||
num_selected_players += keep_connected[index];
|
||||
}
|
||||
// Disconnect Handheld first.
|
||||
auto* handheld = hid_core.GetEmulatedController(Core::HID::NpadIdType::Handheld);
|
||||
if (!keep_connected[hid_core.available_controllers - 2]) handheld->Disconnect();
|
||||
handheld->Disconnect();
|
||||
|
||||
// Deduce the best configuration based on the input parameters.
|
||||
for (std::size_t index = 0; index < hid_core.available_controllers - 2; ++index) {
|
||||
auto* controller = hid_core.GetEmulatedControllerByIndex(index);
|
||||
|
||||
if (keep_connected[index]) continue;
|
||||
// First, disconnect all controllers regardless of the value of keep_controllers_connected.
|
||||
// This makes it easy to connect the desired controllers.
|
||||
controller->Disconnect();
|
||||
|
||||
// only add players still needed to reach the minimum
|
||||
if (num_selected_players >= min_supported_players) continue;
|
||||
++num_selected_players;
|
||||
// Only connect the minimum number of required players.
|
||||
if (index >= min_supported_players) {
|
||||
continue;
|
||||
}
|
||||
|
||||
// Connect controllers based on the following priority list from highest to lowest priority:
|
||||
// Pro Controller -> Dual Joycons -> Left Joycon/Right Joycon -> Handheld
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: 2014 Citra Emulator Project
|
||||
@@ -74,7 +74,7 @@ public:
|
||||
};
|
||||
|
||||
/// Called from GPU thread when a frame is displayed.
|
||||
virtual void OnFrameDisplayed([[maybe_unused]] u32 presented_frames) {}
|
||||
virtual void OnFrameDisplayed() {}
|
||||
|
||||
/**
|
||||
* Returns a GraphicsContext that the frontend provides to be used for rendering.
|
||||
|
||||
@@ -635,6 +635,36 @@ Result KPageTableBase::CheckMemoryState(const KMemoryInfo& info, KMemoryState st
|
||||
R_SUCCEED();
|
||||
}
|
||||
|
||||
bool KPageTableBase::BeginTraversal(const Common::PageTable &impl, TraversalEntry *out_entry, TraversalContext *out_context,
|
||||
Common::ProcessAddress address) const {
|
||||
out_context->next_offset = GetInteger(address);
|
||||
out_context->next_page = GetInteger(address) >> PageBits;
|
||||
|
||||
return ContinueTraversal(impl, out_entry, out_context);
|
||||
}
|
||||
|
||||
bool KPageTableBase::ContinueTraversal(const Common::PageTable &impl, TraversalEntry *out_entry,
|
||||
TraversalContext *context) const {
|
||||
// Setup invalid defaults.
|
||||
out_entry->phys_addr = 0;
|
||||
out_entry->block_size = PageSize;
|
||||
// Validate that we can read the actual entry.
|
||||
if (auto const page = context->next_page; page < impl.entries.size()) {
|
||||
// Validate that the entry is mapped.
|
||||
if (auto const paddr = impl.entries[page].Pointer(true); paddr != 0) {
|
||||
// Populate the results and return true
|
||||
out_entry->phys_addr = GetInteger(m_system.DeviceMemory().GetPhysicalAddr(paddr + context->next_offset));
|
||||
context->next_page += 1;
|
||||
context->next_offset += PageSize;
|
||||
return true;
|
||||
}
|
||||
}
|
||||
context->next_page += 1;
|
||||
context->next_offset += PageSize;
|
||||
// Otherwise return false
|
||||
return false;
|
||||
}
|
||||
|
||||
Result KPageTableBase::CheckMemoryStateContiguous(size_t* out_blocks_needed, KProcessAddress addr,
|
||||
size_t size, KMemoryState state_mask,
|
||||
KMemoryState state, KMemoryPermission perm_mask,
|
||||
@@ -940,7 +970,7 @@ Result KPageTableBase::QueryMappingImpl(KProcessAddress* out, KPhysicalAddress a
|
||||
size_t tot_size = 0;
|
||||
|
||||
next_valid =
|
||||
impl.BeginTraversal(std::addressof(next_entry), std::addressof(context), region_start);
|
||||
BeginTraversal(impl, std::addressof(next_entry), std::addressof(context), region_start);
|
||||
next_entry.block_size =
|
||||
(next_entry.block_size - (GetInteger(region_start) & (next_entry.block_size - 1)));
|
||||
|
||||
@@ -976,7 +1006,7 @@ Result KPageTableBase::QueryMappingImpl(KProcessAddress* out, KPhysicalAddress a
|
||||
break;
|
||||
}
|
||||
|
||||
next_valid = impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context));
|
||||
next_valid = ContinueTraversal(impl, std::addressof(next_entry), std::addressof(context));
|
||||
}
|
||||
|
||||
// Check the last entry.
|
||||
@@ -1754,7 +1784,7 @@ Result KPageTableBase::MakePageGroup(KPageGroup& pg, KProcessAddress addr, size_
|
||||
// Begin traversal.
|
||||
TraversalContext context;
|
||||
TraversalEntry next_entry;
|
||||
R_UNLESS(impl.BeginTraversal(std::addressof(next_entry), std::addressof(context), addr),
|
||||
R_UNLESS(BeginTraversal(impl, std::addressof(next_entry), std::addressof(context), addr),
|
||||
ResultInvalidCurrentMemory);
|
||||
|
||||
// Prepare tracking variables.
|
||||
@@ -1764,7 +1794,7 @@ Result KPageTableBase::MakePageGroup(KPageGroup& pg, KProcessAddress addr, size_
|
||||
|
||||
// Iterate, adding to group as we go.
|
||||
while (tot_size < size) {
|
||||
R_UNLESS(impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context)),
|
||||
R_UNLESS(ContinueTraversal(impl, std::addressof(next_entry), std::addressof(context)),
|
||||
ResultInvalidCurrentMemory);
|
||||
|
||||
if (next_entry.phys_addr != (cur_addr + cur_size)) {
|
||||
@@ -1828,7 +1858,7 @@ bool KPageTableBase::IsValidPageGroup(const KPageGroup& pg, KProcessAddress addr
|
||||
// Begin traversal.
|
||||
TraversalContext context;
|
||||
TraversalEntry next_entry;
|
||||
if (!impl.BeginTraversal(std::addressof(next_entry), std::addressof(context), addr)) {
|
||||
if (!BeginTraversal(impl, std::addressof(next_entry), std::addressof(context), addr)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -1839,7 +1869,7 @@ bool KPageTableBase::IsValidPageGroup(const KPageGroup& pg, KProcessAddress addr
|
||||
|
||||
// Iterate, comparing expected to actual.
|
||||
while (tot_size < size) {
|
||||
if (!impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context))) {
|
||||
if (!ContinueTraversal(impl, std::addressof(next_entry), std::addressof(context))) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -1896,7 +1926,7 @@ Result KPageTableBase::GetContiguousMemoryRangeWithState(
|
||||
// Begin a traversal.
|
||||
TraversalContext context;
|
||||
TraversalEntry cur_entry = {.phys_addr = 0, .block_size = 0};
|
||||
R_UNLESS(impl.BeginTraversal(std::addressof(cur_entry), std::addressof(context), address),
|
||||
R_UNLESS(BeginTraversal(impl, std::addressof(cur_entry), std::addressof(context), address),
|
||||
ResultInvalidCurrentMemory);
|
||||
|
||||
// Traverse until we have enough size or we aren't contiguous any more.
|
||||
@@ -1905,7 +1935,7 @@ Result KPageTableBase::GetContiguousMemoryRangeWithState(
|
||||
for (contig_size =
|
||||
cur_entry.block_size - (GetInteger(phys_address) & (cur_entry.block_size - 1));
|
||||
contig_size < size; contig_size += cur_entry.block_size) {
|
||||
if (!impl.ContinueTraversal(std::addressof(cur_entry), std::addressof(context))) {
|
||||
if (!ContinueTraversal(impl, std::addressof(cur_entry), std::addressof(context))) {
|
||||
break;
|
||||
}
|
||||
if (cur_entry.phys_addr != phys_address + contig_size) {
|
||||
@@ -2334,7 +2364,7 @@ Result KPageTableBase::QueryPhysicalAddress(Svc::lp64::PhysicalMemoryInfo* out,
|
||||
TraversalContext context;
|
||||
TraversalEntry next_entry;
|
||||
bool traverse_valid =
|
||||
m_impl.BeginTraversal(std::addressof(next_entry), std::addressof(context), virt_addr);
|
||||
BeginTraversal(m_impl, std::addressof(next_entry), std::addressof(context), virt_addr);
|
||||
R_UNLESS(traverse_valid, ResultInvalidCurrentMemory);
|
||||
|
||||
// Set tracking variables.
|
||||
@@ -2345,7 +2375,7 @@ Result KPageTableBase::QueryPhysicalAddress(Svc::lp64::PhysicalMemoryInfo* out,
|
||||
while (true) {
|
||||
// Continue the traversal.
|
||||
traverse_valid =
|
||||
m_impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context));
|
||||
ContinueTraversal(m_impl, std::addressof(next_entry), std::addressof(context));
|
||||
if (!traverse_valid) {
|
||||
break;
|
||||
}
|
||||
@@ -2567,7 +2597,7 @@ Result KPageTableBase::UnmapIoRegion(KProcessAddress dst_address, KPhysicalAddre
|
||||
TraversalContext context;
|
||||
TraversalEntry next_entry;
|
||||
ASSERT(
|
||||
impl.BeginTraversal(std::addressof(next_entry), std::addressof(context), dst_address));
|
||||
BeginTraversal(impl, std::addressof(next_entry), std::addressof(context), dst_address));
|
||||
|
||||
// Check that the physical region matches.
|
||||
R_UNLESS(next_entry.phys_addr == phys_addr, ResultInvalidMemoryRegion);
|
||||
@@ -2577,7 +2607,7 @@ Result KPageTableBase::UnmapIoRegion(KProcessAddress dst_address, KPhysicalAddre
|
||||
next_entry.block_size - (GetInteger(phys_addr) & (next_entry.block_size - 1));
|
||||
checked_size < size; checked_size += next_entry.block_size) {
|
||||
// Continue the traversal.
|
||||
ASSERT(impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context)));
|
||||
ASSERT(ContinueTraversal(impl, std::addressof(next_entry), std::addressof(context)));
|
||||
|
||||
// Check that the physical region matches.
|
||||
R_UNLESS(next_entry.phys_addr == phys_addr + checked_size, ResultInvalidMemoryRegion);
|
||||
@@ -3029,7 +3059,7 @@ Result KPageTableBase::InvalidateProcessDataCache(KProcessAddress address, size_
|
||||
TraversalContext context;
|
||||
TraversalEntry next_entry;
|
||||
bool traverse_valid =
|
||||
impl.BeginTraversal(std::addressof(next_entry), std::addressof(context), address);
|
||||
BeginTraversal(impl, std::addressof(next_entry), std::addressof(context), address);
|
||||
R_UNLESS(traverse_valid, ResultInvalidCurrentMemory);
|
||||
|
||||
// Prepare tracking variables.
|
||||
@@ -3041,7 +3071,7 @@ Result KPageTableBase::InvalidateProcessDataCache(KProcessAddress address, size_
|
||||
while (tot_size < size) {
|
||||
// Continue the traversal.
|
||||
traverse_valid =
|
||||
impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context));
|
||||
ContinueTraversal(impl, std::addressof(next_entry), std::addressof(context));
|
||||
R_UNLESS(traverse_valid, ResultInvalidCurrentMemory);
|
||||
|
||||
if (next_entry.phys_addr != (cur_addr + cur_size)) {
|
||||
@@ -3129,7 +3159,7 @@ Result KPageTableBase::ReadDebugMemory(KProcessAddress dst_address, KProcessAddr
|
||||
TraversalContext context;
|
||||
TraversalEntry next_entry;
|
||||
bool traverse_valid =
|
||||
impl.BeginTraversal(std::addressof(next_entry), std::addressof(context), src_address);
|
||||
BeginTraversal(impl, std::addressof(next_entry), std::addressof(context), src_address);
|
||||
R_UNLESS(traverse_valid, ResultInvalidCurrentMemory);
|
||||
|
||||
// Prepare tracking variables.
|
||||
@@ -3167,7 +3197,7 @@ Result KPageTableBase::ReadDebugMemory(KProcessAddress dst_address, KProcessAddr
|
||||
while (tot_size < size) {
|
||||
// Continue the traversal.
|
||||
traverse_valid =
|
||||
impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context));
|
||||
ContinueTraversal(impl, std::addressof(next_entry), std::addressof(context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
if (next_entry.phys_addr != (cur_addr + cur_size)) {
|
||||
@@ -3225,7 +3255,7 @@ Result KPageTableBase::WriteDebugMemory(KProcessAddress dst_address, KProcessAdd
|
||||
TraversalContext context;
|
||||
TraversalEntry next_entry;
|
||||
bool traverse_valid =
|
||||
impl.BeginTraversal(std::addressof(next_entry), std::addressof(context), dst_address);
|
||||
BeginTraversal(impl, std::addressof(next_entry), std::addressof(context), dst_address);
|
||||
R_UNLESS(traverse_valid, ResultInvalidCurrentMemory);
|
||||
|
||||
// Prepare tracking variables.
|
||||
@@ -3267,7 +3297,7 @@ Result KPageTableBase::WriteDebugMemory(KProcessAddress dst_address, KProcessAdd
|
||||
while (tot_size < size) {
|
||||
// Continue the traversal.
|
||||
traverse_valid =
|
||||
impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context));
|
||||
ContinueTraversal(impl, std::addressof(next_entry), std::addressof(context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
if (next_entry.phys_addr != (cur_addr + cur_size)) {
|
||||
@@ -3728,7 +3758,7 @@ Result KPageTableBase::CopyMemoryFromLinearToUser(
|
||||
TraversalContext context;
|
||||
TraversalEntry next_entry;
|
||||
bool traverse_valid =
|
||||
impl.BeginTraversal(std::addressof(next_entry), std::addressof(context), src_addr);
|
||||
BeginTraversal(impl, std::addressof(next_entry), std::addressof(context), src_addr);
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
// Prepare tracking variables.
|
||||
@@ -3768,7 +3798,7 @@ Result KPageTableBase::CopyMemoryFromLinearToUser(
|
||||
while (tot_size < size) {
|
||||
// Continue the traversal.
|
||||
traverse_valid =
|
||||
impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context));
|
||||
ContinueTraversal(impl, std::addressof(next_entry), std::addressof(context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
if (next_entry.phys_addr != (cur_addr + cur_size)) {
|
||||
@@ -3822,7 +3852,7 @@ Result KPageTableBase::CopyMemoryFromLinearToKernel(
|
||||
TraversalContext context;
|
||||
TraversalEntry next_entry;
|
||||
bool traverse_valid =
|
||||
impl.BeginTraversal(std::addressof(next_entry), std::addressof(context), src_addr);
|
||||
BeginTraversal(impl, std::addressof(next_entry), std::addressof(context), src_addr);
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
// Prepare tracking variables.
|
||||
@@ -3845,7 +3875,7 @@ Result KPageTableBase::CopyMemoryFromLinearToKernel(
|
||||
while (tot_size < size) {
|
||||
// Continue the traversal.
|
||||
traverse_valid =
|
||||
impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context));
|
||||
ContinueTraversal(impl, std::addressof(next_entry), std::addressof(context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
if (next_entry.phys_addr != (cur_addr + cur_size)) {
|
||||
@@ -3902,7 +3932,7 @@ Result KPageTableBase::CopyMemoryFromUserToLinear(
|
||||
TraversalContext context;
|
||||
TraversalEntry next_entry;
|
||||
bool traverse_valid =
|
||||
impl.BeginTraversal(std::addressof(next_entry), std::addressof(context), dst_addr);
|
||||
BeginTraversal(impl, std::addressof(next_entry), std::addressof(context), dst_addr);
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
// Prepare tracking variables.
|
||||
@@ -3941,7 +3971,7 @@ Result KPageTableBase::CopyMemoryFromUserToLinear(
|
||||
while (tot_size < size) {
|
||||
// Continue the traversal.
|
||||
traverse_valid =
|
||||
impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context));
|
||||
ContinueTraversal(impl, std::addressof(next_entry), std::addressof(context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
if (next_entry.phys_addr != (cur_addr + cur_size)) {
|
||||
@@ -3997,7 +4027,7 @@ Result KPageTableBase::CopyMemoryFromKernelToLinear(KProcessAddress dst_addr, si
|
||||
TraversalContext context;
|
||||
TraversalEntry next_entry;
|
||||
bool traverse_valid =
|
||||
impl.BeginTraversal(std::addressof(next_entry), std::addressof(context), dst_addr);
|
||||
BeginTraversal(impl, std::addressof(next_entry), std::addressof(context), dst_addr);
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
// Prepare tracking variables.
|
||||
@@ -4020,7 +4050,7 @@ Result KPageTableBase::CopyMemoryFromKernelToLinear(KProcessAddress dst_addr, si
|
||||
while (tot_size < size) {
|
||||
// Continue the traversal.
|
||||
traverse_valid =
|
||||
impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context));
|
||||
ContinueTraversal(impl, std::addressof(next_entry), std::addressof(context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
if (next_entry.phys_addr != (cur_addr + cur_size)) {
|
||||
@@ -4089,10 +4119,10 @@ Result KPageTableBase::CopyMemoryFromHeapToHeap(
|
||||
bool traverse_valid;
|
||||
|
||||
// Begin traversal.
|
||||
traverse_valid = src_impl.BeginTraversal(std::addressof(src_next_entry),
|
||||
traverse_valid = BeginTraversal(src_impl, std::addressof(src_next_entry),
|
||||
std::addressof(src_context), src_addr);
|
||||
ASSERT(traverse_valid);
|
||||
traverse_valid = dst_impl.BeginTraversal(std::addressof(dst_next_entry),
|
||||
traverse_valid = BeginTraversal(dst_impl, std::addressof(dst_next_entry),
|
||||
std::addressof(dst_context), dst_addr);
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
@@ -4127,7 +4157,7 @@ Result KPageTableBase::CopyMemoryFromHeapToHeap(
|
||||
if (ofs + cur_copy_size != size) {
|
||||
if (cur_src_addr + cur_min_size == cur_src_block_addr + cur_src_size) {
|
||||
// Continue the src traversal.
|
||||
traverse_valid = src_impl.ContinueTraversal(std::addressof(src_next_entry),
|
||||
traverse_valid = ContinueTraversal(src_impl, std::addressof(src_next_entry),
|
||||
std::addressof(src_context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
@@ -4138,7 +4168,7 @@ Result KPageTableBase::CopyMemoryFromHeapToHeap(
|
||||
if (cur_dst_addr + cur_min_size ==
|
||||
dst_next_entry.phys_addr + dst_next_entry.block_size) {
|
||||
// Continue the dst traversal.
|
||||
traverse_valid = dst_impl.ContinueTraversal(std::addressof(dst_next_entry),
|
||||
traverse_valid = ContinueTraversal(dst_impl, std::addressof(dst_next_entry),
|
||||
std::addressof(dst_context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
@@ -4223,10 +4253,10 @@ Result KPageTableBase::CopyMemoryFromHeapToHeapWithoutCheckDestination(
|
||||
bool traverse_valid;
|
||||
|
||||
// Begin traversal.
|
||||
traverse_valid = src_impl.BeginTraversal(std::addressof(src_next_entry),
|
||||
traverse_valid = BeginTraversal(src_impl, std::addressof(src_next_entry),
|
||||
std::addressof(src_context), src_addr);
|
||||
ASSERT(traverse_valid);
|
||||
traverse_valid = dst_impl.BeginTraversal(std::addressof(dst_next_entry),
|
||||
traverse_valid = BeginTraversal(dst_impl, std::addressof(dst_next_entry),
|
||||
std::addressof(dst_context), dst_addr);
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
@@ -4261,7 +4291,7 @@ Result KPageTableBase::CopyMemoryFromHeapToHeapWithoutCheckDestination(
|
||||
if (ofs + cur_copy_size != size) {
|
||||
if (cur_src_addr + cur_min_size == cur_src_block_addr + cur_src_size) {
|
||||
// Continue the src traversal.
|
||||
traverse_valid = src_impl.ContinueTraversal(std::addressof(src_next_entry),
|
||||
traverse_valid = ContinueTraversal(src_impl, std::addressof(src_next_entry),
|
||||
std::addressof(src_context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
@@ -4272,7 +4302,7 @@ Result KPageTableBase::CopyMemoryFromHeapToHeapWithoutCheckDestination(
|
||||
if (cur_dst_addr + cur_min_size ==
|
||||
dst_next_entry.phys_addr + dst_next_entry.block_size) {
|
||||
// Continue the dst traversal.
|
||||
traverse_valid = dst_impl.ContinueTraversal(std::addressof(dst_next_entry),
|
||||
traverse_valid = ContinueTraversal(dst_impl, std::addressof(dst_next_entry),
|
||||
std::addressof(dst_context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
@@ -4547,7 +4577,7 @@ Result KPageTableBase::SetupForIpcServer(KProcessAddress* out_addr, size_t size,
|
||||
// Begin traversal.
|
||||
TraversalContext context;
|
||||
TraversalEntry next_entry;
|
||||
bool traverse_valid = src_impl.BeginTraversal(std::addressof(next_entry),
|
||||
bool traverse_valid = BeginTraversal(src_impl, std::addressof(next_entry),
|
||||
std::addressof(context), aligned_src_start);
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
@@ -4597,7 +4627,7 @@ Result KPageTableBase::SetupForIpcServer(KProcessAddress* out_addr, size_t size,
|
||||
// If the block's size was one page, we may need to continue traversal.
|
||||
if (cur_block_size == 0 && aligned_src_size > PageSize) {
|
||||
traverse_valid =
|
||||
src_impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context));
|
||||
ContinueTraversal(src_impl, std::addressof(next_entry), std::addressof(context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
cur_block_addr = next_entry.phys_addr;
|
||||
@@ -4610,7 +4640,7 @@ Result KPageTableBase::SetupForIpcServer(KProcessAddress* out_addr, size_t size,
|
||||
while (aligned_src_start + tot_block_size < mapping_src_end) {
|
||||
// Continue the traversal.
|
||||
traverse_valid =
|
||||
src_impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context));
|
||||
ContinueTraversal(src_impl, std::addressof(next_entry), std::addressof(context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
// Process the block.
|
||||
@@ -4653,7 +4683,7 @@ Result KPageTableBase::SetupForIpcServer(KProcessAddress* out_addr, size_t size,
|
||||
if (mapped_block_end + cur_block_size < aligned_src_end &&
|
||||
cur_block_size == last_block_size) {
|
||||
traverse_valid =
|
||||
src_impl.ContinueTraversal(std::addressof(next_entry), std::addressof(context));
|
||||
ContinueTraversal(src_impl, std::addressof(next_entry), std::addressof(context));
|
||||
ASSERT(traverse_valid);
|
||||
|
||||
cur_block_addr = next_entry.phys_addr;
|
||||
@@ -5601,7 +5631,7 @@ Result KPageTableBase::UnmapProcessMemory(KProcessAddress dst_address, size_t si
|
||||
ContiguousRangeInfo(KPageTableBase& pt, KProcessAddress address, size_t size)
|
||||
: m_pt(pt), m_remaining_size(size) {
|
||||
// Begin a traversal.
|
||||
ASSERT(m_pt.GetImpl().BeginTraversal(std::addressof(m_entry),
|
||||
ASSERT(m_pt.BeginTraversal(m_pt.GetImpl(), std::addressof(m_entry),
|
||||
std::addressof(m_context), address));
|
||||
|
||||
// Setup tracking fields.
|
||||
@@ -5632,7 +5662,7 @@ Result KPageTableBase::UnmapProcessMemory(KProcessAddress dst_address, size_t si
|
||||
void DetermineContiguousBlockExtents() {
|
||||
// Continue traversing until we're not contiguous, or we have enough.
|
||||
while (m_cur_size < m_remaining_size) {
|
||||
ASSERT(m_pt.GetImpl().ContinueTraversal(std::addressof(m_entry),
|
||||
ASSERT(m_pt.ContinueTraversal(m_pt.GetImpl(), std::addressof(m_entry),
|
||||
std::addressof(m_context)));
|
||||
|
||||
// If we're not contiguous, we're done.
|
||||
|
||||
@@ -370,6 +370,10 @@ private:
|
||||
size_t num_pages, size_t alignment, size_t offset,
|
||||
size_t guard_pages) const;
|
||||
|
||||
bool BeginTraversal(const Common::PageTable& impl, TraversalEntry* out_entry, TraversalContext* out_context,
|
||||
Common::ProcessAddress address) const;
|
||||
bool ContinueTraversal(const Common::PageTable& impl, TraversalEntry* out_entry, TraversalContext* context) const;
|
||||
|
||||
Result CheckMemoryStateContiguous(size_t* out_blocks_needed, KProcessAddress addr, size_t size,
|
||||
KMemoryState state_mask, KMemoryState state,
|
||||
KMemoryPermission perm_mask, KMemoryPermission perm,
|
||||
@@ -474,7 +478,14 @@ private:
|
||||
// Validate pre-conditions.
|
||||
ASSERT(this->IsLockedByCurrentThread());
|
||||
|
||||
return this->GetImpl().GetPhysicalAddress(out, virt_addr);
|
||||
if (virt_addr > (1ULL << m_address_space_width)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
*out = m_system.DeviceMemory().GetPhysicalAddr(
|
||||
this->GetImpl().entries[GetInteger(virt_addr) >> PageBits].Pointer(true) + GetInteger(virt_addr));
|
||||
|
||||
return true;
|
||||
}
|
||||
|
||||
public:
|
||||
|
||||
@@ -211,19 +211,14 @@ DAddr NvMap::PinHandle(NvMap::Handle::Id handle, bool low_area_pin) {
|
||||
while ((address = smmu.Allocate(aligned_up)) == 0) {
|
||||
// Free handles until the allocation succeeds
|
||||
std::scoped_lock queueLock(unmap_queue_lock);
|
||||
if (unmap_queue.empty()) {
|
||||
LOG_CRITICAL(Service_NVDRV, "Ran out of SMMU address space!");
|
||||
return 0;
|
||||
}
|
||||
// Handles in the unmap queue are guaranteed not to be pinned so don't bother
|
||||
// checking if they are before unmapping
|
||||
const std::shared_ptr<Handle> freeHandleDesc = unmap_queue.front();
|
||||
std::scoped_lock freeLock(freeHandleDesc->mutex);
|
||||
if (freeHandleDesc->d_address) {
|
||||
UnmapHandle(*freeHandleDesc);
|
||||
if (auto freeHandleDesc{unmap_queue.front()}) {
|
||||
// Handles in the unmap queue are guaranteed not to be pinned so don't bother
|
||||
// checking if they are before unmapping
|
||||
std::scoped_lock freeLock(freeHandleDesc->mutex);
|
||||
if (handle_description->d_address)
|
||||
UnmapHandle(*freeHandleDesc);
|
||||
} else {
|
||||
unmap_queue.pop_front();
|
||||
freeHandleDesc->unmap_queue_entry.reset();
|
||||
LOG_CRITICAL(Service_NVDRV, "Ran out of SMMU address space!");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -233,9 +233,9 @@ NvResult nvhost_as_gpu::FreeSpace(IoctlFreeSpace& params) {
|
||||
if (allocation.page_size != params.page_size || allocation.size != (u64(params.pages) * params.page_size))
|
||||
return NvResult::BadValue;
|
||||
|
||||
for (const auto mapping_offset : allocation.mappings) {
|
||||
void(FreeMappingLocked(mapping_offset));
|
||||
}
|
||||
for (const auto mapping_offset : allocation.mappings)
|
||||
if (!FreeMappingLocked(mapping_offset))
|
||||
return NvResult::BadValue;
|
||||
|
||||
// Unset sparse flag if required
|
||||
if (allocation.sparse)
|
||||
@@ -375,16 +375,39 @@ NvResult nvhost_as_gpu::MapBufferEx(IoctlMapBufferEx& params) {
|
||||
mapping_map.insert_or_assign(params.offset, Mapping(params.handle, device_address, params.offset, size, false, big_page, false));
|
||||
}
|
||||
|
||||
map_buffer_offsets.insert(params.offset);
|
||||
|
||||
return NvResult::Success;
|
||||
}
|
||||
|
||||
NvResult nvhost_as_gpu::UnmapBuffer(IoctlUnmapBuffer& params) {
|
||||
LOG_DEBUG(Service_NVDRV, "called, offset={:#x}", params.offset);
|
||||
std::scoped_lock lock(mutex);
|
||||
if (!vm.initialised) {
|
||||
return NvResult::BadValue;
|
||||
if (auto const offset_it = map_buffer_offsets.find(params.offset); offset_it != map_buffer_offsets.end()) {
|
||||
LOG_DEBUG(Service_NVDRV, "called, offset={:#x}", params.offset);
|
||||
if (!vm.initialised) {
|
||||
return NvResult::BadValue;
|
||||
}
|
||||
|
||||
auto const it = mapping_map.find(params.offset);
|
||||
auto const mapping = it->second;
|
||||
if (!mapping.fixed) {
|
||||
auto& allocator{mapping.big_page ? *vm.big_page_allocator : *vm.small_page_allocator};
|
||||
u32 page_size_bits{mapping.big_page ? vm.big_page_size_bits : VM::PAGE_SIZE_BITS};
|
||||
allocator.Free(u32(mapping.offset >> page_size_bits), u32(mapping.size >> page_size_bits));
|
||||
}
|
||||
|
||||
// Sparse mappings shouldn't be fully unmapped, just returned to their sparse state
|
||||
// Only FreeSpace can unmap them fully
|
||||
if (mapping.sparse_alloc) {
|
||||
gmmu->MapSparse(params.offset, mapping.size, mapping.big_page);
|
||||
} else {
|
||||
gmmu->Unmap(params.offset, mapping.size);
|
||||
}
|
||||
|
||||
nvmap.UnpinHandle(mapping.handle);
|
||||
mapping_map.erase(params.offset);
|
||||
map_buffer_offsets.erase(params.offset);
|
||||
}
|
||||
void(FreeMappingLocked(params.offset));
|
||||
return NvResult::Success;
|
||||
}
|
||||
|
||||
|
||||
@@ -113,6 +113,8 @@ private:
|
||||
};
|
||||
static_assert(sizeof(IoctlRemapEntry) == 20, "IoctlRemapEntry is incorrect size");
|
||||
|
||||
::Common::unordered_set<s64_le> map_buffer_offsets{};
|
||||
|
||||
struct IoctlMapBufferEx {
|
||||
MappingFlags flags{}; // bit0: fixed_offset, bit2: cacheable
|
||||
u32_le kind{}; // -1 is default
|
||||
|
||||
@@ -13,7 +13,6 @@
|
||||
#include "common/assert.h"
|
||||
#include "common/logging.h"
|
||||
#include "common/scope_exit.h"
|
||||
#include "common/settings.h"
|
||||
#include "core/core.h"
|
||||
#include "core/hle/kernel/k_event.h"
|
||||
#include "core/hle/service/nvdrv/core/container.h"
|
||||
@@ -148,12 +147,9 @@ NvResult nvhost_ctrl::IocCtrlEventWait(IocCtrlEventWaitParams& params, bool is_a
|
||||
|
||||
const auto check_failing = [&]() {
|
||||
if (events[slot].fails > 2) {
|
||||
std::unique_lock<std::mutex> stall;
|
||||
if (Settings::values.stall_on_gpu_fence_wait.GetValue()) {
|
||||
stall = system.StallApplication();
|
||||
}
|
||||
host1x_syncpoint_manager.WaitHost(fence_id, target_value);
|
||||
if (stall.owns_lock()) {
|
||||
{
|
||||
auto lk = system.StallApplication();
|
||||
host1x_syncpoint_manager.WaitHost(fence_id, target_value);
|
||||
system.UnstallApplication();
|
||||
}
|
||||
params.value.raw = target_value;
|
||||
|
||||
@@ -312,17 +312,19 @@ static boost::container::small_vector<Tegra::CommandHeader, 512> BuildWaitComman
|
||||
|
||||
static boost::container::small_vector<Tegra::CommandHeader, 512> BuildIncrementCommandList(
|
||||
NvFence fence) {
|
||||
const Tegra::CommandHeader increment =
|
||||
BuildFenceAction(Tegra::Engines::Puller::FenceOperation::Increment, fence.id);
|
||||
return {
|
||||
boost::container::small_vector<Tegra::CommandHeader, 512> result{
|
||||
Tegra::BuildCommandHeader(Tegra::BufferMethods::SyncpointPayload, 1,
|
||||
Tegra::SubmissionMode::Increasing),
|
||||
{},
|
||||
Tegra::BuildCommandHeader(Tegra::BufferMethods::SyncpointOperation, 2,
|
||||
Tegra::SubmissionMode::NonIncreasing),
|
||||
increment,
|
||||
increment,
|
||||
};
|
||||
{}};
|
||||
|
||||
for (u32 count = 0; count < 2; ++count) {
|
||||
result.push_back(Tegra::BuildCommandHeader(Tegra::BufferMethods::SyncpointOperation, 1,
|
||||
Tegra::SubmissionMode::Increasing));
|
||||
result.push_back(
|
||||
BuildFenceAction(Tegra::Engines::Puller::FenceOperation::Increment, fence.id));
|
||||
}
|
||||
|
||||
return result;
|
||||
}
|
||||
|
||||
static boost::container::small_vector<Tegra::CommandHeader, 512> BuildIncrementWithWfiCommandList(
|
||||
@@ -368,14 +370,17 @@ NvResult nvhost_gpu::SubmitGPFIFOImpl(IoctlSubmitGpfifo& params, Tegra::CommandL
|
||||
u32 increment{(flags.fence_increment.Value() != 0 ? 2 : 0) +
|
||||
(flags.increment_value.Value() != 0 ? params.fence.value : 0)};
|
||||
params.fence.value = syncpoint_manager.IncrementSyncpointMaxExt(channel_syncpoint, increment);
|
||||
gpu.PushGPUEntries(bind_id, std::move(entries));
|
||||
|
||||
if (flags.fence_increment.Value()) {
|
||||
if (flags.suppress_wfi.Value()) {
|
||||
entries.prefetch_command_list = BuildIncrementCommandList(params.fence);
|
||||
gpu.PushGPUEntries(bind_id,
|
||||
Tegra::CommandList{BuildIncrementCommandList(params.fence)});
|
||||
} else {
|
||||
entries.prefetch_command_list = BuildIncrementWithWfiCommandList(params.fence);
|
||||
gpu.PushGPUEntries(bind_id,
|
||||
Tegra::CommandList{BuildIncrementWithWfiCommandList(params.fence)});
|
||||
}
|
||||
}
|
||||
gpu.PushGPUEntries(bind_id, std::move(entries));
|
||||
|
||||
flags.raw = 0;
|
||||
|
||||
|
||||
@@ -165,22 +165,13 @@ Module::Module(Core::System& system)
|
||||
|
||||
Module::~Module() {}
|
||||
|
||||
std::shared_ptr<Devices::nvdevice> Module::FindDevice(DeviceFD fd) const {
|
||||
std::scoped_lock lock(open_files_mutex);
|
||||
const auto itr = open_files.find(fd);
|
||||
if (itr == open_files.end()) {
|
||||
return nullptr;
|
||||
}
|
||||
return itr->second;
|
||||
}
|
||||
|
||||
NvResult Module::VerifyFD(DeviceFD fd) const {
|
||||
if (fd < 0) {
|
||||
LOG_ERROR(Service_NVDRV, "Invalid DeviceFD={}!", fd);
|
||||
return NvResult::InvalidState;
|
||||
}
|
||||
|
||||
if (!FindDevice(fd)) {
|
||||
if (open_files.find(fd) == open_files.end()) {
|
||||
LOG_ERROR(Service_NVDRV, "Could not find DeviceFD={}!", fd);
|
||||
return NvResult::NotImplemented;
|
||||
}
|
||||
@@ -196,11 +187,8 @@ DeviceFD Module::Open(const std::string& device_name, NvCore::SessionId session_
|
||||
}
|
||||
|
||||
const DeviceFD fd = next_fd++;
|
||||
std::shared_ptr<Devices::nvdevice> device;
|
||||
{
|
||||
std::scoped_lock lock(open_files_mutex);
|
||||
device = it->second(fd)->second;
|
||||
}
|
||||
auto& builder = it->second;
|
||||
auto device = builder(fd)->second;
|
||||
|
||||
device->OnOpen(session_id, fd);
|
||||
|
||||
@@ -214,13 +202,14 @@ NvResult Module::Ioctl1(DeviceFD fd, Ioctl command, std::span<const u8> input,
|
||||
return NvResult::InvalidState;
|
||||
}
|
||||
|
||||
const auto device = FindDevice(fd);
|
||||
if (!device) {
|
||||
const auto itr = open_files.find(fd);
|
||||
|
||||
if (itr == open_files.end()) {
|
||||
LOG_ERROR(Service_NVDRV, "Could not find DeviceFD={}!", fd);
|
||||
return NvResult::NotImplemented;
|
||||
}
|
||||
|
||||
return device->Ioctl1(fd, command, input, output);
|
||||
return itr->second->Ioctl1(fd, command, input, output);
|
||||
}
|
||||
|
||||
NvResult Module::Ioctl2(DeviceFD fd, Ioctl command, std::span<const u8> input,
|
||||
@@ -230,13 +219,14 @@ NvResult Module::Ioctl2(DeviceFD fd, Ioctl command, std::span<const u8> input,
|
||||
return NvResult::InvalidState;
|
||||
}
|
||||
|
||||
const auto device = FindDevice(fd);
|
||||
if (!device) {
|
||||
const auto itr = open_files.find(fd);
|
||||
|
||||
if (itr == open_files.end()) {
|
||||
LOG_ERROR(Service_NVDRV, "Could not find DeviceFD={}!", fd);
|
||||
return NvResult::NotImplemented;
|
||||
}
|
||||
|
||||
return device->Ioctl2(fd, command, input, inline_input, output);
|
||||
return itr->second->Ioctl2(fd, command, input, inline_input, output);
|
||||
}
|
||||
|
||||
NvResult Module::Ioctl3(DeviceFD fd, Ioctl command, std::span<const u8> input, std::span<u8> output,
|
||||
@@ -246,13 +236,14 @@ NvResult Module::Ioctl3(DeviceFD fd, Ioctl command, std::span<const u8> input, s
|
||||
return NvResult::InvalidState;
|
||||
}
|
||||
|
||||
const auto device = FindDevice(fd);
|
||||
if (!device) {
|
||||
const auto itr = open_files.find(fd);
|
||||
|
||||
if (itr == open_files.end()) {
|
||||
LOG_ERROR(Service_NVDRV, "Could not find DeviceFD={}!", fd);
|
||||
return NvResult::NotImplemented;
|
||||
}
|
||||
|
||||
return device->Ioctl3(fd, command, input, output, inline_output);
|
||||
return itr->second->Ioctl3(fd, command, input, output, inline_output);
|
||||
}
|
||||
|
||||
NvResult Module::Close(DeviceFD fd) {
|
||||
@@ -261,17 +252,16 @@ NvResult Module::Close(DeviceFD fd) {
|
||||
return NvResult::InvalidState;
|
||||
}
|
||||
|
||||
const auto device = FindDevice(fd);
|
||||
if (!device) {
|
||||
const auto itr = open_files.find(fd);
|
||||
|
||||
if (itr == open_files.end()) {
|
||||
LOG_ERROR(Service_NVDRV, "Could not find DeviceFD={}!", fd);
|
||||
return NvResult::NotImplemented;
|
||||
}
|
||||
|
||||
{
|
||||
std::scoped_lock lock(open_files_mutex);
|
||||
open_files.erase(fd);
|
||||
}
|
||||
device->OnClose(fd);
|
||||
itr->second->OnClose(fd);
|
||||
|
||||
open_files.erase(itr);
|
||||
|
||||
return NvResult::Success;
|
||||
}
|
||||
@@ -282,13 +272,14 @@ NvResult Module::QueryEvent(DeviceFD fd, u32 event_id, Kernel::KEvent*& event) {
|
||||
return NvResult::InvalidState;
|
||||
}
|
||||
|
||||
const auto device = FindDevice(fd);
|
||||
if (!device) {
|
||||
const auto itr = open_files.find(fd);
|
||||
|
||||
if (itr == open_files.end()) {
|
||||
LOG_ERROR(Service_NVDRV, "Could not find DeviceFD={}!", fd);
|
||||
return NvResult::NotImplemented;
|
||||
}
|
||||
|
||||
event = device->QueryEvent(event_id);
|
||||
event = itr->second->QueryEvent(event_id);
|
||||
if (!event) {
|
||||
return NvResult::BadParameter;
|
||||
}
|
||||
|
||||
@@ -65,7 +65,10 @@ public:
|
||||
/// Returns a pointer to one of the available devices, identified by its name.
|
||||
template <typename T>
|
||||
std::shared_ptr<T> GetDevice(DeviceFD fd) {
|
||||
return std::static_pointer_cast<T>(FindDevice(fd));
|
||||
auto itr = open_files.find(fd);
|
||||
if (itr == open_files.end())
|
||||
return nullptr;
|
||||
return std::static_pointer_cast<T>(itr->second);
|
||||
}
|
||||
|
||||
NvResult VerifyFD(DeviceFD fd) const;
|
||||
@@ -94,8 +97,6 @@ public:
|
||||
private:
|
||||
friend class EventInterface;
|
||||
|
||||
std::shared_ptr<Devices::nvdevice> FindDevice(DeviceFD fd) const;
|
||||
|
||||
/// Manages syncpoints on the host
|
||||
NvCore::Container container;
|
||||
|
||||
@@ -105,7 +106,6 @@ private:
|
||||
using FilesContainerType = ::Common::unordered_map<DeviceFD, std::shared_ptr<Devices::nvdevice>>;
|
||||
/// Mapping of file descriptors to the devices they reference.
|
||||
FilesContainerType open_files;
|
||||
mutable std::mutex open_files_mutex;
|
||||
|
||||
KernelHelpers::ServiceContext service_context;
|
||||
|
||||
|
||||
@@ -182,12 +182,6 @@ AppLoader_DeconstructedRomDirectory::LoadResult AppLoader_DeconstructedRomDirect
|
||||
}
|
||||
metadata.Print();
|
||||
|
||||
if (Settings::values.relocate_36bit_address_space.GetValue() &&
|
||||
Settings::values.cpu_backend.GetValue() == Settings::CpuBackend::Nce &&
|
||||
metadata.GetAddressSpaceType() == FileSys::ProgramAddressSpaceType::Is36Bit) {
|
||||
metadata.SetAddressSpaceType(FileSys::ProgramAddressSpaceType::Is39Bit);
|
||||
}
|
||||
|
||||
// Enable NCE only for applications with 39-bit address space.
|
||||
const bool is_39bit =
|
||||
metadata.GetAddressSpaceType() == FileSys::ProgramAddressSpaceType::Is39Bit;
|
||||
|
||||
+42
-39
@@ -101,8 +101,10 @@ struct Memory::Impl {
|
||||
}
|
||||
|
||||
u64 protect_bytes = 0, protect_begin = 0;
|
||||
|
||||
current_page_table->entries.CommitRegion(vaddr >> YUZU_PAGEBITS, (vaddr + size) >> YUZU_PAGEBITS);
|
||||
for (u64 addr = vaddr; addr < vaddr + size; addr += YUZU_PAGESIZE) {
|
||||
const Common::PageType page_type = current_page_table->entries[addr >> YUZU_PAGEBITS].ptr.Type();
|
||||
const Common::PageType page_type = current_page_table->entries.GetUnchecked(addr >> YUZU_PAGEBITS).Type();
|
||||
switch (page_type) {
|
||||
case Common::PageType::RasterizerCachedMemory:
|
||||
if (protect_bytes > 0) {
|
||||
@@ -123,16 +125,14 @@ struct Memory::Impl {
|
||||
}
|
||||
|
||||
[[nodiscard]] u8* GetPointerFromRasterizerCachedMemory(u64 vaddr) const {
|
||||
Common::PhysicalAddress const paddr = current_page_table->entries[vaddr >> YUZU_PAGEBITS].addr;
|
||||
if (paddr)
|
||||
return system.DeviceMemory().GetPointer<u8>(paddr + vaddr);
|
||||
if (u64 paddr = current_page_table->entries[vaddr >> YUZU_PAGEBITS].Pointer(true); paddr)
|
||||
return reinterpret_cast<u8*>(paddr) + vaddr;
|
||||
return {};
|
||||
}
|
||||
|
||||
[[nodiscard]] u8* GetPointerFromDebugMemory(u64 vaddr) const {
|
||||
const Common::PhysicalAddress paddr = current_page_table->entries[vaddr >> YUZU_PAGEBITS].addr;
|
||||
if (paddr != 0)
|
||||
return system.DeviceMemory().GetPointer<u8>(paddr + vaddr);
|
||||
if (u64 paddr = current_page_table->entries[vaddr >> YUZU_PAGEBITS].Pointer(true); paddr)
|
||||
return reinterpret_cast<u8*>(paddr) + vaddr;
|
||||
return {};
|
||||
}
|
||||
|
||||
@@ -243,10 +243,12 @@ struct Memory::Impl {
|
||||
std::size_t page_index = addr >> YUZU_PAGEBITS;
|
||||
std::size_t page_offset = addr & YUZU_PAGEMASK;
|
||||
bool user_accessible = true;
|
||||
|
||||
current_page_table->entries.CommitRegion(page_index, page_index + (size >> YUZU_PAGEBITS) + 1);
|
||||
while (remaining_size != 0) {
|
||||
const std::size_t copy_amount = (std::min)(std::size_t(YUZU_PAGESIZE) - page_offset, remaining_size);
|
||||
const auto current_vaddr = u64((page_index << YUZU_PAGEBITS) + page_offset);
|
||||
const auto [pointer, type] = current_page_table->entries[page_index].ptr.PointerType();
|
||||
const auto [pointer, type, _] = current_page_table->entries.GetUnchecked(page_index).PointerTypeBlock();
|
||||
switch (type) {
|
||||
case Common::PageType::Unmapped: {
|
||||
user_accessible = false;
|
||||
@@ -297,10 +299,10 @@ struct Memory::Impl {
|
||||
}
|
||||
|
||||
[[nodiscard]] inline const u8* GetSpan(const VAddr addr, const std::size_t size) const noexcept {
|
||||
return (current_page_table->entries[addr >> YUZU_PAGEBITS].block == current_page_table->entries[(addr + size) >> YUZU_PAGEBITS].block) ? GetPointerSilent(addr) : nullptr;
|
||||
return (current_page_table->entries[addr >> YUZU_PAGEBITS].Block() == current_page_table->entries[(addr + size) >> YUZU_PAGEBITS].Block()) ? GetPointerSilent(addr) : nullptr;
|
||||
}
|
||||
[[nodiscard]] inline u8* GetSpan(const VAddr addr, const std::size_t size) noexcept {
|
||||
return (current_page_table->entries[addr >> YUZU_PAGEBITS].block == current_page_table->entries[(addr + size) >> YUZU_PAGEBITS].block) ? GetPointerSilent(addr) : nullptr;
|
||||
return (current_page_table->entries[addr >> YUZU_PAGEBITS].Block() == current_page_table->entries[(addr + size) >> YUZU_PAGEBITS].Block()) ? GetPointerSilent(addr) : nullptr;
|
||||
}
|
||||
|
||||
bool WriteBlockImpl(const Common::ProcessAddress addr, const void* buffer, const std::size_t size, bool unsafe) {
|
||||
@@ -404,11 +406,14 @@ struct Memory::Impl {
|
||||
// The region is at a granularity of CPU pages.
|
||||
|
||||
const u64 num_pages = ((vaddr + size - 1) >> YUZU_PAGEBITS) - (vaddr >> YUZU_PAGEBITS) + 1;
|
||||
|
||||
current_page_table->entries.CommitRegion(vaddr >> YUZU_PAGEBITS, (vaddr >> YUZU_PAGEBITS) + num_pages);
|
||||
for (u64 i = 0; i < num_pages; ++i, vaddr += YUZU_PAGESIZE) {
|
||||
const Common::PageType page_type = current_page_table->entries[vaddr >> YUZU_PAGEBITS].ptr.Type();
|
||||
auto& entry = current_page_table->entries.GetUnchecked(vaddr >> YUZU_PAGEBITS);
|
||||
const auto [pointer, type, block] = entry.PointerTypeBlock(true);
|
||||
if (debug) {
|
||||
// Switch page type to debug if now debug
|
||||
switch (page_type) {
|
||||
switch (type) {
|
||||
case Common::PageType::Unmapped:
|
||||
ASSERT(false && "Attempted to mark unmapped pages as debug");
|
||||
break;
|
||||
@@ -417,14 +422,14 @@ struct Memory::Impl {
|
||||
// Page is already marked.
|
||||
break;
|
||||
case Common::PageType::Memory:
|
||||
current_page_table->entries[vaddr >> YUZU_PAGEBITS].ptr.Store(0, Common::PageType::DebugMemory);
|
||||
entry.MarkDebug(pointer, block);
|
||||
break;
|
||||
default:
|
||||
UNREACHABLE();
|
||||
}
|
||||
} else {
|
||||
// Switch page type to non-debug if now non-debug
|
||||
switch (page_type) {
|
||||
switch (type) {
|
||||
case Common::PageType::Unmapped:
|
||||
ASSERT(false && "Attempted to mark unmapped pages as non-debug");
|
||||
break;
|
||||
@@ -433,8 +438,7 @@ struct Memory::Impl {
|
||||
// Don't mess with already non-debug or rasterizer memory.
|
||||
break;
|
||||
case Common::PageType::DebugMemory: {
|
||||
u8* const pointer = GetPointerFromDebugMemory(vaddr & ~YUZU_PAGEMASK);
|
||||
current_page_table->entries[vaddr >> YUZU_PAGEBITS].ptr.Store(uintptr_t(pointer) - (vaddr & ~YUZU_PAGEMASK), Common::PageType::Memory);
|
||||
entry.Store(false, Common::PageType::Memory, block, pointer);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
@@ -466,8 +470,10 @@ struct Memory::Impl {
|
||||
// is different). This assumes the specified GPU address region is contiguous as well.
|
||||
|
||||
const u64 num_pages = ((vaddr + size - 1) >> YUZU_PAGEBITS) - (vaddr >> YUZU_PAGEBITS) + 1;
|
||||
current_page_table->entries.CommitRegion(vaddr >> YUZU_PAGEBITS, (vaddr >> YUZU_PAGEBITS) + num_pages);
|
||||
for (u64 i = 0; i < num_pages; ++i, vaddr += YUZU_PAGESIZE) {
|
||||
const Common::PageType page_type= current_page_table->entries[vaddr >> YUZU_PAGEBITS].ptr.Type();
|
||||
auto& entry = current_page_table->entries.GetUnchecked(vaddr >> YUZU_PAGEBITS);
|
||||
const Common::PageType page_type = entry.Type();
|
||||
if (cached) {
|
||||
// Switch page type to cached if now cached
|
||||
switch (page_type) {
|
||||
@@ -477,7 +483,7 @@ struct Memory::Impl {
|
||||
break;
|
||||
case Common::PageType::DebugMemory:
|
||||
case Common::PageType::Memory:
|
||||
current_page_table->entries[vaddr >> YUZU_PAGEBITS].ptr.Store(0, Common::PageType::RasterizerCachedMemory);
|
||||
entry.MarkRasterizerCached();
|
||||
break;
|
||||
case Common::PageType::RasterizerCachedMemory:
|
||||
// There can be more than one GPU region mapped per CPU region, so it's common
|
||||
@@ -499,13 +505,13 @@ struct Memory::Impl {
|
||||
// that this area is already unmarked as cached.
|
||||
break;
|
||||
case Common::PageType::RasterizerCachedMemory: {
|
||||
if (u8* const pointer = GetPointerFromRasterizerCachedMemory(vaddr & ~YUZU_PAGEMASK); pointer == nullptr) {
|
||||
if (auto [ptr, _, block] = entry.PointerTypeBlock(true); ptr == 0) {
|
||||
// It's possible that this function has been called while updating the
|
||||
// pagetable after unmapping a VMA. In that case the underlying VMA will no
|
||||
// longer exist, and we should just leave the pagetable entry blank.
|
||||
current_page_table->entries[vaddr >> YUZU_PAGEBITS].ptr.Store(0, Common::PageType::Unmapped);
|
||||
entry.Store(false, Common::PageType::Unmapped, block, 0);
|
||||
} else {
|
||||
current_page_table->entries[vaddr >> YUZU_PAGEBITS].ptr.Store(uintptr_t(pointer) - (vaddr & ~YUZU_PAGEMASK), Common::PageType::Memory);
|
||||
entry.Store(false, Common::PageType::Memory, block, ptr);
|
||||
}
|
||||
break;
|
||||
}
|
||||
@@ -539,22 +545,18 @@ struct Memory::Impl {
|
||||
ASSERT_MSG(type != Common::PageType::Memory,
|
||||
"Mapping memory page without a pointer @ {:016x}", base * YUZU_PAGESIZE);
|
||||
|
||||
while (base != end) {
|
||||
page_table.entries[base].ptr.Store(0, type);
|
||||
page_table.entries[base].addr = 0;
|
||||
page_table.entries[base].block = 0;
|
||||
base += 1;
|
||||
}
|
||||
page_table.entries.ZeroRegion(base, end);
|
||||
} else {
|
||||
auto orig_base = base;
|
||||
while (base != end) {
|
||||
auto host_ptr = uintptr_t(system.DeviceMemory().GetPointer<u8>(target)) - (base << YUZU_PAGEBITS);
|
||||
auto backing = GetInteger(target) - (base << YUZU_PAGEBITS);
|
||||
page_table.entries[base].ptr.Store(host_ptr, type);
|
||||
page_table.entries[base].addr = backing;
|
||||
page_table.entries[base].block = orig_base << YUZU_PAGEBITS;
|
||||
auto current_block = block_count.fetch_add(1, std::memory_order_relaxed);
|
||||
ASSERT(current_block != 65535);
|
||||
|
||||
ASSERT_MSG(page_table.entries[base].ptr.Pointer(),
|
||||
page_table.entries.CommitRegion(base, end);
|
||||
while (base != end) {
|
||||
auto host_ptr = reinterpret_cast<u64>(system.DeviceMemory().GetPointer<u8>(target)) - (base << YUZU_PAGEBITS);;
|
||||
auto& entry = page_table.entries.GetUnchecked(base);
|
||||
|
||||
entry.Store(false, type, current_block, host_ptr);
|
||||
ASSERT_MSG(page_table.entries[base].Pointer(),
|
||||
"memory mapping base yield a nullptr within the table");
|
||||
|
||||
base += 1;
|
||||
@@ -569,11 +571,11 @@ struct Memory::Impl {
|
||||
vaddr &= 0xffffffffffffULL;
|
||||
if (AddressSpaceContains(*current_page_table, vaddr, 1)) [[likely]] {
|
||||
// Avoid adding any extra logic to this fast-path block
|
||||
const uintptr_t raw_pointer = current_page_table->entries[vaddr >> YUZU_PAGEBITS].ptr.Raw();
|
||||
if (const uintptr_t pointer = Common::PageTable::PageInfo::ExtractPointer(raw_pointer)) [[likely]] {
|
||||
const auto raw = current_page_table->entries[vaddr >> YUZU_PAGEBITS].Raw();
|
||||
if (auto pointer = Common::PageTable::PageEntryData::ExtractPointer(raw); pointer) [[likely]] {
|
||||
return reinterpret_cast<u8*>(pointer + vaddr);
|
||||
} else {
|
||||
switch (Common::PageTable::PageInfo::ExtractType(raw_pointer)) {
|
||||
switch (static_cast<Common::PageType>(raw.type)) {
|
||||
case Common::PageType::Memory:
|
||||
ASSERT_MSG(false, "Mapped memory page without a pointer @ {:#016x}", vaddr);
|
||||
return nullptr;
|
||||
@@ -773,6 +775,7 @@ struct Memory::Impl {
|
||||
#else
|
||||
Common::HostMemory* host_buffer{};
|
||||
#endif
|
||||
std::atomic<u16> block_count = 0;
|
||||
};
|
||||
|
||||
Memory::Memory(Core::System& system_) : system{system_} {
|
||||
@@ -811,7 +814,7 @@ bool Memory::IsValidVirtualAddress(const Common::ProcessAddress vaddr) const {
|
||||
if (page >= page_table.entries.size()) {
|
||||
return false;
|
||||
}
|
||||
const auto [pointer, type] = page_table.entries[page].ptr.PointerType();
|
||||
const auto [pointer, type, _] = page_table.entries[page].PointerTypeBlock();
|
||||
return pointer != 0 || type == Common::PageType::RasterizerCachedMemory ||
|
||||
type == Common::PageType::DebugMemory;
|
||||
}
|
||||
|
||||
@@ -371,8 +371,10 @@ EmitConfig A32AddressSpace::GetEmitConfig() {
|
||||
|
||||
.page_table_pointer = std::bit_cast<u64>(conf.page_table),
|
||||
.page_table_address_space_bits = 32,
|
||||
.page_table_pointer_mask_bits = conf.page_table_pointer_mask_bits,
|
||||
.page_table_pointer_mask = conf.page_table_pointer_mask,
|
||||
.page_table_log2_stride = conf.page_table_log2_stride,
|
||||
.page_table_marked_bit = conf.page_table_marked_bit,
|
||||
.page_table_sign_extension = conf.page_table_sign_extension,
|
||||
.silently_mirror_page_table = true,
|
||||
.absolute_offset_page_table = conf.absolute_offset_page_table,
|
||||
.detect_misaligned_access_via_page_table = conf.detect_misaligned_access_via_page_table,
|
||||
|
||||
@@ -545,8 +545,10 @@ EmitConfig A64AddressSpace::GetEmitConfig() {
|
||||
|
||||
.page_table_pointer = std::bit_cast<u64>(conf.page_table),
|
||||
.page_table_address_space_bits = conf.page_table_address_space_bits,
|
||||
.page_table_pointer_mask_bits = conf.page_table_pointer_mask_bits,
|
||||
.page_table_pointer_mask = conf.page_table_pointer_mask,
|
||||
.page_table_log2_stride = conf.page_table_log2_stride,
|
||||
.page_table_marked_bit = conf.page_table_marked_bit,
|
||||
.page_table_sign_extension = conf.page_table_sign_extension,
|
||||
.silently_mirror_page_table = conf.silently_mirror_page_table,
|
||||
.absolute_offset_page_table = conf.absolute_offset_page_table,
|
||||
.detect_misaligned_access_via_page_table = conf.detect_misaligned_access_via_page_table,
|
||||
|
||||
@@ -128,8 +128,10 @@ struct EmitConfig {
|
||||
// Page table
|
||||
u64 page_table_pointer;
|
||||
std::size_t page_table_address_space_bits;
|
||||
int page_table_pointer_mask_bits;
|
||||
u64 page_table_pointer_mask;
|
||||
std::size_t page_table_log2_stride;
|
||||
std::optional<std::uint8_t> page_table_marked_bit;
|
||||
std::optional<std::uint8_t> page_table_sign_extension;
|
||||
bool silently_mirror_page_table;
|
||||
bool absolute_offset_page_table;
|
||||
u8 detect_misaligned_access_via_page_table;
|
||||
|
||||
@@ -273,9 +273,18 @@ std::pair<oaknut::XReg, oaknut::XReg> InlinePageTableEmitVAddrLookup(oaknut::Cod
|
||||
// load x0 = *<(u8*)pagetable + index>
|
||||
code.LDR(Xscratch0, Xpagetable, Xscratch0);
|
||||
|
||||
if (ctx.conf.page_table_pointer_mask_bits != 0) {
|
||||
const u64 mask = u64(~u64(0)) << ctx.conf.page_table_pointer_mask_bits;
|
||||
code.AND(Xscratch0, Xscratch0, mask);
|
||||
if (ctx.conf.page_table_marked_bit) {
|
||||
code.TST(Xscratch0, 1ULL << *ctx.conf.page_table_marked_bit);
|
||||
code.B(NE, *fallback);
|
||||
}
|
||||
|
||||
if (ctx.conf.page_table_pointer_mask != 0) {
|
||||
code.AND(Xscratch0, Xscratch0, ctx.conf.page_table_pointer_mask);
|
||||
}
|
||||
|
||||
// TODO: combine this with page_table_pointer_mask
|
||||
if (ctx.conf.page_table_sign_extension) {
|
||||
code.SBFM(Xscratch0, Xscratch0, 0, *ctx.conf.page_table_sign_extension);
|
||||
}
|
||||
|
||||
code.CBZ(Xscratch0, *fallback);
|
||||
|
||||
@@ -9,6 +9,7 @@
|
||||
#pragma once
|
||||
|
||||
#include <bit>
|
||||
#include <utility>
|
||||
#include "dynarmic/backend/x64/xbyak.h"
|
||||
|
||||
#include "dynarmic/backend/x64/a32_emit_x64.h"
|
||||
@@ -78,27 +79,46 @@ Xbyak::RegExp EmitVAddrLookup(BlockOfCode& code, EmitContext& ctx, size_t bitsiz
|
||||
template<>
|
||||
[[maybe_unused]] Xbyak::RegExp EmitVAddrLookup<A32EmitContext>(BlockOfCode& code, A32EmitContext& ctx, size_t bitsize, Xbyak::Label& abort, Xbyak::Reg64 vaddr) {
|
||||
const Xbyak::Reg64 page = ctx.reg_alloc.ScratchGpr(code);
|
||||
const Xbyak::Reg32 tmp = ctx.conf.absolute_offset_page_table ? page.cvt32() : ctx.reg_alloc.ScratchGpr(code).cvt32();
|
||||
const Xbyak::Reg64 tmp = ctx.conf.absolute_offset_page_table && ctx.conf.page_table_pointer_mask == 0 ? page : ctx.reg_alloc.ScratchGpr(code);
|
||||
|
||||
EmitDetectMisalignedVAddr(code, ctx, bitsize, abort, vaddr, tmp.cvt64());
|
||||
EmitDetectMisalignedVAddr(code, ctx, bitsize, abort, vaddr, tmp);
|
||||
|
||||
// TODO: This code assumes vaddr has been zext from 32-bits to 64-bits.
|
||||
|
||||
code.mov(tmp, vaddr.cvt32());
|
||||
code.mov(tmp, vaddr);
|
||||
code.shr(tmp, int(page_table_const_bits));
|
||||
code.shl(tmp, int(ctx.conf.page_table_log2_stride));
|
||||
code.mov(page, qword[r14 + tmp.cvt64()]);
|
||||
if (ctx.conf.page_table_pointer_mask_bits == 0) {
|
||||
code.test(page, page);
|
||||
|
||||
if (ctx.conf.page_table_log2_stride > 3) {
|
||||
code.shl(tmp, int(ctx.conf.page_table_log2_stride));
|
||||
code.mov(page, qword[r14 + tmp.cvt64()]);
|
||||
} else {
|
||||
code.and_(page, ~u32(0) << ctx.conf.page_table_pointer_mask_bits);
|
||||
code.mov(page, qword[r14 + tmp.cvt64() * int(ctx.conf.page_table_log2_stride)]);
|
||||
}
|
||||
|
||||
// check for marked bit, use as unmapped if marked
|
||||
if (ctx.conf.page_table_marked_bit) {
|
||||
code.bt(page, *ctx.conf.page_table_marked_bit);
|
||||
code.jc(abort, code.T_NEAR);
|
||||
}
|
||||
// mask away attributes
|
||||
if (ctx.conf.page_table_pointer_mask == 0) {
|
||||
code.test(page, page);
|
||||
} else if (std::in_range<s32>(ctx.conf.page_table_pointer_mask)) {
|
||||
code.and_(page, ctx.conf.page_table_pointer_mask);
|
||||
} else {
|
||||
code.mov(tmp, ctx.conf.page_table_pointer_mask);
|
||||
code.and_(page, tmp);
|
||||
}
|
||||
// check for sign bit, apply sign extension as needed
|
||||
if (ctx.conf.page_table_sign_extension) {
|
||||
code.shl(page, 63 - int(*ctx.conf.page_table_sign_extension));
|
||||
code.sar(page, 63 - int(*ctx.conf.page_table_sign_extension));
|
||||
}
|
||||
|
||||
code.jz(abort, code.T_NEAR);
|
||||
if (ctx.conf.absolute_offset_page_table) {
|
||||
return page + vaddr;
|
||||
}
|
||||
code.mov(tmp, vaddr.cvt32());
|
||||
code.and_(tmp, static_cast<u32>(page_table_const_mask));
|
||||
code.mov(tmp, vaddr);
|
||||
code.and_(tmp, u32(page_table_const_mask));
|
||||
return page + tmp.cvt64();
|
||||
}
|
||||
|
||||
@@ -108,7 +128,7 @@ template<>
|
||||
const size_t unused_top_bits = 64 - ctx.conf.page_table_address_space_bits;
|
||||
|
||||
const Xbyak::Reg64 page = ctx.reg_alloc.ScratchGpr(code);
|
||||
const Xbyak::Reg64 tmp = ctx.conf.absolute_offset_page_table ? page : ctx.reg_alloc.ScratchGpr(code);
|
||||
const Xbyak::Reg64 tmp = ctx.conf.absolute_offset_page_table && ctx.conf.page_table_pointer_mask == 0 ? page : ctx.reg_alloc.ScratchGpr(code);
|
||||
|
||||
EmitDetectMisalignedVAddr(code, ctx, bitsize, abort, vaddr, tmp);
|
||||
|
||||
@@ -143,11 +163,26 @@ template<>
|
||||
|
||||
code.shl(tmp, int(ctx.conf.page_table_log2_stride));
|
||||
code.mov(page, qword[r14 + tmp]);
|
||||
if (ctx.conf.page_table_pointer_mask_bits == 0) {
|
||||
code.test(page, page);
|
||||
} else {
|
||||
code.and_(page, ~u32(0) << ctx.conf.page_table_pointer_mask_bits);
|
||||
|
||||
// check for marked bit, use as unmapped if marked
|
||||
if (ctx.conf.page_table_marked_bit) {
|
||||
code.bt(page, *ctx.conf.page_table_marked_bit);
|
||||
code.jc(abort, code.T_NEAR);
|
||||
}
|
||||
// mask away attributes
|
||||
if (ctx.conf.page_table_pointer_mask == 0) {
|
||||
code.test(page, page);
|
||||
} else if (std::in_range<s32>(ctx.conf.page_table_pointer_mask)) {
|
||||
code.and_(page, ctx.conf.page_table_pointer_mask);
|
||||
} else {
|
||||
code.mov(tmp, ctx.conf.page_table_pointer_mask);
|
||||
code.and_(page, tmp);
|
||||
}
|
||||
if (ctx.conf.page_table_sign_extension) {
|
||||
code.shl(page, *ctx.conf.page_table_sign_extension);
|
||||
code.sar(page, *ctx.conf.page_table_sign_extension);
|
||||
}
|
||||
|
||||
code.jz(abort, code.T_NEAR);
|
||||
if (ctx.conf.absolute_offset_page_table) {
|
||||
return page + vaddr;
|
||||
|
||||
@@ -159,14 +159,23 @@ struct UserConfig {
|
||||
/// Maximum size is limited by the maximum length of a x86_64 / arm64 jump.
|
||||
std::uint32_t code_cache_size = 128 * 1024 * 1024; // bytes
|
||||
|
||||
/// Masks out the first N bits in host pointers from the page table.
|
||||
/// Applies a bit mask to the bits in host pointers from the page table.
|
||||
/// The intention behind this is to allow users of Dynarmic to pack attributes in the
|
||||
/// same integer and update the pointer attribute pair atomically.
|
||||
/// If the configured value is 3, all pointers will be forcefully aligned to 8 bytes.
|
||||
std::int32_t page_table_pointer_mask_bits = 0;
|
||||
/// If the configured value is ~(0b111ULL), all pointers will be forcefully aligned to 8 bytes.
|
||||
std::uint64_t page_table_pointer_mask = 0;
|
||||
|
||||
// Log2 of the size per page entry, value should be either 3 or 4
|
||||
std::size_t page_table_log2_stride = 3;
|
||||
/// Log2 of the size per page entry, value should be either 3 or 4
|
||||
std::uint32_t page_table_log2_stride = 3;
|
||||
|
||||
/// Setting this value has Dynarmic check the specified bit of the page pointer provided by page table.
|
||||
/// If the bit is set to 1, Dynarmic will treat it as unmapped.
|
||||
/// This bit should be included as part of `page_table_pointer_mask_bits`.
|
||||
std::optional<std::uint8_t> page_table_marked_bit = std::nullopt;
|
||||
|
||||
/// If this value is set, Dynarmic will sign extend the page table pointer by this bit.
|
||||
/// Useful for compacting bits into the page table and should be used as part of `page_table_pointer_mask`.
|
||||
std::optional<std::uint8_t> page_table_sign_extension = std::nullopt;
|
||||
|
||||
/// Select the architecture version to use.
|
||||
/// There are minor behavioural differences between versions.
|
||||
|
||||
@@ -173,14 +173,23 @@ struct UserConfig {
|
||||
/// This is only used if page_table is not nullptr.
|
||||
std::uint32_t page_table_address_space_bits = 36;
|
||||
|
||||
/// Masks out the first N bits in host pointers from the page table.
|
||||
/// Applies a bit mask to the bits in host pointers from the page table.
|
||||
/// The intention behind this is to allow users of Dynarmic to pack attributes in the
|
||||
/// same integer and update the pointer attribute pair atomically.
|
||||
/// If the configured value is 3, all pointers will be forcefully aligned to 8 bytes.
|
||||
std::int32_t page_table_pointer_mask_bits = 0;
|
||||
/// If the configured value is ~(0b111ULL), all pointers will be forcefully aligned to 8 bytes.
|
||||
std::uint64_t page_table_pointer_mask = 0;
|
||||
|
||||
// Log2 of the size per page entry, value should be either 3 or 4
|
||||
std::size_t page_table_log2_stride = 3;
|
||||
/// Log2 of the size per page entry, value should be either 3 or 4
|
||||
std::uint32_t page_table_log2_stride = 3;
|
||||
|
||||
/// Setting this value has Dynarmic check the specified bit of the page pointer provided by page table.
|
||||
/// If the bit is set to 1, Dynarmic will treat it as unmapped.
|
||||
/// This bit should be included as part of `page_table_pointer_mask`.
|
||||
std::optional<std::uint8_t> page_table_marked_bit = std::nullopt;
|
||||
|
||||
/// If this value is set, Dynarmic will sign extend the page table pointer by this bit.
|
||||
/// Useful for compacting bits into the page table and should be used as part of `page_table_pointer_mask`.
|
||||
std::optional<std::uint8_t> page_table_sign_extension = std::nullopt;
|
||||
|
||||
/// Counter-timer frequency register. The value of the register is not interpreted by
|
||||
/// dynarmic.
|
||||
|
||||
@@ -154,25 +154,6 @@ std::unique_ptr<TranslationMap> InitializeTranslations(QObject* parent) {
|
||||
INSERT(Settings, post_shader_preset, QString(), QString());
|
||||
INSERT(Settings, post_shader_enabled, tr("Enable post-processing effects"),
|
||||
tr("Applies post-processing effects to the final image."));
|
||||
INSERT(Settings, frame_gen, tr("Frame generation (Lossless Scaling)"),
|
||||
tr("Generates intermediate frames with the Lossless Scaling frame generation "
|
||||
"shaders.\nRequires your own legal copy of Lossless.dll and forces VSync (FIFO)."));
|
||||
INSERT(Settings, frame_gen_multiplier, tr("Frame generation multiplier:"),
|
||||
tr("How many frames are shown for each frame the game renders.\nOnly used when the "
|
||||
"target frame rate is 0."));
|
||||
INSERT(Settings, frame_gen_target_rate, tr("Frame generation target rate:"),
|
||||
tr("Pick the rate your display can actually show. The multiplier then rises or falls "
|
||||
"on its own to hold it.\n0 uses the fixed multiplier."));
|
||||
INSERT(Settings, frame_gen_flow_scale_auto, tr("Match motion estimation to the game"),
|
||||
tr("Estimate motion at the resolution the game actually renders instead of the "
|
||||
"upscaled output."));
|
||||
INSERT(Settings, frame_gen_flow_scale, tr("Motion estimation resolution:"),
|
||||
tr("Resolution of the optical flow pass, as a fraction of the game's render "
|
||||
"resolution.\nLowering it is the cheapest way to reclaim performance."));
|
||||
INSERT(Settings, frame_gen_queue_target, tr("Frame generation queue target:"),
|
||||
tr("How many finished frames may wait ahead of the display.\nLarger queues absorb GPU "
|
||||
"spikes at the cost of input latency."));
|
||||
INSERT(Settings, frame_gen_dump_flow, QString(), QString());
|
||||
INSERT(Settings, fullscreen_mode, tr("Fullscreen Mode:"),
|
||||
tr("The method used to render the window in fullscreen.\nBorderless offers the best "
|
||||
"compatibility with the on-screen keyboard that some games request for "
|
||||
@@ -198,6 +179,14 @@ std::unique_ptr<TranslationMap> InitializeTranslations(QObject* parent) {
|
||||
"GPU: Use the GPU's compute shaders to decode ASTC textures (recommended).\n"
|
||||
"CPU Asynchronously: Use the CPU to decode ASTC textures on demand. Eliminates"
|
||||
"ASTC decoding\nstuttering but may present artifacts."));
|
||||
INSERT(Settings, astc_recompression, tr("ASTC Recompression Method:"),
|
||||
tr("Most GPUs lack support for ASTC textures and must decompress to an"
|
||||
"intermediate format: RGBA8.\n"
|
||||
"BC1/BC3: The intermediate format will be recompressed to BC1 or BC3 format,\n"
|
||||
" saving VRAM but degrading image quality."));
|
||||
INSERT(Settings, frame_pacing_mode, tr("Frame Pacing Mode (Vulkan only)"),
|
||||
tr("Controls how the emulator manages frame pacing to reduce stuttering and make the "
|
||||
"frame rate smoother and more consistent."));
|
||||
INSERT(Settings, vram_usage_mode, tr("VRAM Usage Mode:"),
|
||||
tr("Selects whether the emulator should prefer to conserve memory or make maximum usage "
|
||||
"of available video memory for performance.\nAggressive mode may impact performance "
|
||||
@@ -222,10 +211,6 @@ std::unique_ptr<TranslationMap> InitializeTranslations(QObject* parent) {
|
||||
tr("Ensures data consistency between compute and memory operations.\nThis option fixes "
|
||||
"issues in games, but may degrade performance.\nUnreal Engine 4 games often see the "
|
||||
"most significant changes thereof."));
|
||||
INSERT(Settings, stall_on_gpu_fence_wait, tr("Pause CPU on GPU fence timeouts"),
|
||||
tr("When a game keeps timing out while waiting for the GPU, pauses every CPU thread "
|
||||
"until the GPU catches up.\nDisabling it keeps the other threads running, which can "
|
||||
"reduce stutter but may break games that rely on it."));
|
||||
INSERT(Settings, async_presentation, tr("Enable asynchronous presentation (Vulkan only)"),
|
||||
tr("Slightly improves performance by moving presentation to a separate CPU thread."));
|
||||
INSERT(
|
||||
@@ -249,6 +234,23 @@ std::unique_ptr<TranslationMap> InitializeTranslations(QObject* parent) {
|
||||
INSERT(Settings, gpu_clock, tr("GPU Clocks"),
|
||||
tr("Makes the game believe GPU work finishes faster than it does, so it stops lowering "
|
||||
"resolution and render distance to fit the Switch's clocks."));
|
||||
INSERT(Settings, gpu_unswizzle_enabled, tr("GPU Unswizzle"),
|
||||
tr("Accelerates BCn 3D texture decoding using GPU compute.\n"
|
||||
"Disable if experiencing crashes or graphical glitches."));
|
||||
INSERT(Settings, gpu_unswizzle_texture_size, tr("GPU Unswizzle Max Texture Size"),
|
||||
tr("Sets the maximum size (MiB) for GPU-based texture unswizzling.\n"
|
||||
"While the GPU is faster for medium and large textures, the CPU may be more "
|
||||
"efficient for very small ones.\n"
|
||||
"Adjust this to find the balance between GPU acceleration and CPU overhead."));
|
||||
INSERT(Settings, gpu_unswizzle_stream_size, tr("GPU Unswizzle Stream Size"),
|
||||
tr("Sets the maximum amount of texture data (in MiB) processed per frame.\n"
|
||||
"Higher values can reduce stutter during texture loading but may impact frame "
|
||||
"consistency."));
|
||||
INSERT(Settings, gpu_unswizzle_chunk_size, tr("GPU Unswizzle Chunk Size"),
|
||||
tr("Determines the number of depth slices processed in a single dispatch.\n"
|
||||
"Increasing this can improve throughput on high-end GPUs but may cause TDR or driver "
|
||||
"timeouts on weaker hardware."));
|
||||
|
||||
INSERT(Settings, use_vulkan_driver_pipeline_cache, tr("Use Vulkan pipeline cache"),
|
||||
tr("Enables GPU vendor-specific pipeline cache.\nThis option can improve shader loading "
|
||||
"time significantly in cases where the Vulkan driver does not store pipeline cache "
|
||||
@@ -394,6 +396,21 @@ std::unique_ptr<ComboboxTranslationMap> ComboboxEnumeration(QObject* parent) {
|
||||
PAIR(AstcDecodeMode, Gpu, tr("GPU")),
|
||||
PAIR(AstcDecodeMode, CpuAsynchronous, tr("CPU Asynchronous")),
|
||||
}});
|
||||
translations->insert(
|
||||
{Settings::EnumMetadata<Settings::AstcRecompression>::Index(),
|
||||
{
|
||||
PAIR(AstcRecompression, Uncompressed, tr("Uncompressed (Best quality)")),
|
||||
PAIR(AstcRecompression, Bc1, tr("BC1 (Low quality)")),
|
||||
PAIR(AstcRecompression, Bc3, tr("BC3 (Medium quality)")),
|
||||
}});
|
||||
translations->insert({Settings::EnumMetadata<Settings::FramePacingMode>::Index(),
|
||||
{
|
||||
PAIR(FramePacingMode, Target_Auto, tr("Auto")),
|
||||
PAIR(FramePacingMode, Target_30, tr("30 FPS")),
|
||||
PAIR(FramePacingMode, Target_60, tr("60 FPS")),
|
||||
PAIR(FramePacingMode, Target_90, tr("90 FPS")),
|
||||
PAIR(FramePacingMode, Target_120, tr("120 FPS")),
|
||||
}});
|
||||
translations->insert({Settings::EnumMetadata<Settings::VramUsageMode>::Index(),
|
||||
{
|
||||
PAIR(VramUsageMode, Conservative, tr("Conservative")),
|
||||
@@ -637,6 +654,30 @@ std::unique_ptr<ComboboxTranslationMap> ComboboxEnumeration(QObject* parent) {
|
||||
PAIR(GpuClock, Boost, tr("Boost")),
|
||||
PAIR(GpuClock, Overclock, tr("Overclock")),
|
||||
}});
|
||||
translations->insert({Settings::EnumMetadata<Settings::GpuUnswizzleSize>::Index(),
|
||||
{
|
||||
PAIR(GpuUnswizzleSize, VerySmall, tr("Very Small (16 MB)")),
|
||||
PAIR(GpuUnswizzleSize, Small, tr("Small (32 MB)")),
|
||||
PAIR(GpuUnswizzleSize, Normal, tr("Normal (128 MB)")),
|
||||
PAIR(GpuUnswizzleSize, Large, tr("Large (256 MB)")),
|
||||
PAIR(GpuUnswizzleSize, VeryLarge, tr("Very Large (512 MB)")),
|
||||
}});
|
||||
translations->insert({Settings::EnumMetadata<Settings::GpuUnswizzle>::Index(),
|
||||
{
|
||||
PAIR(GpuUnswizzle, VeryLow, tr("Very Low (4 MB)")),
|
||||
PAIR(GpuUnswizzle, Low, tr("Low (8 MB)")),
|
||||
PAIR(GpuUnswizzle, Normal, tr("Normal (16 MB)")),
|
||||
PAIR(GpuUnswizzle, Medium, tr("Medium (32 MB)")),
|
||||
PAIR(GpuUnswizzle, High, tr("High (64 MB)")),
|
||||
}});
|
||||
translations->insert({Settings::EnumMetadata<Settings::GpuUnswizzleChunk>::Index(),
|
||||
{
|
||||
PAIR(GpuUnswizzleChunk, VeryLow, tr("Very Low (32)")),
|
||||
PAIR(GpuUnswizzleChunk, Low, tr("Low (64)")),
|
||||
PAIR(GpuUnswizzleChunk, Normal, tr("Normal (128)")),
|
||||
PAIR(GpuUnswizzleChunk, Medium, tr("Medium (256)")),
|
||||
PAIR(GpuUnswizzleChunk, High, tr("High (512)")),
|
||||
}});
|
||||
|
||||
translations->insert({Settings::EnumMetadata<Settings::ExtendedDynamicState>::Index(),
|
||||
{
|
||||
|
||||
@@ -262,5 +262,6 @@ Q_DECLARE_METATYPE(Settings::ResolutionSetup);
|
||||
Q_DECLARE_METATYPE(Settings::ScalingFilter);
|
||||
Q_DECLARE_METATYPE(Settings::AntiAliasing);
|
||||
Q_DECLARE_METATYPE(Settings::RendererBackend);
|
||||
Q_DECLARE_METATYPE(Settings::AstcRecompression);
|
||||
Q_DECLARE_METATYPE(Settings::AstcDecodeMode);
|
||||
Q_DECLARE_METATYPE(Settings::SpirvOptimizeMode);
|
||||
|
||||
@@ -172,7 +172,6 @@ add_library(shader_recompiler STATIC
|
||||
ir_opt/constant_propagation_pass.cpp
|
||||
ir_opt/dead_code_elimination_pass.cpp
|
||||
ir_opt/dual_vertex_pass.cpp
|
||||
ir_opt/geometry_compaction_pass.cpp
|
||||
ir_opt/global_memory_to_storage_buffer_pass.cpp
|
||||
ir_opt/identity_removal_pass.cpp
|
||||
ir_opt/layer_pass.cpp
|
||||
|
||||
@@ -1,6 +1,3 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
@@ -35,10 +32,10 @@ void GlobalStorageOp(EmitContext& ctx, Register address, bool pointer_based, std
|
||||
std::string_view else_expr = {}) {
|
||||
const size_t num_buffers{ctx.info.storage_buffers_descriptors.size()};
|
||||
for (size_t index = 0; index < num_buffers; ++index) {
|
||||
const auto& ssbo{ctx.info.storage_buffers_descriptors[index]};
|
||||
if (!ssbo.is_global_fallback) {
|
||||
if (!ctx.info.nvn_buffer_used[index]) {
|
||||
continue;
|
||||
}
|
||||
const auto& ssbo{ctx.info.storage_buffers_descriptors[index]};
|
||||
const u64 ssbo_align_mask{~(ctx.profile.min_ssbo_alignment - 1U)};
|
||||
ctx.Add("LDC.U64 DC.x,c{}[{}];" // unaligned_ssbo_addr
|
||||
"AND.U64 DC.x,DC.x,{};" // ssbo_addr = unaligned_ssbo_addr & ssbo_align_mask
|
||||
@@ -68,10 +65,9 @@ void GlobalStorageOp(EmitContext& ctx, Register address, bool pointer_based, std
|
||||
if (!else_expr.empty()) {
|
||||
ctx.Add("{}", else_expr);
|
||||
}
|
||||
for (const auto& ssbo : ctx.info.storage_buffers_descriptors) {
|
||||
if (ssbo.is_global_fallback) {
|
||||
ctx.Add("ENDIF;");
|
||||
}
|
||||
const size_t num_used_buffers{ctx.info.nvn_buffer_used.count()};
|
||||
for (size_t index = 0; index < num_used_buffers; ++index) {
|
||||
ctx.Add("ENDIF;");
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -633,7 +633,7 @@ std::string EmitContext::DefineGlobalMemoryFunctions() {
|
||||
std::string load_func_128{"uvec4 LoadGlobal128(uint64_t addr){"};
|
||||
const size_t num_buffers{info.storage_buffers_descriptors.size()};
|
||||
for (size_t index = 0; index < num_buffers; ++index) {
|
||||
if (!info.storage_buffers_descriptors[index].is_global_fallback) {
|
||||
if (!info.nvn_buffer_used[index]) {
|
||||
continue;
|
||||
}
|
||||
define_body(write_func, index, "{0}[uint(addr-{1})>>2]=data;return;}}");
|
||||
|
||||
@@ -4,7 +4,6 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <algorithm>
|
||||
#include <span>
|
||||
#include <tuple>
|
||||
#include <type_traits>
|
||||
@@ -440,13 +439,10 @@ void SetupCapabilities(const Profile& profile, const Info& info, EmitContext& ct
|
||||
ctx.AddCapability(spv::Capability::DrawParameters);
|
||||
}
|
||||
if ((info.uses_subgroup_vote || info.uses_subgroup_invocation_id ||
|
||||
info.uses_subgroup_shuffles || info.uses_subgroup_mask) &&
|
||||
info.uses_subgroup_shuffles) &&
|
||||
profile.support_vote && profile.SupportsSubgroupStage(ctx.stage)) {
|
||||
ctx.AddCapability(spv::Capability::GroupNonUniformBallot);
|
||||
ctx.AddCapability(spv::Capability::GroupNonUniformShuffle);
|
||||
if (info.uses_subgroup_shuffles && profile.support_shuffle_relative) {
|
||||
ctx.AddCapability(spv::Capability::GroupNonUniformShuffleRelative);
|
||||
}
|
||||
if (!profile.warp_size_potentially_larger_than_guest) {
|
||||
// vote ops are only used when not taking the long path
|
||||
ctx.AddCapability(spv::Capability::GroupNonUniformVote);
|
||||
@@ -525,29 +521,6 @@ void PatchPhiNodes(IR::Program& program, EmitContext& ctx) {
|
||||
return { ctx.Def(phi->Arg(phi_arg)), parent };
|
||||
});
|
||||
}
|
||||
|
||||
void RewriteOpcodes(std::vector<u32>& code, std::span<const std::pair<u32, spv::Op>> rewrites) {
|
||||
if (rewrites.empty()) {
|
||||
return;
|
||||
}
|
||||
size_t offset = 5;
|
||||
while (offset + 2 < code.size()) {
|
||||
const u32 word_count{code[offset] >> 16};
|
||||
const auto opcode{static_cast<spv::Op>(code[offset] & 0xFFFFu)};
|
||||
if (word_count == 0) {
|
||||
return;
|
||||
}
|
||||
if (opcode == spv::Op::OpGroupNonUniformShuffleXor ||
|
||||
opcode == spv::Op::OpGroupNonUniformQuadBroadcast) {
|
||||
const auto it{std::ranges::find(rewrites, code[offset + 2],
|
||||
&std::pair<u32, spv::Op>::first)};
|
||||
if (it != rewrites.end()) {
|
||||
code[offset] = (code[offset] & 0xFFFF0000u) | static_cast<u32>(it->second);
|
||||
}
|
||||
}
|
||||
offset += word_count;
|
||||
}
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
std::vector<u32> EmitSPIRV(const Profile& profile, const RuntimeInfo& runtime_info, IR::Program& program, Bindings& bindings) {
|
||||
@@ -562,9 +535,7 @@ std::vector<u32> EmitSPIRV(const Profile& profile, const RuntimeInfo& runtime_in
|
||||
SetupCapabilities(profile, program.info, ctx);
|
||||
SetupTransformFeedbackCapabilities(ctx, main);
|
||||
PatchPhiNodes(program, ctx);
|
||||
std::vector<u32> code{ctx.Assemble()};
|
||||
RewriteOpcodes(code, ctx.opcode_rewrites);
|
||||
return code;
|
||||
return ctx.Assemble();
|
||||
}
|
||||
|
||||
Id EmitPhi(EmitContext& ctx, IR::Inst* inst) {
|
||||
|
||||
@@ -417,48 +417,48 @@ Id EmitStorageAtomicMaxF32x2(EmitContext& ctx, const IR::Value& binding, const I
|
||||
return ctx.OpPackHalf2x16(ctx.U32[1], result);
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicIAdd32(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicIAdd32, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicIAdd32(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicSMin32(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicSMin32, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicSMin32(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicUMin32(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicUMin32, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicUMin32(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicSMax32(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicSMax32, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicSMax32(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicUMax32(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicUMax32, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicUMax32(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicInc32(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicInc32, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicInc32(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicDec32(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicDec32, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicDec32(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicAnd32(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicAnd32, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicAnd32(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicOr32(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicOr32, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicOr32(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicXor32(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicXor32, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicXor32(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicExchange32(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicExchange32, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicExchange32(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicIAdd64(EmitContext&) {
|
||||
@@ -549,32 +549,32 @@ Id EmitGlobalAtomicExchange32x2(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicAddF32(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicAddF32, ctx.F32[1], address, value);
|
||||
Id EmitGlobalAtomicAddF32(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicAddF16x2(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicAddF16x2, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicAddF16x2(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicAddF32x2(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicAddF32x2, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicAddF32x2(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicMinF16x2(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicMinF16x2, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicMinF16x2(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicMinF32x2(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicMinF32x2, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicMinF32x2(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicMaxF16x2(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicMaxF16x2, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicMaxF16x2(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitGlobalAtomicMaxF32x2(EmitContext& ctx, Id address, Id value) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::GlobalAtomicMaxF32x2, ctx.U32[1], address, value);
|
||||
Id EmitGlobalAtomicMaxF32x2(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
} // namespace Shader::Backend::SPIRV
|
||||
|
||||
@@ -90,17 +90,17 @@ Id EmitUndefU8(EmitContext& ctx);
|
||||
Id EmitUndefU16(EmitContext& ctx);
|
||||
Id EmitUndefU32(EmitContext& ctx);
|
||||
Id EmitUndefU64(EmitContext& ctx);
|
||||
Id EmitLoadGlobalU8(EmitContext& ctx, Id address);
|
||||
Id EmitLoadGlobalS8(EmitContext& ctx, Id address);
|
||||
Id EmitLoadGlobalU16(EmitContext& ctx, Id address);
|
||||
Id EmitLoadGlobalS16(EmitContext& ctx, Id address);
|
||||
void EmitLoadGlobalU8(EmitContext& ctx);
|
||||
void EmitLoadGlobalS8(EmitContext& ctx);
|
||||
void EmitLoadGlobalU16(EmitContext& ctx);
|
||||
void EmitLoadGlobalS16(EmitContext& ctx);
|
||||
Id EmitLoadGlobal32(EmitContext& ctx, Id address);
|
||||
Id EmitLoadGlobal64(EmitContext& ctx, Id address);
|
||||
Id EmitLoadGlobal128(EmitContext& ctx, Id address);
|
||||
void EmitWriteGlobalU8(EmitContext& ctx, Id address, Id value);
|
||||
void EmitWriteGlobalS8(EmitContext& ctx, Id address, Id value);
|
||||
void EmitWriteGlobalU16(EmitContext& ctx, Id address, Id value);
|
||||
void EmitWriteGlobalS16(EmitContext& ctx, Id address, Id value);
|
||||
void EmitWriteGlobalU8(EmitContext& ctx);
|
||||
void EmitWriteGlobalS8(EmitContext& ctx);
|
||||
void EmitWriteGlobalU16(EmitContext& ctx);
|
||||
void EmitWriteGlobalS16(EmitContext& ctx);
|
||||
void EmitWriteGlobal32(EmitContext& ctx, Id address, Id value);
|
||||
void EmitWriteGlobal64(EmitContext& ctx, Id address, Id value);
|
||||
void EmitWriteGlobal128(EmitContext& ctx, Id address, Id value);
|
||||
@@ -415,17 +415,17 @@ Id EmitStorageAtomicMaxF16x2(EmitContext& ctx, const IR::Value& binding, const I
|
||||
Id value);
|
||||
Id EmitStorageAtomicMaxF32x2(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value);
|
||||
Id EmitGlobalAtomicIAdd32(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicSMin32(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicUMin32(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicSMax32(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicUMax32(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicInc32(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicDec32(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicAnd32(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicOr32(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicXor32(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicExchange32(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicIAdd32(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicSMin32(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicUMin32(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicSMax32(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicUMax32(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicInc32(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicDec32(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicAnd32(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicOr32(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicXor32(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicExchange32(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicIAdd64(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicSMin64(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicUMin64(EmitContext& ctx);
|
||||
@@ -448,13 +448,13 @@ Id EmitGlobalAtomicAnd32x2(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicOr32x2(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicXor32x2(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicExchange32x2(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicAddF32(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicAddF16x2(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicAddF32x2(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicMinF16x2(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicMinF32x2(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicMaxF16x2(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicMaxF32x2(EmitContext& ctx, Id address, Id value);
|
||||
Id EmitGlobalAtomicAddF32(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicAddF16x2(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicAddF32x2(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicMinF16x2(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicMinF32x2(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicMaxF16x2(EmitContext& ctx);
|
||||
Id EmitGlobalAtomicMaxF32x2(EmitContext& ctx);
|
||||
Id EmitLogicalOr(EmitContext& ctx, Id a, Id b);
|
||||
Id EmitLogicalAnd(EmitContext& ctx, Id a, Id b);
|
||||
Id EmitLogicalXor(EmitContext& ctx, Id a, Id b);
|
||||
|
||||
@@ -69,71 +69,93 @@ void WriteStorage32(EmitContext& ctx, const IR::Value& binding, const IR::Value&
|
||||
&StorageDefinitions::U32, index_offset);
|
||||
}
|
||||
|
||||
void WriteStorageBits(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value, Id bit_offset, Id bit_count) {
|
||||
void WriteStorageByCasLoop(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset,
|
||||
Id value, Id bit_offset, Id bit_count) {
|
||||
const Id pointer{StoragePointer(ctx, binding, offset, ctx.storage_types.U32, sizeof(u32),
|
||||
&StorageDefinitions::U32)};
|
||||
ctx.AtomicBitFieldInsert(pointer, value, bit_offset, bit_count);
|
||||
ctx.OpFunctionCall(ctx.TypeVoid(), ctx.write_storage_cas_loop_func, pointer, value, bit_offset,
|
||||
bit_count);
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
Id EmitLoadGlobalU8(EmitContext& ctx, Id address) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::LoadGlobalU8, ctx.U32[1], address, ctx.u32_zero_value);
|
||||
void EmitLoadGlobalU8(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitLoadGlobalS8(EmitContext& ctx, Id address) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::LoadGlobalS8, ctx.U32[1], address, ctx.u32_zero_value);
|
||||
void EmitLoadGlobalS8(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitLoadGlobalU16(EmitContext& ctx, Id address) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::LoadGlobalU16, ctx.U32[1], address,
|
||||
ctx.u32_zero_value);
|
||||
void EmitLoadGlobalU16(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitLoadGlobalS16(EmitContext& ctx, Id address) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::LoadGlobalS16, ctx.U32[1], address,
|
||||
ctx.u32_zero_value);
|
||||
void EmitLoadGlobalS16(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
Id EmitLoadGlobal32(EmitContext& ctx, Id address) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::LoadGlobal32, ctx.U32[1], address, ctx.u32_zero_value);
|
||||
if (ctx.profile.support_int64) {
|
||||
return ctx.OpFunctionCall(ctx.U32[1], ctx.load_global_func_u32, address);
|
||||
}
|
||||
LOG_WARNING(Shader_SPIRV, "Int64 not supported, ignoring memory operation");
|
||||
return ctx.Const(0u);
|
||||
}
|
||||
|
||||
Id EmitLoadGlobal64(EmitContext& ctx, Id address) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::LoadGlobal64, ctx.U32[2], address, ctx.u32_zero_value);
|
||||
if (ctx.profile.support_int64) {
|
||||
return ctx.OpFunctionCall(ctx.U32[2], ctx.load_global_func_u32x2, address);
|
||||
}
|
||||
LOG_WARNING(Shader_SPIRV, "Int64 not supported, ignoring memory operation");
|
||||
return ctx.Const(0u, 0u);
|
||||
}
|
||||
|
||||
Id EmitLoadGlobal128(EmitContext& ctx, Id address) {
|
||||
return ctx.CallGlobalMemory(IR::Opcode::LoadGlobal128, ctx.U32[4], address,
|
||||
ctx.u32_zero_value);
|
||||
if (ctx.profile.support_int64) {
|
||||
return ctx.OpFunctionCall(ctx.U32[4], ctx.load_global_func_u32x4, address);
|
||||
}
|
||||
LOG_WARNING(Shader_SPIRV, "Int64 not supported, ignoring memory operation");
|
||||
return ctx.Const(0u, 0u, 0u, 0u);
|
||||
}
|
||||
|
||||
void EmitWriteGlobalU8(EmitContext& ctx, Id address, Id value) {
|
||||
ctx.CallGlobalMemory(IR::Opcode::WriteGlobalU8, ctx.void_id, address, value);
|
||||
void EmitWriteGlobalU8(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
void EmitWriteGlobalS8(EmitContext& ctx, Id address, Id value) {
|
||||
ctx.CallGlobalMemory(IR::Opcode::WriteGlobalS8, ctx.void_id, address, value);
|
||||
void EmitWriteGlobalS8(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
void EmitWriteGlobalU16(EmitContext& ctx, Id address, Id value) {
|
||||
ctx.CallGlobalMemory(IR::Opcode::WriteGlobalU16, ctx.void_id, address, value);
|
||||
void EmitWriteGlobalU16(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
void EmitWriteGlobalS16(EmitContext& ctx, Id address, Id value) {
|
||||
ctx.CallGlobalMemory(IR::Opcode::WriteGlobalS16, ctx.void_id, address, value);
|
||||
void EmitWriteGlobalS16(EmitContext&) {
|
||||
throw NotImplementedException("SPIR-V Instruction");
|
||||
}
|
||||
|
||||
void EmitWriteGlobal32(EmitContext& ctx, Id address, Id value) {
|
||||
ctx.CallGlobalMemory(IR::Opcode::WriteGlobal32, ctx.void_id, address, value);
|
||||
if (ctx.profile.support_int64) {
|
||||
ctx.OpFunctionCall(ctx.void_id, ctx.write_global_func_u32, address, value);
|
||||
return;
|
||||
}
|
||||
LOG_WARNING(Shader_SPIRV, "Int64 not supported, ignoring memory operation");
|
||||
}
|
||||
|
||||
void EmitWriteGlobal64(EmitContext& ctx, Id address, Id value) {
|
||||
ctx.CallGlobalMemory(IR::Opcode::WriteGlobal64, ctx.void_id, address, value);
|
||||
if (ctx.profile.support_int64) {
|
||||
ctx.OpFunctionCall(ctx.void_id, ctx.write_global_func_u32x2, address, value);
|
||||
return;
|
||||
}
|
||||
LOG_WARNING(Shader_SPIRV, "Int64 not supported, ignoring memory operation");
|
||||
}
|
||||
|
||||
void EmitWriteGlobal128(EmitContext& ctx, Id address, Id value) {
|
||||
ctx.CallGlobalMemory(IR::Opcode::WriteGlobal128, ctx.void_id, address, value);
|
||||
if (ctx.profile.support_int64) {
|
||||
ctx.OpFunctionCall(ctx.void_id, ctx.write_global_func_u32x4, address, value);
|
||||
return;
|
||||
}
|
||||
LOG_WARNING(Shader_SPIRV, "Int64 not supported, ignoring memory operation");
|
||||
}
|
||||
|
||||
Id EmitLoadStorageU8(EmitContext& ctx, const IR::Value& binding, const IR::Value& offset) {
|
||||
@@ -217,7 +239,7 @@ void EmitWriteStorageU8(EmitContext& ctx, const IR::Value& binding, const IR::Va
|
||||
WriteStorage(ctx, binding, offset, ctx.OpSConvert(ctx.U8, value), ctx.storage_types.U8,
|
||||
sizeof(u8), &StorageDefinitions::U8);
|
||||
} else {
|
||||
WriteStorageBits(ctx, binding, offset, value, ctx.BitOffset8(offset), ctx.Const(8u));
|
||||
WriteStorageByCasLoop(ctx, binding, offset, value, ctx.BitOffset8(offset), ctx.Const(8u));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -228,7 +250,7 @@ void EmitWriteStorageS8(EmitContext& ctx, const IR::Value& binding, const IR::Va
|
||||
WriteStorage(ctx, binding, offset, ctx.OpSConvert(ctx.S8, value), ctx.storage_types.S8,
|
||||
sizeof(s8), &StorageDefinitions::S8);
|
||||
} else {
|
||||
WriteStorageBits(ctx, binding, offset, value, ctx.BitOffset8(offset), ctx.Const(8u));
|
||||
WriteStorageByCasLoop(ctx, binding, offset, value, ctx.BitOffset8(offset), ctx.Const(8u));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -239,7 +261,7 @@ void EmitWriteStorageU16(EmitContext& ctx, const IR::Value& binding, const IR::V
|
||||
WriteStorage(ctx, binding, offset, ctx.OpSConvert(ctx.U16, value), ctx.storage_types.U16,
|
||||
sizeof(u16), &StorageDefinitions::U16);
|
||||
} else {
|
||||
WriteStorageBits(ctx, binding, offset, value, ctx.BitOffset16(offset), ctx.Const(16u));
|
||||
WriteStorageByCasLoop(ctx, binding, offset, value, ctx.BitOffset16(offset), ctx.Const(16u));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -250,7 +272,7 @@ void EmitWriteStorageS16(EmitContext& ctx, const IR::Value& binding, const IR::V
|
||||
WriteStorage(ctx, binding, offset, ctx.OpSConvert(ctx.S16, value), ctx.storage_types.S16,
|
||||
sizeof(s16), &StorageDefinitions::S16);
|
||||
} else {
|
||||
WriteStorageBits(ctx, binding, offset, value, ctx.BitOffset16(offset), ctx.Const(16u));
|
||||
WriteStorageByCasLoop(ctx, binding, offset, value, ctx.BitOffset16(offset), ctx.Const(16u));
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
@@ -77,57 +77,20 @@ Id GetMaxThreadId(EmitContext& ctx, Id thread_id, Id clamp, Id segmentation_mask
|
||||
return ComputeMaxThreadId(ctx, min_thread_id, clamp, not_seg_mask);
|
||||
}
|
||||
|
||||
Id HostThreadId(EmitContext& ctx, Id thread_id) {
|
||||
if (!ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
return thread_id;
|
||||
Id SelectValue(EmitContext& ctx, Id in_range, Id value, Id src_thread_id) {
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return value;
|
||||
}
|
||||
return ctx.OpSelect(
|
||||
ctx.U32[1], in_range,
|
||||
ctx.OpGroupNonUniformShuffle(ctx.U32[1], SubgroupScope(ctx), value, src_thread_id), value);
|
||||
}
|
||||
|
||||
Id AddPartitionBase(EmitContext& ctx, Id thread_id) {
|
||||
const Id partition_idx{ctx.OpShiftRightLogical(ctx.U32[1], GetThreadId(ctx), ctx.Const(5u))};
|
||||
const Id partition_base{ctx.OpShiftLeftLogical(ctx.U32[1], partition_idx, ctx.Const(5u))};
|
||||
return ctx.OpIAdd(ctx.U32[1], thread_id, partition_base);
|
||||
}
|
||||
|
||||
Id GuestLane(EmitContext& ctx, Id index) {
|
||||
return ctx.OpBitwiseAnd(ctx.U32[1], index, ctx.Const(31U));
|
||||
}
|
||||
|
||||
Id ShuffleAbsolute(EmitContext& ctx, Id value, Id src_thread_id) {
|
||||
if (!ctx.profile.has_broken_spirv_subgroup_shuffle) {
|
||||
return ctx.OpGroupNonUniformShuffle(ctx.U32[1], SubgroupScope(ctx), value, src_thread_id);
|
||||
}
|
||||
Id result{ctx.u32_zero_value};
|
||||
for (u32 lane = 0; lane < ctx.profile.max_subgroup_size; ++lane) {
|
||||
const Id read{
|
||||
ctx.OpGroupNonUniformBroadcast(ctx.U32[1], SubgroupScope(ctx), value, ctx.Const(lane))};
|
||||
const Id matches{ctx.OpIEqual(ctx.U1, src_thread_id, ctx.Const(lane))};
|
||||
result = ctx.OpSelect(ctx.U32[1], matches, read, result);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
Id ShuffleRelative(EmitContext& ctx, Id value, Id delta, Id src_thread_id, spv::Op op) {
|
||||
if (!ctx.profile.support_shuffle_relative) {
|
||||
return ShuffleAbsolute(ctx, value, HostThreadId(ctx, src_thread_id));
|
||||
}
|
||||
const Id result{ctx.OpGroupNonUniformShuffleXor(ctx.U32[1], SubgroupScope(ctx), value, delta)};
|
||||
ctx.opcode_rewrites.emplace_back(result.value, op);
|
||||
return result;
|
||||
}
|
||||
|
||||
Id BroadcastLane(EmitContext& ctx, Id value, u32 lane) {
|
||||
Id result{
|
||||
ctx.OpGroupNonUniformBroadcast(ctx.U32[1], SubgroupScope(ctx), value, ctx.Const(lane))};
|
||||
if (!ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
return result;
|
||||
}
|
||||
const Id partition_idx{ctx.OpShiftRightLogical(ctx.U32[1], GetThreadId(ctx), ctx.Const(5u))};
|
||||
for (u32 base = 32; base < ctx.profile.max_subgroup_size; base += 32) {
|
||||
const Id read{ctx.OpGroupNonUniformBroadcast(ctx.U32[1], SubgroupScope(ctx), value,
|
||||
ctx.Const(base + lane))};
|
||||
const Id matches{ctx.OpIEqual(ctx.U1, partition_idx, ctx.Const(base >> 5))};
|
||||
result = ctx.OpSelect(ctx.U32[1], matches, read, result);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
Id EmitLaneId(EmitContext& ctx) {
|
||||
@@ -240,75 +203,61 @@ Id EmitShuffleIndex(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id cla
|
||||
const Id min_thread_id{ComputeMinThreadId(ctx, thread_id, segmentation_mask)};
|
||||
const Id max_thread_id{ComputeMaxThreadId(ctx, min_thread_id, clamp, not_seg_mask)};
|
||||
|
||||
const Id lhs{ctx.OpBitwiseAnd(ctx.U32[1], GuestLane(ctx, index), not_seg_mask)};
|
||||
const Id src_thread_id{ctx.OpBitwiseOr(ctx.U32[1], lhs, min_thread_id)};
|
||||
const Id lhs{ctx.OpBitwiseAnd(ctx.U32[1], index, not_seg_mask)};
|
||||
Id src_thread_id{ctx.OpBitwiseOr(ctx.U32[1], lhs, min_thread_id)};
|
||||
const Id in_range{ctx.OpSLessThanEqual(ctx.U1, src_thread_id, max_thread_id)};
|
||||
|
||||
if (ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
src_thread_id = AddPartitionBase(ctx, src_thread_id);
|
||||
}
|
||||
|
||||
SetInBoundsFlag(inst, in_range);
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return value;
|
||||
}
|
||||
const IR::Value lane{inst->Arg(1).Resolve()};
|
||||
const IR::Value segment{inst->Arg(3).Resolve()};
|
||||
if (lane.IsImmediate() && segment.IsImmediate() && segment.U32() == 0) {
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, BroadcastLane(ctx, value, lane.U32() & 31),
|
||||
value);
|
||||
}
|
||||
const Id shuffled{ShuffleAbsolute(ctx, value, HostThreadId(ctx, src_thread_id))};
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, shuffled, value);
|
||||
return SelectValue(ctx, in_range, value, src_thread_id);
|
||||
}
|
||||
|
||||
Id EmitShuffleUp(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id delta{GuestLane(ctx, index)};
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id max_thread_id{GetMaxThreadId(ctx, thread_id, clamp, segmentation_mask)};
|
||||
const Id src_thread_id{ctx.OpISub(ctx.U32[1], thread_id, delta)};
|
||||
Id src_thread_id{ctx.OpISub(ctx.U32[1], thread_id, index)};
|
||||
const Id in_range{ctx.OpSGreaterThanEqual(ctx.U1, src_thread_id, max_thread_id)};
|
||||
|
||||
if (ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
src_thread_id = AddPartitionBase(ctx, src_thread_id);
|
||||
}
|
||||
|
||||
SetInBoundsFlag(inst, in_range);
|
||||
const Id shuffled{ShuffleRelative(ctx, value, delta, src_thread_id,
|
||||
spv::Op::OpGroupNonUniformShuffleUp)};
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, shuffled, value);
|
||||
return SelectValue(ctx, in_range, value, src_thread_id);
|
||||
}
|
||||
|
||||
Id EmitShuffleDown(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id delta{GuestLane(ctx, index)};
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id max_thread_id{GetMaxThreadId(ctx, thread_id, clamp, segmentation_mask)};
|
||||
const Id src_thread_id{ctx.OpIAdd(ctx.U32[1], thread_id, delta)};
|
||||
Id src_thread_id{ctx.OpIAdd(ctx.U32[1], thread_id, index)};
|
||||
const Id in_range{ctx.OpSLessThanEqual(ctx.U1, src_thread_id, max_thread_id)};
|
||||
|
||||
if (ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
src_thread_id = AddPartitionBase(ctx, src_thread_id);
|
||||
}
|
||||
|
||||
SetInBoundsFlag(inst, in_range);
|
||||
const Id shuffled{ShuffleRelative(ctx, value, delta, src_thread_id,
|
||||
spv::Op::OpGroupNonUniformShuffleDown)};
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, shuffled, value);
|
||||
return SelectValue(ctx, in_range, value, src_thread_id);
|
||||
}
|
||||
|
||||
Id EmitShuffleButterfly(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id mask{GuestLane(ctx, index)};
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id max_thread_id{GetMaxThreadId(ctx, thread_id, clamp, segmentation_mask)};
|
||||
const Id src_thread_id{ctx.OpBitwiseXor(ctx.U32[1], thread_id, mask)};
|
||||
Id src_thread_id{ctx.OpBitwiseXor(ctx.U32[1], thread_id, index)};
|
||||
const Id in_range{ctx.OpSLessThanEqual(ctx.U1, src_thread_id, max_thread_id)};
|
||||
|
||||
if (ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
src_thread_id = AddPartitionBase(ctx, src_thread_id);
|
||||
}
|
||||
|
||||
SetInBoundsFlag(inst, in_range);
|
||||
const Id shuffled{ctx.OpGroupNonUniformShuffleXor(ctx.U32[1], SubgroupScope(ctx), value, mask)};
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, shuffled, value);
|
||||
return SelectValue(ctx, in_range, value, src_thread_id);
|
||||
}
|
||||
|
||||
Id EmitQuadBroadcast(EmitContext& ctx, Id value, Id lane) {
|
||||
@@ -318,16 +267,10 @@ Id EmitQuadBroadcast(EmitContext& ctx, Id value, Id lane) {
|
||||
const Id base{ctx.OpBitwiseAnd(ctx.U32[1], GetThreadId(ctx), ctx.Const(~3u))};
|
||||
const Id local_lane{ctx.OpBitwiseAnd(ctx.U32[1], lane, ctx.Const(3u))};
|
||||
const Id src_thread_id{ctx.OpBitwiseOr(ctx.U32[1], base, local_lane)};
|
||||
return ShuffleAbsolute(ctx, value, src_thread_id);
|
||||
return ctx.OpGroupNonUniformShuffle(ctx.U32[1], SubgroupScope(ctx), value, src_thread_id);
|
||||
}
|
||||
|
||||
Id EmitQuadSwap(EmitContext& ctx, Id value, Id direction) {
|
||||
if (ctx.profile.support_quad_shuffles) {
|
||||
const Id result{
|
||||
ctx.OpGroupNonUniformQuadBroadcast(ctx.U32[1], SubgroupScope(ctx), value, direction)};
|
||||
ctx.opcode_rewrites.emplace_back(result.value, spv::Op::OpGroupNonUniformQuadSwap);
|
||||
return result;
|
||||
}
|
||||
const Id xor_mask{ctx.OpIAdd(ctx.U32[1], direction, ctx.Const(1u))};
|
||||
return ctx.OpGroupNonUniformShuffleXor(ctx.U32[1], SubgroupScope(ctx), value, xor_mask);
|
||||
}
|
||||
|
||||
@@ -486,7 +486,8 @@ EmitContext::EmitContext(const Profile& profile_, const RuntimeInfo& runtime_inf
|
||||
DefineTextures(program.info, texture_binding, bindings.texture_scaling_index);
|
||||
DefineImages(program.info, image_binding, bindings.image_scaling_index);
|
||||
DefineAttributeMemAccess(program.info);
|
||||
DefineGlobalMemoryFunctions(program);
|
||||
DefineWriteStorageCasLoopFunction(program.info);
|
||||
DefineGlobalMemoryFunctions(program.info);
|
||||
DefineRescalingInput(program.info);
|
||||
DefineRenderArea(program.info);
|
||||
}
|
||||
@@ -531,24 +532,6 @@ Id EmitContext::BitOffset16(const IR::Value& offset) {
|
||||
return OpBitwiseAnd(U32[1], OpShiftLeftLogical(U32[1], Def(offset), Const(3u)), Const(16u));
|
||||
}
|
||||
|
||||
Id EmitContext::CallGlobalMemory(IR::Opcode opcode, Id result_type, Id address, Id value) {
|
||||
if (profile.support_int64) {
|
||||
return OpFunctionCall(result_type, global_memory_funcs.at(opcode), address, value);
|
||||
}
|
||||
if (result_type.value == void_id.value) {
|
||||
return Id{};
|
||||
}
|
||||
return ConstantNull(result_type);
|
||||
}
|
||||
|
||||
void EmitContext::AtomicBitFieldInsert(Id pointer, Id value, Id offset, Id count) {
|
||||
const Id scope{Const(static_cast<u32>(spv::Scope::Device))};
|
||||
const Id mask{OpBitFieldInsert(U32[1], u32_zero_value, Const(0xFFFFFFFFU), offset, count)};
|
||||
const Id bits{OpBitFieldInsert(U32[1], u32_zero_value, value, offset, count)};
|
||||
OpAtomicAnd(U32[1], pointer, scope, u32_zero_value, OpNot(U32[1], mask));
|
||||
OpAtomicOr(U32[1], pointer, scope, u32_zero_value, bits);
|
||||
}
|
||||
|
||||
void EmitContext::DefineCommonTypes(const Info& info) {
|
||||
void_id = TypeVoid();
|
||||
|
||||
@@ -906,256 +889,135 @@ void EmitContext::DefineAttributeMemAccess(const Info& info) {
|
||||
}
|
||||
}
|
||||
|
||||
void EmitContext::DefineGlobalMemoryFunctions(const IR::Program& program) {
|
||||
const Info& info{program.info};
|
||||
void EmitContext::DefineWriteStorageCasLoopFunction(const Info& info) {
|
||||
if (profile.support_int8 && profile.support_int16) {
|
||||
return;
|
||||
}
|
||||
if (!info.uses_int8 && !info.uses_int16) {
|
||||
return;
|
||||
}
|
||||
|
||||
AddCapability(spv::Capability::VariablePointersStorageBuffer);
|
||||
|
||||
const Id ptr_type{TypePointer(spv::StorageClass::StorageBuffer, U32[1])};
|
||||
const Id func_type{TypeFunction(void_id, ptr_type, U32[1], U32[1], U32[1])};
|
||||
const Id func{OpFunction(void_id, spv::FunctionControlMask::MaskNone, func_type)};
|
||||
const Id pointer{OpFunctionParameter(ptr_type)};
|
||||
const Id value{OpFunctionParameter(U32[1])};
|
||||
const Id bit_offset{OpFunctionParameter(U32[1])};
|
||||
const Id bit_count{OpFunctionParameter(U32[1])};
|
||||
|
||||
AddLabel();
|
||||
const Id scope_device{Const(1u)};
|
||||
const Id ordering_relaxed{u32_zero_value};
|
||||
const Id body_label{OpLabel()};
|
||||
const Id continue_label{OpLabel()};
|
||||
const Id endloop_label{OpLabel()};
|
||||
const Id beginloop_label{OpLabel()};
|
||||
OpBranch(beginloop_label);
|
||||
|
||||
AddLabel(beginloop_label);
|
||||
OpLoopMerge(endloop_label, continue_label, spv::LoopControlMask::MaskNone);
|
||||
OpBranch(body_label);
|
||||
|
||||
AddLabel(body_label);
|
||||
const Id expected_value{OpLoad(U32[1], pointer)};
|
||||
const Id desired_value{OpBitFieldInsert(U32[1], expected_value, value, bit_offset, bit_count)};
|
||||
const Id actual_value{OpAtomicCompareExchange(U32[1], pointer, scope_device, ordering_relaxed,
|
||||
ordering_relaxed, desired_value, expected_value)};
|
||||
const Id store_successful{OpIEqual(U1, expected_value, actual_value)};
|
||||
OpBranchConditional(store_successful, endloop_label, continue_label);
|
||||
|
||||
AddLabel(endloop_label);
|
||||
OpReturn();
|
||||
|
||||
AddLabel(continue_label);
|
||||
OpBranch(beginloop_label);
|
||||
|
||||
OpFunctionEnd();
|
||||
|
||||
write_storage_cas_loop_func = func;
|
||||
}
|
||||
|
||||
void EmitContext::DefineGlobalMemoryFunctions(const Info& info) {
|
||||
if (!info.uses_global_memory || !profile.support_int64) {
|
||||
return;
|
||||
}
|
||||
using DefPtr = Id StorageDefinitions::*;
|
||||
const Id zero{u32_zero_value};
|
||||
const Id scope{Const(static_cast<u32>(spv::Scope::Device))};
|
||||
const Id align_mask{Const(~(static_cast<u32>(profile.min_ssbo_alignment) - 1U))};
|
||||
const auto word_pointer{[&](Id ssbo, Id word, u32 element) {
|
||||
return OpAccessChain(storage_types.U32.element, ssbo, zero,
|
||||
OpIAdd(U32[1], word, Const(element)));
|
||||
}};
|
||||
const auto cbuf_word{[&](u32 index, u32 offset) {
|
||||
if (profile.support_descriptor_aliasing) {
|
||||
return OpLoad(U32[1], OpAccessChain(uniform_types.U32, cbufs[index].U32, zero,
|
||||
Const(offset / 4)));
|
||||
}
|
||||
const Id vector{OpLoad(U32[4], OpAccessChain(uniform_types.U32x4, cbufs[index].U32x4,
|
||||
zero, Const(offset / 16)))};
|
||||
return OpCompositeExtract(U32[1], vector, (offset / 4) % 4);
|
||||
}};
|
||||
const auto bits{[&](Id offset, u32 count) {
|
||||
return OpBitwiseAnd(U32[1], OpShiftLeftLogical(U32[1], offset, Const(3U)),
|
||||
Const(32U - count));
|
||||
}};
|
||||
const auto define{[&](IR::Opcode opcode, Id result_type, Id value_type, auto&& callback) {
|
||||
const std::array<Id, 2> params{U64, value_type};
|
||||
const Id func{OpFunction(result_type, spv::FunctionControlMask::MaskNone,
|
||||
TypeFunction(result_type, params))};
|
||||
const Id addr{OpFunctionParameter(U64)};
|
||||
const Id value{OpFunctionParameter(value_type)};
|
||||
const bool returns_value{result_type.value != void_id.value};
|
||||
const auto define_body{[&](DefPtr ssbo_member, Id addr, Id element_pointer, u32 shift,
|
||||
auto&& callback) {
|
||||
AddLabel();
|
||||
const Id addr_words{OpBitcast(U32[2], addr)};
|
||||
const Id addr_low{OpCompositeExtract(U32[1], addr_words, 0U)};
|
||||
const Id addr_high{OpCompositeExtract(U32[1], addr_words, 1U)};
|
||||
for (size_t index = 0; index < info.storage_buffers_descriptors.size(); ++index) {
|
||||
const auto& desc{info.storage_buffers_descriptors[index]};
|
||||
if (!desc.is_global_fallback) {
|
||||
const size_t num_buffers{info.storage_buffers_descriptors.size()};
|
||||
for (size_t index = 0; index < num_buffers; ++index) {
|
||||
if (!info.nvn_buffer_used[index]) {
|
||||
continue;
|
||||
}
|
||||
const Id ssbo_low{
|
||||
OpBitwiseAnd(U32[1], cbuf_word(desc.cbuf_index, desc.cbuf_offset), align_mask)};
|
||||
const Id ssbo_high{cbuf_word(desc.cbuf_index, desc.cbuf_offset + 4)};
|
||||
const Id ssbo_size{cbuf_word(desc.cbuf_index, desc.cbuf_offset + 8)};
|
||||
const Id offset{OpISub(U32[1], addr_low, ssbo_low)};
|
||||
const Id borrow{
|
||||
OpSelect(U32[1], OpULessThan(U1, addr_low, ssbo_low), Const(1U), zero)};
|
||||
const Id cond{OpLogicalAnd(U1, OpULessThan(U1, offset, ssbo_size),
|
||||
OpIEqual(U1, OpISub(U32[1], addr_high, borrow), ssbo_high))};
|
||||
const auto& ssbo{info.storage_buffers_descriptors[index]};
|
||||
const Id ssbo_addr_cbuf_offset{Const(ssbo.cbuf_offset / 8)};
|
||||
const Id ssbo_size_cbuf_offset{Const(ssbo.cbuf_offset / 4 + 2)};
|
||||
const Id ssbo_addr_pointer{OpAccessChain(
|
||||
uniform_types.U32x2, cbufs[ssbo.cbuf_index].U32x2, zero, ssbo_addr_cbuf_offset)};
|
||||
const Id ssbo_size_pointer{OpAccessChain(uniform_types.U32, cbufs[ssbo.cbuf_index].U32,
|
||||
zero, ssbo_size_cbuf_offset)};
|
||||
|
||||
const u64 ssbo_align_mask{~(profile.min_ssbo_alignment - 1U)};
|
||||
const Id unaligned_addr{OpBitcast(U64, OpLoad(U32[2], ssbo_addr_pointer))};
|
||||
const Id ssbo_addr{OpBitwiseAnd(U64, unaligned_addr, Constant(U64, ssbo_align_mask))};
|
||||
const Id ssbo_size{OpUConvert(U64, OpLoad(U32[1], ssbo_size_pointer))};
|
||||
const Id ssbo_end{OpIAdd(U64, ssbo_addr, ssbo_size)};
|
||||
const Id cond{OpLogicalAnd(U1, OpUGreaterThanEqual(U1, addr, ssbo_addr),
|
||||
OpULessThan(U1, addr, ssbo_end))};
|
||||
const Id then_label{OpLabel()};
|
||||
const Id else_label{OpLabel()};
|
||||
OpSelectionMerge(else_label, spv::SelectionControlMask::MaskNone);
|
||||
OpBranchConditional(cond, then_label, else_label);
|
||||
AddLabel(then_label);
|
||||
const Id word{OpShiftRightLogical(U32[1], offset, Const(2U))};
|
||||
const Id result{callback(ssbos[index], word, offset, value)};
|
||||
if (returns_value) {
|
||||
OpReturnValue(result);
|
||||
} else {
|
||||
OpReturn();
|
||||
}
|
||||
const Id ssbo_id{ssbos[index].*ssbo_member};
|
||||
const Id ssbo_offset{OpUConvert(U32[1], OpISub(U64, addr, ssbo_addr))};
|
||||
const Id ssbo_index{OpShiftRightLogical(U32[1], ssbo_offset, Const(shift))};
|
||||
const Id ssbo_pointer{OpAccessChain(element_pointer, ssbo_id, zero, ssbo_index)};
|
||||
callback(ssbo_pointer);
|
||||
AddLabel(else_label);
|
||||
}
|
||||
if (returns_value) {
|
||||
OpReturnValue(ConstantNull(result_type));
|
||||
} else {
|
||||
OpReturn();
|
||||
}
|
||||
}};
|
||||
const auto define_load{[&](DefPtr ssbo_member, Id element_pointer, Id type, u32 shift) {
|
||||
const Id function_type{TypeFunction(type, U64)};
|
||||
const Id func_id{OpFunction(type, spv::FunctionControlMask::MaskNone, function_type)};
|
||||
const Id addr{OpFunctionParameter(U64)};
|
||||
define_body(ssbo_member, addr, element_pointer, shift,
|
||||
[&](Id ssbo_pointer) { OpReturnValue(OpLoad(type, ssbo_pointer)); });
|
||||
OpReturnValue(ConstantNull(type));
|
||||
OpFunctionEnd();
|
||||
global_memory_funcs.emplace(opcode, func);
|
||||
return func_id;
|
||||
}};
|
||||
const auto vector_pointer{[&](const StorageDefinitions& ssbo, Id word, u32 count) {
|
||||
if (count == 2) {
|
||||
return OpAccessChain(storage_types.U32x2.element, ssbo.U32x2, zero,
|
||||
OpShiftRightLogical(U32[1], word, Const(1U)));
|
||||
}
|
||||
return OpAccessChain(storage_types.U32x4.element, ssbo.U32x4, zero,
|
||||
OpShiftRightLogical(U32[1], word, Const(2U)));
|
||||
const auto define_write{[&](DefPtr ssbo_member, Id element_pointer, Id type, u32 shift) {
|
||||
const Id function_type{TypeFunction(void_id, U64, type)};
|
||||
const Id func_id{OpFunction(void_id, spv::FunctionControlMask::MaskNone, function_type)};
|
||||
const Id addr{OpFunctionParameter(U64)};
|
||||
const Id data{OpFunctionParameter(type)};
|
||||
define_body(ssbo_member, addr, element_pointer, shift, [&](Id ssbo_pointer) {
|
||||
OpStore(ssbo_pointer, data);
|
||||
OpReturn();
|
||||
});
|
||||
OpReturn();
|
||||
OpFunctionEnd();
|
||||
return func_id;
|
||||
}};
|
||||
const auto load{[&](Id type, u32 count) {
|
||||
return [&, type, count](const StorageDefinitions& ssbo, Id word, Id, Id) {
|
||||
if (count > 1 && profile.support_descriptor_aliasing) {
|
||||
return OpLoad(type, vector_pointer(ssbo, word, count));
|
||||
}
|
||||
std::array<Id, 4> words{};
|
||||
for (u32 element = 0; element < count; ++element) {
|
||||
words[element] = OpLoad(U32[1], word_pointer(ssbo.U32, word, element));
|
||||
}
|
||||
if (count == 1) {
|
||||
return words[0];
|
||||
}
|
||||
return OpCompositeConstruct(type, std::span<const Id>(words.data(), count));
|
||||
};
|
||||
}};
|
||||
const auto store{[&](u32 count) {
|
||||
return [&, count](const StorageDefinitions& ssbo, Id word, Id, Id value) {
|
||||
if (count > 1 && profile.support_descriptor_aliasing) {
|
||||
OpStore(vector_pointer(ssbo, word, count), value);
|
||||
return Id{};
|
||||
}
|
||||
if (count == 1) {
|
||||
OpStore(word_pointer(ssbo.U32, word, 0), value);
|
||||
return Id{};
|
||||
}
|
||||
for (u32 element = 0; element < count; ++element) {
|
||||
OpStore(word_pointer(ssbo.U32, word, element),
|
||||
OpCompositeExtract(U32[1], value, element));
|
||||
}
|
||||
return Id{};
|
||||
};
|
||||
}};
|
||||
const auto extract{[&](bool is_signed, u32 count) {
|
||||
return [&, is_signed, count](const StorageDefinitions& ssbo, Id word, Id offset, Id) {
|
||||
const Id loaded{OpLoad(U32[1], word_pointer(ssbo.U32, word, 0))};
|
||||
if (is_signed) {
|
||||
return OpBitFieldSExtract(U32[1], loaded, bits(offset, count), Const(count));
|
||||
}
|
||||
return OpBitFieldUExtract(U32[1], loaded, bits(offset, count), Const(count));
|
||||
};
|
||||
}};
|
||||
const auto insert{[&](u32 count) {
|
||||
return [&, count](const StorageDefinitions& ssbo, Id word, Id offset, Id value) {
|
||||
AtomicBitFieldInsert(word_pointer(ssbo.U32, word, 0), value, bits(offset, count),
|
||||
Const(count));
|
||||
return Id{};
|
||||
};
|
||||
}};
|
||||
const auto atomic{[&](Id (Sirit::Module::*func)(Id, Id, Id, Id, Id)) {
|
||||
return [&, func](const StorageDefinitions& ssbo, Id word, Id, Id value) {
|
||||
return (this->*func)(U32[1], word_pointer(ssbo.U32, word, 0), scope, zero, value);
|
||||
};
|
||||
}};
|
||||
const auto cas{[&](Id type, Id helper) {
|
||||
return [&, type, helper](const StorageDefinitions& ssbo, Id word, Id, Id value) {
|
||||
return OpFunctionCall(type, helper, word, value, ssbo.U32);
|
||||
};
|
||||
}};
|
||||
const auto packed{[&](bool is_half, Id helper) {
|
||||
return [&, is_half, helper](const StorageDefinitions& ssbo, Id word, Id, Id value) {
|
||||
if (is_half) {
|
||||
return OpBitcast(U32[1], OpFunctionCall(F16[2], helper, word, value, ssbo.U32));
|
||||
}
|
||||
return OpPackHalf2x16(U32[1], OpFunctionCall(F32[2], helper, word, value, ssbo.U32));
|
||||
};
|
||||
}};
|
||||
for (const IR::Block* const block : program.post_order_blocks) {
|
||||
for (const IR::Inst& inst : block->Instructions()) {
|
||||
const IR::Opcode opcode{inst.GetOpcode()};
|
||||
if (global_memory_funcs.contains(opcode)) {
|
||||
continue;
|
||||
}
|
||||
switch (opcode) {
|
||||
case IR::Opcode::LoadGlobalU8:
|
||||
define(opcode, U32[1], U32[1], extract(false, 8));
|
||||
break;
|
||||
case IR::Opcode::LoadGlobalS8:
|
||||
define(opcode, U32[1], U32[1], extract(true, 8));
|
||||
break;
|
||||
case IR::Opcode::LoadGlobalU16:
|
||||
define(opcode, U32[1], U32[1], extract(false, 16));
|
||||
break;
|
||||
case IR::Opcode::LoadGlobalS16:
|
||||
define(opcode, U32[1], U32[1], extract(true, 16));
|
||||
break;
|
||||
case IR::Opcode::LoadGlobal32:
|
||||
define(opcode, U32[1], U32[1], load(U32[1], 1));
|
||||
break;
|
||||
case IR::Opcode::LoadGlobal64:
|
||||
define(opcode, U32[2], U32[1], load(U32[2], 2));
|
||||
break;
|
||||
case IR::Opcode::LoadGlobal128:
|
||||
define(opcode, U32[4], U32[1], load(U32[4], 4));
|
||||
break;
|
||||
case IR::Opcode::WriteGlobalU8:
|
||||
case IR::Opcode::WriteGlobalS8:
|
||||
define(opcode, void_id, U32[1], insert(8));
|
||||
break;
|
||||
case IR::Opcode::WriteGlobalU16:
|
||||
case IR::Opcode::WriteGlobalS16:
|
||||
define(opcode, void_id, U32[1], insert(16));
|
||||
break;
|
||||
case IR::Opcode::WriteGlobal32:
|
||||
define(opcode, void_id, U32[1], store(1));
|
||||
break;
|
||||
case IR::Opcode::WriteGlobal64:
|
||||
define(opcode, void_id, U32[2], store(2));
|
||||
break;
|
||||
case IR::Opcode::WriteGlobal128:
|
||||
define(opcode, void_id, U32[4], store(4));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicIAdd32:
|
||||
define(opcode, U32[1], U32[1], atomic(&Sirit::Module::OpAtomicIAdd));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicSMin32:
|
||||
define(opcode, U32[1], U32[1], atomic(&Sirit::Module::OpAtomicSMin));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicUMin32:
|
||||
define(opcode, U32[1], U32[1], atomic(&Sirit::Module::OpAtomicUMin));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicSMax32:
|
||||
define(opcode, U32[1], U32[1], atomic(&Sirit::Module::OpAtomicSMax));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicUMax32:
|
||||
define(opcode, U32[1], U32[1], atomic(&Sirit::Module::OpAtomicUMax));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicAnd32:
|
||||
define(opcode, U32[1], U32[1], atomic(&Sirit::Module::OpAtomicAnd));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicOr32:
|
||||
define(opcode, U32[1], U32[1], atomic(&Sirit::Module::OpAtomicOr));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicXor32:
|
||||
define(opcode, U32[1], U32[1], atomic(&Sirit::Module::OpAtomicXor));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicExchange32:
|
||||
define(opcode, U32[1], U32[1], atomic(&Sirit::Module::OpAtomicExchange));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicInc32:
|
||||
define(opcode, U32[1], U32[1], cas(U32[1], increment_cas_ssbo));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicDec32:
|
||||
define(opcode, U32[1], U32[1], cas(U32[1], decrement_cas_ssbo));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicAddF32:
|
||||
define(opcode, F32[1], F32[1], cas(F32[1], f32_add_cas));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicAddF16x2:
|
||||
define(opcode, U32[1], F16[2], packed(true, f16x2_add_cas));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicMinF16x2:
|
||||
define(opcode, U32[1], F16[2], packed(true, f16x2_min_cas));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicMaxF16x2:
|
||||
define(opcode, U32[1], F16[2], packed(true, f16x2_max_cas));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicAddF32x2:
|
||||
define(opcode, U32[1], F32[2], packed(false, f32x2_add_cas));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicMinF32x2:
|
||||
define(opcode, U32[1], F32[2], packed(false, f32x2_min_cas));
|
||||
break;
|
||||
case IR::Opcode::GlobalAtomicMaxF32x2:
|
||||
define(opcode, U32[1], F32[2], packed(false, f32x2_max_cas));
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
}
|
||||
}
|
||||
const auto define{
|
||||
[&](DefPtr ssbo_member, const StorageTypeDefinition& type_def, Id type, size_t size) {
|
||||
const Id element_type{type_def.element};
|
||||
const u32 shift{static_cast<u32>(std::countr_zero(size))};
|
||||
const Id load_func{define_load(ssbo_member, element_type, type, shift)};
|
||||
const Id write_func{define_write(ssbo_member, element_type, type, shift)};
|
||||
return std::make_pair(load_func, write_func);
|
||||
}};
|
||||
std::tie(load_global_func_u32, write_global_func_u32) =
|
||||
define(&StorageDefinitions::U32, storage_types.U32, U32[1], sizeof(u32));
|
||||
std::tie(load_global_func_u32x2, write_global_func_u32x2) =
|
||||
define(&StorageDefinitions::U32x2, storage_types.U32x2, U32[2], sizeof(u32[2]));
|
||||
std::tie(load_global_func_u32x4, write_global_func_u32x4) =
|
||||
define(&StorageDefinitions::U32x4, storage_types.U32x4, U32[4], sizeof(u32[4]));
|
||||
}
|
||||
|
||||
void EmitContext::DefineRescalingInput(const Info& info) {
|
||||
|
||||
@@ -7,7 +7,6 @@
|
||||
#pragma once
|
||||
|
||||
#include <array>
|
||||
#include <unordered_map>
|
||||
|
||||
#include <sirit/sirit.h>
|
||||
#include "common/container/unordered_set.h"
|
||||
@@ -174,9 +173,6 @@ public:
|
||||
[[nodiscard]] Id BitOffset8(const IR::Value& offset);
|
||||
[[nodiscard]] Id BitOffset16(const IR::Value& offset);
|
||||
|
||||
Id CallGlobalMemory(IR::Opcode opcode, Id result_type, Id address, Id value);
|
||||
void AtomicBitFieldInsert(Id pointer, Id value, Id offset, Id count);
|
||||
|
||||
Id Const(u32 value) {
|
||||
return Constant(U32[1], value);
|
||||
}
|
||||
@@ -339,7 +335,14 @@ public:
|
||||
Id f32x2_min_cas{};
|
||||
Id f32x2_max_cas{};
|
||||
|
||||
std::unordered_map<IR::Opcode, Id> global_memory_funcs;
|
||||
Id write_storage_cas_loop_func{};
|
||||
|
||||
Id load_global_func_u32{};
|
||||
Id load_global_func_u32x2{};
|
||||
Id load_global_func_u32x4{};
|
||||
Id write_global_func_u32{};
|
||||
Id write_global_func_u32x2{};
|
||||
Id write_global_func_u32x4{};
|
||||
|
||||
bool need_input_position_indirect{};
|
||||
Id input_position{};
|
||||
@@ -358,7 +361,6 @@ public:
|
||||
Id frag_depth{};
|
||||
|
||||
std::vector<Id> interfaces;
|
||||
std::vector<std::pair<u32, spv::Op>> opcode_rewrites;
|
||||
|
||||
Id load_const_func_u8{};
|
||||
Id load_const_func_u16{};
|
||||
@@ -390,7 +392,8 @@ private:
|
||||
void DefineTextures(const Info& info, u32& binding, u32& scaling_index);
|
||||
void DefineImages(const Info& info, u32& binding, u32& scaling_index);
|
||||
void DefineAttributeMemAccess(const Info& info);
|
||||
void DefineGlobalMemoryFunctions(const IR::Program& program);
|
||||
void DefineWriteStorageCasLoopFunction(const Info& info);
|
||||
void DefineGlobalMemoryFunctions(const Info& info);
|
||||
void DefineRescalingInput(const Info& info);
|
||||
void DefineRescalingInputPushConstant();
|
||||
void DefineRescalingInputUniformConstant();
|
||||
|
||||
+7
-2
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
@@ -93,7 +93,7 @@ IR::U64 AtomOffset(TranslatorVisitor& v, u64 insn) {
|
||||
} const mem{insn};
|
||||
|
||||
const IR::U64 address{[&]() -> IR::U64 {
|
||||
if (mem.e == 0 || mem.addr_reg == IR::Reg::RZ)
|
||||
if (mem.e == 0)
|
||||
return v.ir.UConvert(64, v.X(mem.addr_reg));
|
||||
return v.L(mem.addr_reg);
|
||||
}()};
|
||||
@@ -108,9 +108,14 @@ IR::U64 AtomOffset(TranslatorVisitor& v, u64 insn) {
|
||||
return v.ir.IAdd(address, v.ir.Imm64(addr_offset));
|
||||
}
|
||||
|
||||
// INC, DEC for U32/S32/U64 does nothing
|
||||
// ADD, INC, DEC for S64 does nothing
|
||||
// Only ADD does something for F32
|
||||
// Only ADD, MIN and MAX does something for F16x2
|
||||
bool AtomOpNotApplicable(AtomSize size, AtomOp op) {
|
||||
// TODO: SAFEADD
|
||||
switch (size) {
|
||||
case AtomSize::U32:
|
||||
case AtomSize::S32:
|
||||
case AtomSize::U64:
|
||||
return (op == AtomOp::INC || op == AtomOp::DEC);
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
@@ -36,12 +36,6 @@ void TranslatorVisitor::DEPBAR(u64) {
|
||||
// DEPBAR is a no-op
|
||||
}
|
||||
|
||||
void TranslatorVisitor::CCTL(u64) {}
|
||||
|
||||
void TranslatorVisitor::CCTLL(u64) {}
|
||||
|
||||
void TranslatorVisitor::CCTLT(u64) {}
|
||||
|
||||
void TranslatorVisitor::BAR(u64 insn) {
|
||||
enum class Mode {
|
||||
RedPopc,
|
||||
|
||||
@@ -1,6 +1,3 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
@@ -58,7 +55,7 @@ IR::U64 Address(TranslatorVisitor& v, u64 insn) {
|
||||
} const mem{insn};
|
||||
|
||||
const IR::U64 address{[&]() -> IR::U64 {
|
||||
if (mem.e == 0 || mem.addr_reg == IR::Reg::RZ) {
|
||||
if (mem.e == 0) {
|
||||
// LDG/STG without .E uses a 32-bit pointer, zero-extend it
|
||||
return v.ir.UConvert(64, v.X(mem.addr_reg));
|
||||
}
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
@@ -43,6 +43,18 @@ void TranslatorVisitor::CAL(u64) {
|
||||
// CAL is a no-op
|
||||
}
|
||||
|
||||
void TranslatorVisitor::CCTL(u64) {
|
||||
ThrowNotImplemented(Opcode::CCTL);
|
||||
}
|
||||
|
||||
void TranslatorVisitor::CCTLL(u64) {
|
||||
ThrowNotImplemented(Opcode::CCTLL);
|
||||
}
|
||||
|
||||
void TranslatorVisitor::CCTLT(u64) {
|
||||
ThrowNotImplemented(Opcode::CCTLT);
|
||||
}
|
||||
|
||||
void TranslatorVisitor::CONT(u64) {
|
||||
ThrowNotImplemented(Opcode::CONT);
|
||||
}
|
||||
|
||||
@@ -5,10 +5,7 @@
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <algorithm>
|
||||
#include <array>
|
||||
#include <bitset>
|
||||
#include <memory>
|
||||
#include <optional>
|
||||
#include <vector>
|
||||
#include <queue>
|
||||
|
||||
@@ -121,12 +118,9 @@ void AddNVNStorageBuffers(IR::Program& program) {
|
||||
continue;
|
||||
}
|
||||
const u32 offset{base + index * descriptor_size};
|
||||
const auto it{std::ranges::find_if(descs, [&](const StorageBufferDescriptor& desc) {
|
||||
return desc.cbuf_index == driver_cbuf && desc.cbuf_offset == offset;
|
||||
})};
|
||||
const auto it{std::ranges::find(descs, offset, &StorageBufferDescriptor::cbuf_offset)};
|
||||
if (it != descs.end()) {
|
||||
it->is_written |= program.info.stores_global_memory;
|
||||
it->is_global_fallback = true;
|
||||
continue;
|
||||
}
|
||||
descs.push_back({
|
||||
@@ -134,7 +128,6 @@ void AddNVNStorageBuffers(IR::Program& program) {
|
||||
.cbuf_offset = offset,
|
||||
.count = 1,
|
||||
.is_written = program.info.stores_global_memory,
|
||||
.is_global_fallback = true,
|
||||
});
|
||||
}
|
||||
}
|
||||
@@ -179,17 +172,7 @@ std::map<IR::Attribute, IR::Attribute> GenerateLegacyToGenericMappings(
|
||||
void EmitGeometryPassthrough(IR::IREmitter& ir, const IR::Program& program,
|
||||
const Shader::VaryingState& passthrough_mask,
|
||||
bool passthrough_position,
|
||||
std::optional<IR::Attribute> passthrough_layer_attr,
|
||||
std::optional<IR::Reg> viewport_mask_reg) {
|
||||
constexpr std::array CULLED_POSITION{2.0f, 2.0f, 2.0f, 1.0f};
|
||||
IR::U1 culled{ir.Imm1(false)};
|
||||
IR::U32 viewport{ir.Imm32(0)};
|
||||
if (viewport_mask_reg) {
|
||||
const IR::U32 mask{ir.GetReg(*viewport_mask_reg)};
|
||||
const IR::U32 lowest_bit{ir.BitwiseAnd(mask, IR::U32{ir.INeg(mask)})};
|
||||
culled = ir.IEqual(mask, ir.Imm32(0));
|
||||
viewport = IR::U32{ir.Select(culled, ir.Imm32(0), ir.FindUMsb(lowest_bit))};
|
||||
}
|
||||
std::optional<IR::Attribute> passthrough_layer_attr) {
|
||||
for (u32 i = 0; i < program.output_vertices; i++) {
|
||||
// Assign generics from input
|
||||
for (u32 j = 0; j < 32; j++) {
|
||||
@@ -207,19 +190,10 @@ void EmitGeometryPassthrough(IR::IREmitter& ir, const IR::Program& program,
|
||||
if (passthrough_position) {
|
||||
// Assign position from input
|
||||
const IR::Attribute attr = IR::Attribute::PositionX;
|
||||
for (u32 component = 0; component < 4; ++component) {
|
||||
IR::F32 value{ir.GetAttribute(attr + component, ir.Imm32(i))};
|
||||
if (viewport_mask_reg) {
|
||||
value = IR::F32{
|
||||
ir.Select(culled, ir.Imm32(CULLED_POSITION[component]), value)};
|
||||
}
|
||||
ir.SetAttribute(attr + component, value, ir.Imm32(0));
|
||||
}
|
||||
}
|
||||
|
||||
if (viewport_mask_reg) {
|
||||
ir.SetAttribute(IR::Attribute::ViewportIndex, ir.BitCast<IR::F32>(viewport),
|
||||
ir.Imm32(0));
|
||||
ir.SetAttribute(attr + 0, ir.GetAttribute(attr + 0, ir.Imm32(i)), ir.Imm32(0));
|
||||
ir.SetAttribute(attr + 1, ir.GetAttribute(attr + 1, ir.Imm32(i)), ir.Imm32(0));
|
||||
ir.SetAttribute(attr + 2, ir.GetAttribute(attr + 2, ir.Imm32(i)), ir.Imm32(0));
|
||||
ir.SetAttribute(attr + 3, ir.GetAttribute(attr + 3, ir.Imm32(i)), ir.Imm32(0));
|
||||
}
|
||||
|
||||
if (passthrough_layer_attr) {
|
||||
@@ -245,91 +219,19 @@ u32 GetOutputTopologyVertices(OutputTopology output_topology) {
|
||||
}
|
||||
}
|
||||
|
||||
std::optional<IR::Reg> FindFreeRegister(const IR::Program& program) {
|
||||
std::bitset<IR::NUM_REGS> used;
|
||||
for (IR::Block* const block : program.blocks) {
|
||||
for (const IR::Inst& inst : block->Instructions()) {
|
||||
const IR::Opcode opcode{inst.GetOpcode()};
|
||||
if (opcode == IR::Opcode::GetRegister || opcode == IR::Opcode::SetRegister) {
|
||||
used.set(IR::RegIndex(inst.Arg(0).Reg()));
|
||||
}
|
||||
}
|
||||
}
|
||||
for (size_t index = IR::NUM_USER_REGS; index-- > 0;) {
|
||||
if (!used.test(index)) {
|
||||
return static_cast<IR::Reg>(index);
|
||||
}
|
||||
}
|
||||
return std::nullopt;
|
||||
}
|
||||
|
||||
std::optional<IR::Reg> LowerViewportMask(const IR::Program& program) {
|
||||
const std::optional<IR::Reg> reg{FindFreeRegister(program)};
|
||||
if (!reg || program.blocks.empty()) {
|
||||
return std::nullopt;
|
||||
}
|
||||
bool stores_mask{};
|
||||
for (IR::Block* const block : program.blocks) {
|
||||
for (IR::Inst& inst : block->Instructions()) {
|
||||
if (inst.GetOpcode() != IR::Opcode::SetAttribute ||
|
||||
inst.Arg(0).Attribute() != IR::Attribute::ViewportMask) {
|
||||
continue;
|
||||
}
|
||||
IR::IREmitter ir{*block, IR::Block::InstructionList::s_iterator_to(inst)};
|
||||
ir.SetReg(*reg, ir.BitCast<IR::U32>(IR::F32{inst.Arg(1)}));
|
||||
inst.Invalidate();
|
||||
stores_mask = true;
|
||||
}
|
||||
}
|
||||
if (!stores_mask) {
|
||||
return std::nullopt;
|
||||
}
|
||||
IR::Block& entry{*program.blocks.front()};
|
||||
IR::IREmitter ir{entry, entry.begin()};
|
||||
ir.SetReg(*reg, ir.Imm32(1));
|
||||
return reg;
|
||||
}
|
||||
|
||||
void LowerGeometryPassthrough(const IR::Program& program, const HostTranslateInfo& host_info) {
|
||||
std::optional<IR::Reg> viewport_mask_reg;
|
||||
if (!host_info.support_viewport_mask) {
|
||||
viewport_mask_reg = LowerViewportMask(program);
|
||||
}
|
||||
for (IR::Block* const block : program.blocks) {
|
||||
for (IR::Inst& inst : block->Instructions()) {
|
||||
if (inst.GetOpcode() == IR::Opcode::Epilogue) {
|
||||
IR::IREmitter ir{*block, IR::Block::InstructionList::s_iterator_to(inst)};
|
||||
EmitGeometryPassthrough(
|
||||
ir, program, program.info.passthrough,
|
||||
program.info.passthrough.AnyComponent(IR::Attribute::PositionX), {},
|
||||
viewport_mask_reg);
|
||||
program.info.passthrough.AnyComponent(IR::Attribute::PositionX), {});
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void TightenOutputVertices(IR::Program& program) {
|
||||
if (program.stage != Stage::Geometry || program.is_geometry_passthrough) {
|
||||
return;
|
||||
}
|
||||
const bool has_loops = std::ranges::any_of(program.syntax_list, [](const auto& node) {
|
||||
return node.type == IR::AbstractSyntaxNode::Type::Loop;
|
||||
});
|
||||
if (has_loops) {
|
||||
return;
|
||||
}
|
||||
u32 num_emits = 0;
|
||||
for (IR::Block* const block : program.blocks) {
|
||||
for (const IR::Inst& inst : block->Instructions()) {
|
||||
if (inst.GetOpcode() == IR::Opcode::EmitVertex) {
|
||||
++num_emits;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (num_emits != 0) {
|
||||
program.output_vertices = (std::min)(program.output_vertices, num_emits);
|
||||
}
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
IR::Program TranslateProgram(ObjectPool<IR::Inst>& inst_pool, ObjectPool<IR::Block>& block_pool,
|
||||
@@ -397,14 +299,12 @@ IR::Program TranslateProgram(ObjectPool<IR::Inst>& inst_pool, ObjectPool<IR::Blo
|
||||
Optimization::PositionPass(env, program);
|
||||
|
||||
Optimization::GlobalMemoryToStorageBufferPass(program, normalized_host_info);
|
||||
Optimization::GeometryCompactionPass(program, normalized_host_info);
|
||||
Optimization::TexturePass(env, program, normalized_host_info);
|
||||
|
||||
if (Settings::values.resolution_info.active || Settings::values.rescale_hack.GetValue()) {
|
||||
Optimization::RescalingPass(program);
|
||||
}
|
||||
Optimization::DeadCodeEliminationPass(program);
|
||||
TightenOutputVertices(program);
|
||||
if (Settings::values.renderer_debug) {
|
||||
Optimization::VerificationPass(program);
|
||||
}
|
||||
@@ -533,7 +433,7 @@ IR::Program GenerateGeometryPassthrough(ObjectPool<IR::Inst>& inst_pool,
|
||||
|
||||
IR::IREmitter ir{*current_block};
|
||||
EmitGeometryPassthrough(ir, program, program.info.stores, true,
|
||||
source_program.info.emulated_layer, std::nullopt);
|
||||
source_program.info.emulated_layer);
|
||||
|
||||
IR::Block* return_block{block_pool.Create(inst_pool)};
|
||||
IR::IREmitter{*return_block}.Epilogue();
|
||||
|
||||
@@ -34,12 +34,10 @@ struct HostTranslateInfo {
|
||||
bool needs_demote_reorder{}; ///< True when the device needs DemoteToHelperInvocation reordered
|
||||
bool support_snorm_render_buffer{}; ///< True when the device supports SNORM render buffers
|
||||
bool support_viewport_index_layer{}; ///< True when the device supports gl_Layer in VS
|
||||
bool support_viewport_mask{};
|
||||
bool support_geometry_shader_passthrough{}; ///< True when the device supports geometry
|
||||
///< passthrough shaders
|
||||
bool support_conditional_barrier{}; ///< True when the device supports barriers in conditional
|
||||
///< control flow
|
||||
bool single_lane_geometry_subgroups{};
|
||||
|
||||
void ApplyDescriptorLimitPolicy() noexcept {
|
||||
if (min_ssbo_alignment == 0) {
|
||||
|
||||
@@ -424,8 +424,8 @@ void VisitUsages(Info& info, IR::Inst& inst) {
|
||||
case IR::Opcode::LoadGlobal128:
|
||||
info.uses_int64 = true;
|
||||
info.uses_global_memory = true;
|
||||
info.used_constant_buffer_types |= IR::Type::U32;
|
||||
info.used_storage_buffer_types |= IR::Type::U32;
|
||||
info.used_constant_buffer_types |= IR::Type::U32 | IR::Type::U32x2;
|
||||
info.used_storage_buffer_types |= IR::Type::U32 | IR::Type::U32x2 | IR::Type::U32x4;
|
||||
break;
|
||||
case IR::Opcode::LoadLocal:
|
||||
case IR::Opcode::WriteLocal:
|
||||
@@ -632,8 +632,6 @@ void VisitUsages(Info& info, IR::Inst& inst) {
|
||||
case IR::Opcode::StorageAtomicExchange32:
|
||||
info.used_storage_buffer_types |= IR::Type::U32;
|
||||
break;
|
||||
case IR::Opcode::LoadGlobal64:
|
||||
case IR::Opcode::WriteGlobal64:
|
||||
case IR::Opcode::LoadStorage64:
|
||||
case IR::Opcode::WriteStorage64:
|
||||
case IR::Opcode::StorageAtomicIAdd32x2:
|
||||
@@ -647,8 +645,6 @@ void VisitUsages(Info& info, IR::Inst& inst) {
|
||||
case IR::Opcode::StorageAtomicExchange32x2:
|
||||
info.used_storage_buffer_types |= IR::Type::U32x2;
|
||||
break;
|
||||
case IR::Opcode::LoadGlobal128:
|
||||
case IR::Opcode::WriteGlobal128:
|
||||
case IR::Opcode::LoadStorage128:
|
||||
case IR::Opcode::WriteStorage128:
|
||||
info.used_storage_buffer_types |= IR::Type::U32x4;
|
||||
|
||||
@@ -1,4 +1,4 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-FileCopyrightText: Copyright 2025 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
@@ -414,14 +414,6 @@ void FoldSelect(IR::Inst& inst) {
|
||||
}
|
||||
}
|
||||
|
||||
void FoldAtomicWrap(IR::Inst& inst, IR::Opcode opcode, u32 addend) {
|
||||
const IR::Value limit{inst.Arg(1)};
|
||||
if (limit.IsImmediate() && limit.U32() == 0xFFFFFFFFU) {
|
||||
inst.ReplaceOpcode(opcode);
|
||||
inst.SetArg(1, IR::Value{addend});
|
||||
}
|
||||
}
|
||||
|
||||
void FoldFPAdd32(IR::Inst& inst) {
|
||||
if (FoldWhenAllImmediates(inst, [](f32 a, f32 b) { return a + b; })) {
|
||||
return;
|
||||
@@ -659,29 +651,6 @@ IR::Value GetThroughCast(IR::Value value, IR::Opcode expected_cast) {
|
||||
return value;
|
||||
}
|
||||
|
||||
u32 QuadButterflyMask(const IR::Inst& inst) {
|
||||
if (inst.GetOpcode() == IR::Opcode::QuadSwap) {
|
||||
const IR::Value direction{inst.Arg(1)};
|
||||
if (!direction.IsImmediate()) {
|
||||
return 0;
|
||||
}
|
||||
return direction.U32() + 1;
|
||||
}
|
||||
if (inst.GetOpcode() != IR::Opcode::ShuffleButterfly) {
|
||||
return 0;
|
||||
}
|
||||
const IR::Value index{inst.Arg(1)};
|
||||
const IR::Value clamp{inst.Arg(2)};
|
||||
const IR::Value segmentation_mask{inst.Arg(3)};
|
||||
if (!index.IsImmediate() || !clamp.IsImmediate() || !segmentation_mask.IsImmediate()) {
|
||||
return 0;
|
||||
}
|
||||
if (clamp.U32() != 3 || segmentation_mask.U32() != 28) {
|
||||
return 0;
|
||||
}
|
||||
return index.U32();
|
||||
}
|
||||
|
||||
void FoldFSwizzleAdd(IR::Block& block, IR::Inst& inst) {
|
||||
const IR::Value swizzle{inst.Arg(2)};
|
||||
if (!swizzle.IsImmediate()) {
|
||||
@@ -697,8 +666,7 @@ void FoldFSwizzleAdd(IR::Block& block, IR::Inst& inst) {
|
||||
return;
|
||||
}
|
||||
IR::Inst* const inst2{value_1.InstRecursive()};
|
||||
const u32 lane_mask{QuadButterflyMask(*inst2)};
|
||||
if (lane_mask == 0) {
|
||||
if (inst2->GetOpcode() != IR::Opcode::ShuffleButterfly) {
|
||||
return;
|
||||
}
|
||||
const IR::Value value_3{GetThroughCast(inst2->Arg(0).Resolve(), IR::Opcode::BitCastU32F32)};
|
||||
@@ -710,15 +678,24 @@ void FoldFSwizzleAdd(IR::Block& block, IR::Inst& inst) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
const IR::Value index{inst2->Arg(1)};
|
||||
const IR::Value clamp{inst2->Arg(2)};
|
||||
const IR::Value segmentation_mask{inst2->Arg(3)};
|
||||
if (!index.IsImmediate() || !clamp.IsImmediate() || !segmentation_mask.IsImmediate()) {
|
||||
return;
|
||||
}
|
||||
if (clamp.U32() != 3 || segmentation_mask.U32() != 28) {
|
||||
return;
|
||||
}
|
||||
if (swizzle_value == 0x99) {
|
||||
// DPdxFine
|
||||
if (lane_mask == 1) {
|
||||
if (index.U32() == 1) {
|
||||
IR::IREmitter ir{block, IR::Block::InstructionList::s_iterator_to(inst)};
|
||||
inst.ReplaceUsesWith(ir.DPdxFine(IR::F32{inst.Arg(1)}));
|
||||
}
|
||||
} else if (swizzle_value == 0xA5) {
|
||||
// DPdyFine
|
||||
if (lane_mask == 2) {
|
||||
if (index.U32() == 2) {
|
||||
IR::IREmitter ir{block, IR::Block::InstructionList::s_iterator_to(inst)};
|
||||
inst.ReplaceUsesWith(ir.DPdyFine(IR::F32{inst.Arg(1)}));
|
||||
}
|
||||
@@ -732,13 +709,6 @@ bool FindGradient3DDerivatives(std::array<IR::Value, 3>& results, IR::Value coor
|
||||
const auto check_through_shuffle = [](IR::Value input, IR::Value& result) {
|
||||
const IR::Value value_1{GetThroughCast(input.Resolve(), IR::Opcode::BitCastF32U32)};
|
||||
IR::Inst* const inst2{value_1.InstRecursive()};
|
||||
if (inst2->GetOpcode() == IR::Opcode::QuadBroadcast) {
|
||||
if (!inst2->Arg(1).Resolve().IsImmediate()) {
|
||||
return false;
|
||||
}
|
||||
result = GetThroughCast(inst2->Arg(0).Resolve(), IR::Opcode::BitCastU32F32);
|
||||
return true;
|
||||
}
|
||||
if (inst2->GetOpcode() != IR::Opcode::ShuffleIndex) {
|
||||
return false;
|
||||
}
|
||||
@@ -1112,14 +1082,6 @@ void ConstantPropagation(Environment& env, IR::Block& block, IR::Inst& inst) {
|
||||
IR::Opcode::CompositeInsertF16x4);
|
||||
case IR::Opcode::FSwizzleAdd:
|
||||
return FoldFSwizzleAdd(block, inst);
|
||||
case IR::Opcode::GlobalAtomicInc32:
|
||||
return FoldAtomicWrap(inst, IR::Opcode::GlobalAtomicIAdd32, 1U);
|
||||
case IR::Opcode::GlobalAtomicDec32:
|
||||
return FoldAtomicWrap(inst, IR::Opcode::GlobalAtomicIAdd32, 0xFFFFFFFFU);
|
||||
case IR::Opcode::SharedAtomicInc32:
|
||||
return FoldAtomicWrap(inst, IR::Opcode::SharedAtomicIAdd32, 1U);
|
||||
case IR::Opcode::SharedAtomicDec32:
|
||||
return FoldAtomicWrap(inst, IR::Opcode::SharedAtomicIAdd32, 0xFFFFFFFFU);
|
||||
case IR::Opcode::GetCbufF32:
|
||||
case IR::Opcode::GetCbufU32:
|
||||
if (env.HasHLEMacroState()) {
|
||||
|
||||
@@ -1,129 +0,0 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
#include <boost/container/small_vector.hpp>
|
||||
|
||||
#include "shader_recompiler/frontend/ir/ir_emitter.h"
|
||||
#include "shader_recompiler/host_translate_info.h"
|
||||
#include "shader_recompiler/ir_opt/passes.h"
|
||||
|
||||
namespace Shader::Optimization {
|
||||
namespace {
|
||||
constexpr int MAX_BALLOT_DEPTH = 8;
|
||||
|
||||
struct SlotAtomic {
|
||||
IR::Block* block;
|
||||
IR::Inst* atomic;
|
||||
IR::Inst* ballot;
|
||||
};
|
||||
|
||||
IR::Inst* FindBallot(const IR::Value& value, int depth) {
|
||||
if (depth > MAX_BALLOT_DEPTH || value.IsImmediate()) {
|
||||
return nullptr;
|
||||
}
|
||||
IR::Inst* const inst{value.InstRecursive()};
|
||||
if (inst->GetOpcode() == IR::Opcode::SubgroupBallot) {
|
||||
return inst;
|
||||
}
|
||||
if (inst->GetOpcode() == IR::Opcode::Phi) {
|
||||
return nullptr;
|
||||
}
|
||||
for (size_t index = 0; index < inst->NumArgs(); ++index) {
|
||||
if (IR::Inst* const ballot{FindBallot(inst->Arg(index), depth + 1)}) {
|
||||
return ballot;
|
||||
}
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
bool IsShuffle(IR::Opcode opcode) {
|
||||
switch (opcode) {
|
||||
case IR::Opcode::ShuffleIndex:
|
||||
case IR::Opcode::ShuffleUp:
|
||||
case IR::Opcode::ShuffleDown:
|
||||
case IR::Opcode::ShuffleButterfly:
|
||||
return true;
|
||||
default:
|
||||
return false;
|
||||
}
|
||||
}
|
||||
|
||||
IR::U32 PrimitiveSlot(IR::IREmitter& ir, u32 invocations, const IR::U32& amount) {
|
||||
IR::U32 key{ir.GetAttributeU32(IR::Attribute::PrimitiveId)};
|
||||
if (invocations > 1) {
|
||||
key = ir.IAdd(ir.IMul(key, ir.Imm32(invocations)), ir.InvocationId());
|
||||
}
|
||||
return ir.IMul(key, amount);
|
||||
}
|
||||
|
||||
bool MergesAtomic(const IR::Inst& phi, const IR::Block& block, const IR::Inst& atomic) {
|
||||
for (size_t index = 0; index < phi.NumArgs(); ++index) {
|
||||
const IR::Value arg{phi.Arg(index)};
|
||||
if (phi.PhiBlock(index) == &block && !arg.IsImmediate() &&
|
||||
arg.InstRecursive() == &atomic) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
return false;
|
||||
}
|
||||
|
||||
void RewriteMergedSlots(IR::Block& block, IR::Inst& atomic, u32 invocations) {
|
||||
for (IR::Block* const successor : block.ImmSuccessors()) {
|
||||
for (IR::Inst& phi : successor->Instructions()) {
|
||||
if (phi.GetOpcode() != IR::Opcode::Phi || !MergesAtomic(phi, block, atomic)) {
|
||||
continue;
|
||||
}
|
||||
for (size_t index = 0; index < phi.NumArgs(); ++index) {
|
||||
const IR::Value arg{phi.Arg(index)};
|
||||
if (arg.IsImmediate() || !IsShuffle(arg.InstRecursive()->GetOpcode())) {
|
||||
continue;
|
||||
}
|
||||
IR::IREmitter ir{*phi.PhiBlock(index)};
|
||||
const IR::U32 amount{arg.InstRecursive()->Arg(0)};
|
||||
phi.SetArg(index, PrimitiveSlot(ir, invocations, amount));
|
||||
}
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void RewriteAtomic(IR::Block& block, IR::Inst& atomic, u32 invocations) {
|
||||
const auto insert_point{IR::Block::InstructionList::s_iterator_to(atomic)};
|
||||
IR::IREmitter ir{block, insert_point};
|
||||
const IR::U32 amount{atomic.Arg(2)};
|
||||
const IR::U32 slot{PrimitiveSlot(ir, invocations, amount)};
|
||||
block.PrependNewInst(insert_point, IR::Opcode::StorageAtomicUMax32,
|
||||
{atomic.Arg(0), atomic.Arg(1), ir.IAdd(slot, amount)});
|
||||
atomic.ReplaceUsesWith(slot);
|
||||
}
|
||||
|
||||
void NeutralizePredicate(IR::Inst& ballot) {
|
||||
const IR::Value pred{ballot.Arg(0)};
|
||||
if (!pred.IsImmediate()) {
|
||||
pred.InstRecursive()->ReplaceUsesWith(IR::Value{true});
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void GeometryCompactionPass(IR::Program& program, const HostTranslateInfo& host_info) {
|
||||
if (program.stage != Stage::Geometry || !host_info.single_lane_geometry_subgroups) {
|
||||
return;
|
||||
}
|
||||
boost::container::small_vector<SlotAtomic, 4> slot_atomics;
|
||||
for (IR::Block* const block : program.post_order_blocks) {
|
||||
for (IR::Inst& inst : block->Instructions()) {
|
||||
if (inst.GetOpcode() != IR::Opcode::StorageAtomicIAdd32) {
|
||||
continue;
|
||||
}
|
||||
if (IR::Inst* const ballot{FindBallot(inst.Arg(2), 0)}) {
|
||||
slot_atomics.push_back({block, &inst, ballot});
|
||||
}
|
||||
}
|
||||
}
|
||||
for (const SlotAtomic& slot_atomic : slot_atomics) {
|
||||
RewriteMergedSlots(*slot_atomic.block, *slot_atomic.atomic, program.invocations);
|
||||
RewriteAtomic(*slot_atomic.block, *slot_atomic.atomic, program.invocations);
|
||||
NeutralizePredicate(*slot_atomic.ballot);
|
||||
}
|
||||
}
|
||||
|
||||
}
|
||||
@@ -4,15 +4,14 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <map>
|
||||
#include <optional>
|
||||
#include <unordered_set>
|
||||
|
||||
#include <boost/container/flat_set.hpp>
|
||||
#include <boost/container/small_vector.hpp>
|
||||
|
||||
#include "common/alignment.h"
|
||||
#include "shader_recompiler/frontend/ir/basic_block.h"
|
||||
#include "shader_recompiler/frontend/ir/breadth_first_search.h"
|
||||
#include "shader_recompiler/frontend/ir/ir_emitter.h"
|
||||
#include "shader_recompiler/frontend/ir/value.h"
|
||||
#include "shader_recompiler/host_translate_info.h"
|
||||
@@ -50,7 +49,6 @@ using StorageBufferSet =
|
||||
using StorageInstVector = small_vector<StorageInst, 24>;
|
||||
using StorageWritesSet =
|
||||
flat_set<StorageBufferAddr, std::less<StorageBufferAddr>, small_vector<StorageBufferAddr, 16>>;
|
||||
using LocalStores = std::multimap<u32, IR::Value>;
|
||||
|
||||
struct StorageInfo {
|
||||
StorageBufferSet set;
|
||||
@@ -335,7 +333,7 @@ std::optional<LowAddrInfo> TrackLowAddress(IR::Inst* inst) {
|
||||
}
|
||||
|
||||
/// Tries to track the storage buffer address used by a global memory instruction
|
||||
StorageBufferSet Track(const IR::Value& value, const Bias* bias, const LocalStores& local_stores) {
|
||||
std::optional<StorageBufferAddr> Track(const IR::Value& value, const Bias* bias) {
|
||||
const auto pred{[bias](const IR::Inst* inst) -> std::optional<StorageBufferAddr> {
|
||||
if (inst->GetOpcode() != IR::Opcode::GetCbufU32 &&
|
||||
inst->GetOpcode() != IR::Opcode::GetCbufU32x2) {
|
||||
@@ -368,83 +366,11 @@ StorageBufferSet Track(const IR::Value& value, const Bias* bias, const LocalStor
|
||||
}
|
||||
return storage_buffer;
|
||||
}};
|
||||
StorageBufferSet result;
|
||||
std::unordered_set<const IR::Inst*> visited;
|
||||
small_vector<const IR::Inst*, 32> pending;
|
||||
const auto push{[&](const IR::Value& arg) {
|
||||
if (!arg.IsImmediate() && visited.insert(arg.InstRecursive()).second) {
|
||||
pending.push_back(arg.InstRecursive());
|
||||
}
|
||||
}};
|
||||
push(value);
|
||||
while (!pending.empty()) {
|
||||
const IR::Inst* const inst{pending.back()};
|
||||
pending.pop_back();
|
||||
if (const std::optional<StorageBufferAddr> storage_buffer{pred(inst)}) {
|
||||
result.insert(*storage_buffer);
|
||||
continue;
|
||||
}
|
||||
switch (inst->GetOpcode()) {
|
||||
case IR::Opcode::LoadLocal:
|
||||
if (inst->Arg(0).IsImmediate()) {
|
||||
const auto [begin, end]{local_stores.equal_range(inst->Arg(0).U32())};
|
||||
for (auto it = begin; it != end; ++it) {
|
||||
push(it->second);
|
||||
}
|
||||
}
|
||||
continue;
|
||||
case IR::Opcode::SelectU32:
|
||||
case IR::Opcode::SelectU64:
|
||||
push(inst->Arg(1));
|
||||
push(inst->Arg(2));
|
||||
continue;
|
||||
case IR::Opcode::GetCbufU8:
|
||||
case IR::Opcode::GetCbufS8:
|
||||
case IR::Opcode::GetCbufU16:
|
||||
case IR::Opcode::GetCbufS16:
|
||||
case IR::Opcode::GetCbufU32:
|
||||
case IR::Opcode::GetCbufF32:
|
||||
case IR::Opcode::GetCbufU32x2:
|
||||
case IR::Opcode::LoadSharedU8:
|
||||
case IR::Opcode::LoadSharedS8:
|
||||
case IR::Opcode::LoadSharedU16:
|
||||
case IR::Opcode::LoadSharedS16:
|
||||
case IR::Opcode::LoadSharedU32:
|
||||
case IR::Opcode::LoadSharedU64:
|
||||
case IR::Opcode::LoadSharedU128:
|
||||
continue;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
if (IsGlobalMemory(*inst) || inst->MayHaveSideEffects()) {
|
||||
continue;
|
||||
}
|
||||
for (size_t arg = 0; arg < inst->NumArgs(); ++arg) {
|
||||
push(inst->Arg(arg));
|
||||
}
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
LocalStores GatherLocalStores(const IR::Program& program) {
|
||||
LocalStores stores;
|
||||
for (IR::Block* const block : program.post_order_blocks) {
|
||||
for (const IR::Inst& inst : block->Instructions()) {
|
||||
if (inst.GetOpcode() != IR::Opcode::WriteLocal) {
|
||||
continue;
|
||||
}
|
||||
if (!inst.Arg(0).IsImmediate()) {
|
||||
return {};
|
||||
}
|
||||
stores.emplace(inst.Arg(0).U32(), inst.Arg(1));
|
||||
}
|
||||
}
|
||||
return stores;
|
||||
return BreadthFirstSearch(value, pred);
|
||||
}
|
||||
|
||||
/// Collects the storage buffer used by a global memory instruction and the instruction itself
|
||||
void CollectStorageBuffers(IR::Block& block, IR::Inst& inst, StorageInfo& info,
|
||||
const LocalStores& local_stores) {
|
||||
void CollectStorageBuffers(IR::Block& block, IR::Inst& inst, StorageInfo& info) {
|
||||
// NVN puts storage buffers in a specific range, we have to bias towards these addresses to
|
||||
// avoid getting false positives
|
||||
static constexpr Bias nvn_bias{
|
||||
@@ -461,24 +387,25 @@ void CollectStorageBuffers(IR::Block& block, IR::Inst& inst, StorageInfo& info,
|
||||
}
|
||||
// First try to find storage buffers in the NVN address
|
||||
const IR::U32 low_addr{low_addr_info->value};
|
||||
StorageBufferSet candidates{Track(low_addr, &nvn_bias, local_stores)};
|
||||
if (candidates.empty()) {
|
||||
std::optional<StorageBufferAddr> storage_buffer{Track(low_addr, &nvn_bias)};
|
||||
if (!storage_buffer) {
|
||||
// If it fails, track without a bias
|
||||
candidates = Track(low_addr, nullptr, local_stores);
|
||||
storage_buffer = Track(low_addr, nullptr);
|
||||
if (!storage_buffer) {
|
||||
// If that also fails, use NVN fallbacks
|
||||
LOG_WARNING(Shader, "Storage buffer failed to track, using global memory fallbacks");
|
||||
return;
|
||||
}
|
||||
LOG_WARNING(Shader, "Storage buffer tracked without bias, index {} offset {}",
|
||||
storage_buffer->index, storage_buffer->offset);
|
||||
}
|
||||
if (candidates.size() != 1) {
|
||||
// If that also fails, use NVN fallbacks
|
||||
LOG_WARNING(Shader, "Storage buffer failed to track, using global memory fallbacks");
|
||||
return;
|
||||
}
|
||||
const StorageBufferAddr storage_buffer{*candidates.begin()};
|
||||
// Collect storage buffer and the instruction
|
||||
if (IsGlobalMemoryWrite(inst)) {
|
||||
info.writes.insert(storage_buffer);
|
||||
info.writes.insert(*storage_buffer);
|
||||
}
|
||||
info.set.insert(storage_buffer);
|
||||
info.set.insert(*storage_buffer);
|
||||
info.to_replace.push_back(StorageInst{
|
||||
.storage_buffer{storage_buffer},
|
||||
.storage_buffer{*storage_buffer},
|
||||
.inst = &inst,
|
||||
.block = &block,
|
||||
});
|
||||
@@ -598,13 +525,12 @@ void Replace(IR::Block& block, IR::Inst& inst, const IR::U32& storage_index,
|
||||
|
||||
void GlobalMemoryToStorageBufferPass(IR::Program& program, const HostTranslateInfo& host_info) {
|
||||
StorageInfo info;
|
||||
const LocalStores local_stores{GatherLocalStores(program)};
|
||||
for (IR::Block* const block : program.post_order_blocks) {
|
||||
for (IR::Inst& inst : block->Instructions()) {
|
||||
if (!IsGlobalMemory(inst)) {
|
||||
continue;
|
||||
}
|
||||
CollectStorageBuffers(*block, inst, info, local_stores);
|
||||
CollectStorageBuffers(*block, inst, info);
|
||||
}
|
||||
}
|
||||
for (const StorageBufferAddr& storage_buffer : info.set) {
|
||||
@@ -636,7 +562,6 @@ void JoinStorageInfo(Info& base, Info& source) {
|
||||
})};
|
||||
if (it != descriptors.end()) {
|
||||
it->is_written |= desc.is_written;
|
||||
it->is_global_fallback |= desc.is_global_fallback;
|
||||
continue;
|
||||
}
|
||||
descriptors.push_back(desc);
|
||||
|
||||
@@ -1,6 +1,3 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2026 Eden Emulator Project
|
||||
// SPDX-License-Identifier: GPL-3.0-or-later
|
||||
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
@@ -19,7 +16,6 @@ void CollectShaderInfoPass(Environment& env, IR::Program& program);
|
||||
void ConditionalBarrierPass(IR::Program& program);
|
||||
void ConstantPropagationPass(Environment& env, IR::Program& program);
|
||||
void DeadCodeEliminationPass(IR::Program& program);
|
||||
void GeometryCompactionPass(IR::Program& program, const HostTranslateInfo& host_info);
|
||||
void GlobalMemoryToStorageBufferPass(IR::Program& program, const HostTranslateInfo& host_info);
|
||||
void IdentityRemovalPass(IR::Program& program);
|
||||
void LowerFp64ToFp32(IR::Program& program);
|
||||
|
||||
@@ -33,8 +33,7 @@ using TextureInstVector = boost::container::small_vector<TextureInst, 24>;
|
||||
|
||||
constexpr u32 DESCRIPTOR_SIZE = 8;
|
||||
constexpr u32 DESCRIPTOR_SIZE_SHIFT = u32(std::countr_zero(DESCRIPTOR_SIZE));
|
||||
constexpr u32 DESCRIPTOR_MAX_COUNT = 128;
|
||||
constexpr u32 DESCRIPTOR_CBUF_BYTES = 16 * 1024;
|
||||
constexpr u32 DESCRIPTOR_MAX_COUNT = 1024;
|
||||
|
||||
u32 DynamicDescriptorSizeShift(const IR::U32& dynamic_offset) {
|
||||
const IR::Inst* const inst = dynamic_offset.InstRecursive();
|
||||
@@ -49,10 +48,11 @@ u32 DynamicDescriptorSizeShift(const IR::U32& dynamic_offset) {
|
||||
|
||||
u32 DynamicDescriptorCount(u32 base_offset, u32 size_shift, u32 max_descriptors) {
|
||||
auto const descriptor_limit = (std::max)(1U, max_descriptors);
|
||||
if (size_shift >= 31 || base_offset >= DESCRIPTOR_CBUF_BYTES)
|
||||
auto const max_cbuf_bytes = 16 * descriptor_limit;
|
||||
if (size_shift >= 31 || base_offset >= max_cbuf_bytes)
|
||||
return 1;
|
||||
auto const stride = 1U << size_shift;
|
||||
auto const available = DESCRIPTOR_CBUF_BYTES - base_offset;
|
||||
auto const available = max_cbuf_bytes - base_offset;
|
||||
if (available < DESCRIPTOR_SIZE)
|
||||
return 1;
|
||||
auto const available_count = 1U + (available - DESCRIPTOR_SIZE) / stride;
|
||||
|
||||
@@ -40,7 +40,6 @@ struct Profile {
|
||||
bool support_shader_quad_control{};
|
||||
bool support_quad_shuffles{};
|
||||
bool support_vote{};
|
||||
bool support_shuffle_relative{};
|
||||
u32 supported_subgroup_stages{0x7F};
|
||||
bool support_viewport_index_layer_non_geometry{};
|
||||
bool support_viewport_mask{};
|
||||
@@ -103,8 +102,6 @@ struct Profile {
|
||||
bool ignore_nan_fp_comparisons{};
|
||||
/// Some drivers have broken support for OpVectorExtractDynamic on subgroup mask inputs
|
||||
bool has_broken_spirv_subgroup_mask_vector_extract_dynamic{};
|
||||
bool has_broken_spirv_subgroup_shuffle{};
|
||||
u32 max_subgroup_size{};
|
||||
|
||||
u32 gl_max_compute_smem_size{};
|
||||
|
||||
|
||||
@@ -172,7 +172,6 @@ struct StorageBufferDescriptor {
|
||||
u32 cbuf_offset;
|
||||
u32 count;
|
||||
bool is_written;
|
||||
bool is_global_fallback{};
|
||||
|
||||
auto operator<=>(const StorageBufferDescriptor&) const = default;
|
||||
};
|
||||
|
||||
@@ -22,7 +22,6 @@ add_library(video_core STATIC
|
||||
buffer_cache/usage_tracker.h
|
||||
buffer_cache/virtual_range_cache.h
|
||||
buffer_cache/word_manager.h
|
||||
cache_reclaim.h
|
||||
cache_types.h
|
||||
capture.h
|
||||
cdma_pusher.cpp
|
||||
@@ -168,8 +167,6 @@ add_library(video_core STATIC
|
||||
renderer_vulkan/vk_fence_manager.h
|
||||
renderer_vulkan/vk_graphics_pipeline.cpp
|
||||
renderer_vulkan/vk_graphics_pipeline.h
|
||||
renderer_vulkan/vk_guest_memory.cpp
|
||||
renderer_vulkan/vk_guest_memory.h
|
||||
renderer_vulkan/vk_multi_range_buffer.cpp
|
||||
renderer_vulkan/vk_multi_range_buffer.h
|
||||
renderer_vulkan/vk_master_semaphore.cpp
|
||||
@@ -259,6 +256,8 @@ add_library(video_core STATIC
|
||||
texture_cache/util.h
|
||||
textures/astc.h
|
||||
textures/astc.cpp
|
||||
textures/bcn.cpp
|
||||
textures/bcn.h
|
||||
textures/decoders.cpp
|
||||
textures/decoders.h
|
||||
textures/texture.cpp
|
||||
@@ -407,7 +406,7 @@ if (ENABLE_OPENGL)
|
||||
endif()
|
||||
|
||||
target_link_libraries(video_core PUBLIC common core)
|
||||
target_link_libraries(video_core PUBLIC shader_recompiler bc_decoder gpu_logging)
|
||||
target_link_libraries(video_core PUBLIC shader_recompiler stb bc_decoder gpu_logging)
|
||||
if (ENABLE_OPENGL)
|
||||
target_link_libraries(video_core PUBLIC glad)
|
||||
endif()
|
||||
|
||||
@@ -129,49 +129,13 @@ public:
|
||||
write_tick = write_tick_;
|
||||
}
|
||||
|
||||
u64 ContentSerial() const noexcept {
|
||||
return content_serial;
|
||||
}
|
||||
|
||||
void MarkContentModified() noexcept {
|
||||
content_serial = ++next_content_serial;
|
||||
}
|
||||
|
||||
[[nodiscard]] bool HasDrawHazard(u64 pass, u64 wfi, DAddr addr, u64 size,
|
||||
bool check_feedback) const noexcept {
|
||||
return draw_write_pass == pass &&
|
||||
(draw_write_wfi < wfi || (check_feedback && draw_write_feedback)) &&
|
||||
addr < draw_write_end && addr + size > draw_write_begin;
|
||||
}
|
||||
|
||||
void MarkDrawWrite(u64 pass, u64 wfi, DAddr addr, u64 size, bool feedback) noexcept {
|
||||
if (draw_write_pass != pass) {
|
||||
draw_write_pass = pass;
|
||||
draw_write_begin = addr;
|
||||
draw_write_end = addr + size;
|
||||
draw_write_feedback = false;
|
||||
}
|
||||
draw_write_wfi = wfi;
|
||||
draw_write_feedback |= feedback;
|
||||
draw_write_begin = (std::min)(draw_write_begin, addr);
|
||||
draw_write_end = (std::max)(draw_write_end, addr + size);
|
||||
}
|
||||
|
||||
private:
|
||||
static inline u64 next_content_serial = 0;
|
||||
|
||||
VAddr cpu_addr = 0;
|
||||
BufferFlagBits flags{};
|
||||
int stream_score = 0;
|
||||
size_t lru_id = SIZE_MAX;
|
||||
size_t size_bytes = 0;
|
||||
u64 write_tick = 0;
|
||||
u64 content_serial = ++next_content_serial;
|
||||
u64 draw_write_pass = 0;
|
||||
u64 draw_write_wfi = 0;
|
||||
DAddr draw_write_begin = 0;
|
||||
DAddr draw_write_end = 0;
|
||||
bool draw_write_feedback = false;
|
||||
};
|
||||
|
||||
} // namespace VideoCommon
|
||||
|
||||
@@ -30,54 +30,34 @@ BufferCache<P>::BufferCache(Tegra::MaxwellDeviceMemoryManager& device_memory_, R
|
||||
#ifdef YUZU_LEGACY
|
||||
immediately_free = (Settings::values.vram_usage_mode.GetValue() == Settings::VramUsageMode::Aggressive);
|
||||
#endif
|
||||
device_local_memory = runtime.GetDeviceLocalMemory();
|
||||
const auto thresholds = VideoCommon::MakeReclaimThresholds(
|
||||
device_local_memory, static_cast<u64>(TARGET_THRESHOLD),
|
||||
static_cast<u64>(DEFAULT_EXPECTED_MEMORY), static_cast<u64>(DEFAULT_CRITICAL_MEMORY),
|
||||
HEAP_PRESSURE_HEADROOM);
|
||||
minimum_memory = thresholds.minimum;
|
||||
expected_memory = thresholds.expected;
|
||||
critical_memory = thresholds.critical;
|
||||
heap_headroom = thresholds.headroom;
|
||||
if (!runtime.CanReportMemoryUsage()) {
|
||||
minimum_memory = DEFAULT_EXPECTED_MEMORY;
|
||||
critical_memory = DEFAULT_CRITICAL_MEMORY;
|
||||
return;
|
||||
}
|
||||
|
||||
const s64 device_local_memory = static_cast<s64>(runtime.GetDeviceLocalMemory());
|
||||
const s64 min_spacing_expected = device_local_memory - 1_GiB;
|
||||
const s64 min_spacing_critical = device_local_memory - 512_MiB;
|
||||
const s64 mem_threshold = (std::min)(device_local_memory, TARGET_THRESHOLD);
|
||||
const s64 min_vacancy_expected = (6 * mem_threshold) / 10;
|
||||
const s64 min_vacancy_critical = (2 * mem_threshold) / 10;
|
||||
minimum_memory = static_cast<u64>(
|
||||
(std::max)((std::min)(device_local_memory - min_vacancy_expected, min_spacing_expected),
|
||||
DEFAULT_EXPECTED_MEMORY));
|
||||
critical_memory = static_cast<u64>(
|
||||
(std::max)((std::min)(device_local_memory - min_vacancy_critical, min_spacing_critical),
|
||||
DEFAULT_CRITICAL_MEMORY));
|
||||
}
|
||||
|
||||
template <class P>
|
||||
BufferCache<P>::~BufferCache() = default;
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::ReclaimInline() {
|
||||
if (total_used_memory < minimum_memory) {
|
||||
return;
|
||||
}
|
||||
int num_iterations = 8;
|
||||
const auto clean_up = [this, &num_iterations](BufferId buffer_id) {
|
||||
if (num_iterations == 0) {
|
||||
return true;
|
||||
}
|
||||
--num_iterations;
|
||||
Buffer& buffer = slot_buffers[buffer_id];
|
||||
if (memory_tracker.IsRegionGpuModified(buffer.CpuAddr(), buffer.SizeBytes())) {
|
||||
return false;
|
||||
}
|
||||
DeleteBuffer(buffer_id);
|
||||
return false;
|
||||
};
|
||||
lru_cache.ForEachItemBelow(frame_tick - INLINE_TICKS_TO_DESTROY, clean_up);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::RunGarbageCollector() {
|
||||
const bool aggressive_gc = heap_pressure || total_used_memory >= critical_memory;
|
||||
const bool priority_gc = aggressive_gc || total_used_memory >= expected_memory;
|
||||
u64 ticks_to_destroy = 120;
|
||||
int num_iterations = 32;
|
||||
if (aggressive_gc) {
|
||||
ticks_to_destroy = 30;
|
||||
num_iterations = 64;
|
||||
} else if (priority_gc) {
|
||||
ticks_to_destroy = 60;
|
||||
num_iterations = 48;
|
||||
}
|
||||
const bool aggressive_gc = total_used_memory >= critical_memory;
|
||||
const u64 ticks_to_destroy = aggressive_gc ? 60 : 120;
|
||||
int num_iterations = aggressive_gc ? 64 : 32;
|
||||
const auto clean_up = [this, &num_iterations](BufferId buffer_id) {
|
||||
if (num_iterations == 0) {
|
||||
return true;
|
||||
@@ -116,11 +96,11 @@ void BufferCache<P>::TickFrame() {
|
||||
const bool skip_preferred = hits * 256 < shots * 251;
|
||||
channel_state->uniform_buffer_skip_cache_size = skip_preferred ? DEFAULT_SKIP_CACHE_SIZE : 0;
|
||||
|
||||
heap_pressure = false;
|
||||
if (device_local_memory != 0 && runtime.CanReportMemoryUsage()) {
|
||||
heap_pressure = runtime.GetDeviceMemoryUsage() + heap_headroom >= device_local_memory;
|
||||
// If we can obtain the memory info, use it instead of the estimate.
|
||||
if (runtime.CanReportMemoryUsage()) {
|
||||
total_used_memory = runtime.GetDeviceMemoryUsage();
|
||||
}
|
||||
if (total_used_memory >= minimum_memory || heap_pressure) {
|
||||
if (total_used_memory >= minimum_memory) {
|
||||
RunGarbageCollector();
|
||||
}
|
||||
++frame_tick;
|
||||
@@ -266,7 +246,6 @@ bool BufferCache<P>::DMACopy(GPUVAddr src_address, GPUVAddr dest_address, u64 am
|
||||
const auto& copy = copies[0];
|
||||
src_buffer.MarkUsage(copy.src_offset, copy.size);
|
||||
dest_buffer.MarkUsage(copy.dst_offset, copy.size);
|
||||
dest_buffer.MarkContentModified();
|
||||
runtime.CopyBuffer(dest_buffer, src_buffer, copies, true);
|
||||
if (has_new_downloads) {
|
||||
memory_tracker.MarkRegionAsGpuModified(*cpu_dest_address, amount);
|
||||
@@ -298,7 +277,6 @@ bool BufferCache<P>::DMAClear(GPUVAddr dst_address, u64 amount, u32 value) {
|
||||
const u32 offset = dest_buffer.Offset(*cpu_dst_address);
|
||||
runtime.ClearBuffer(dest_buffer, offset, size, value);
|
||||
dest_buffer.MarkUsage(offset, size);
|
||||
dest_buffer.MarkContentModified();
|
||||
return true;
|
||||
}
|
||||
|
||||
@@ -327,7 +305,6 @@ std::pair<typename P::Buffer*, u32> BufferCache<P>::ObtainCPUBuffer(
|
||||
default:
|
||||
break;
|
||||
}
|
||||
buffer.MarkUsage(buffer.Offset(device_addr), size);
|
||||
|
||||
switch (post_op) {
|
||||
case ObtainBufferOperation::MarkAsWritten:
|
||||
@@ -367,45 +344,14 @@ void BufferCache<P>::DisableGraphicsUniformBuffer(size_t stage, u32 index) {
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::UpdateGraphicsBuffers(bool is_indexed) {
|
||||
if constexpr (!IS_OPENGL) {
|
||||
draw_writes.clear();
|
||||
draw_pass = runtime.RenderPassSerial();
|
||||
draw_wfi = runtime.WaitForIdleSerial();
|
||||
draw_hazard = false;
|
||||
recording_draw = true;
|
||||
}
|
||||
ReclaimInline();
|
||||
do {
|
||||
channel_state->has_deleted_buffers = false;
|
||||
DoUpdateGraphicsBuffers(is_indexed);
|
||||
} while (channel_state->has_deleted_buffers);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
bool BufferCache<P>::TakeDrawHazard() noexcept {
|
||||
return std::exchange(draw_hazard, false);
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::CommitDrawWrites() {
|
||||
if constexpr (!IS_OPENGL) {
|
||||
recording_draw = false;
|
||||
if (draw_writes.empty()) {
|
||||
return;
|
||||
}
|
||||
const u64 pass = runtime.RenderPassSerial();
|
||||
for (const DrawWrite& write : draw_writes) {
|
||||
Buffer& buffer = slot_buffers[write.buffer_id];
|
||||
buffer.setWriteTick(runtime.CurrentTick());
|
||||
buffer.MarkDrawWrite(pass, draw_wfi, write.device_addr, write.size, write.feedback);
|
||||
}
|
||||
runtime.MarkRenderPassWrites();
|
||||
}
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::UpdateComputeBuffers() {
|
||||
ReclaimInline();
|
||||
do {
|
||||
channel_state->has_deleted_buffers = false;
|
||||
DoUpdateComputeBuffers();
|
||||
@@ -802,35 +748,17 @@ void BufferCache<P>::BindHostIndexBuffer() {
|
||||
const u32 offset = buffer.Offset(channel_state->index_buffer.device_addr);
|
||||
const u32 size = channel_state->index_buffer.size;
|
||||
const auto& draw_state = maxwell3d->draw_manager.draw_state;
|
||||
if constexpr (!HAS_FULL_INDEX_AND_PRIMITIVE_SUPPORT) {
|
||||
if (draw_state.topology == Maxwell::PrimitiveTopology::Quads ||
|
||||
draw_state.topology == Maxwell::PrimitiveTopology::QuadStrip) {
|
||||
const DAddr device_addr = channel_state->index_buffer.device_addr;
|
||||
std::span<const u8> indices = draw_state.inline_index_draw_indexes;
|
||||
if (indices.empty() && !IsRegionGpuModified(device_addr, size)) {
|
||||
indices = ImmediateBufferWithData(device_addr, size);
|
||||
}
|
||||
if (runtime.BindQuadIndices(draw_state.topology, draw_state.index_buffer.format,
|
||||
draw_state.index_buffer.first,
|
||||
draw_state.index_buffer.count, indices)) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
if (draw_state.inline_index_draw_indexes.empty()) {
|
||||
SynchronizeBuffer(buffer, channel_state->index_buffer.device_addr, size);
|
||||
} else if constexpr (USE_MEMORY_MAPS_FOR_UPLOADS) {
|
||||
const auto upload_staging = runtime.UploadStagingBuffer(size);
|
||||
std::memcpy(upload_staging.mapped_span.data(),
|
||||
draw_state.inline_index_draw_indexes.data(), size);
|
||||
runtime.BindIndexBuffer(draw_state.topology, draw_state.index_buffer.format,
|
||||
draw_state.index_buffer.first, draw_state.index_buffer.count,
|
||||
upload_staging.buffer, static_cast<u32>(upload_staging.offset),
|
||||
size);
|
||||
return;
|
||||
} else {
|
||||
buffer.MarkContentModified();
|
||||
buffer.ImmediateUpload(0, draw_state.inline_index_draw_indexes);
|
||||
if constexpr (USE_MEMORY_MAPS_FOR_UPLOADS) {
|
||||
auto upload_staging = runtime.UploadStagingBuffer(size);
|
||||
std::array<BufferCopy, 1> copies{{BufferCopy{.src_offset = upload_staging.offset, .dst_offset = 0, .size = size}}};
|
||||
std::memcpy(upload_staging.mapped_span.data(), draw_state.inline_index_draw_indexes.data(), size);
|
||||
runtime.CopyBuffer(buffer, upload_staging.buffer, copies, true);
|
||||
} else {
|
||||
buffer.ImmediateUpload(0, draw_state.inline_index_draw_indexes);
|
||||
}
|
||||
}
|
||||
if constexpr (HAS_FULL_INDEX_AND_PRIMITIVE_SUPPORT) {
|
||||
const u32 new_offset = offset + draw_state.index_buffer.first * u32(draw_state.index_buffer.FormatSizeInBytes());
|
||||
@@ -975,7 +903,6 @@ void BufferCache<P>::BindHostDrawIndirectBuffers() {
|
||||
Buffer& buffer = slot_buffers[binding.buffer_id];
|
||||
TouchBuffer(buffer, binding.buffer_id);
|
||||
SynchronizeBuffer(buffer, binding.device_addr, binding.size);
|
||||
buffer.MarkUsage(buffer.Offset(binding.device_addr), binding.size);
|
||||
};
|
||||
if (current_draw_indirect->include_count) {
|
||||
bind_buffer(channel_state->count_buffer_binding);
|
||||
@@ -1079,14 +1006,15 @@ void BufferCache<P>::BindHostGraphicsUniformBuffer(size_t stage, u32 index, u32
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::ResolveMultiRangeStorage(Binding& binding,
|
||||
std::vector<MultiRangeSegment>& pool,
|
||||
bool is_written) {
|
||||
void BufferCache<P>::ResolveMultiRangeStorage(Binding& binding, bool is_written,
|
||||
std::vector<MultiRangeSegment>& pool) {
|
||||
binding.segment_first = 0;
|
||||
binding.segment_count = 0;
|
||||
if constexpr (requires { runtime.BindMultiRangeStorageBuffer(u64{}, bool{}); }) {
|
||||
if (binding.gpu_addr == 0 || binding.size == 0 ||
|
||||
(is_written && !runtime.PrefersSparseSources())) {
|
||||
if (binding.gpu_addr == 0 || binding.size == 0) {
|
||||
return;
|
||||
}
|
||||
if (is_written && !runtime.PrefersSparseSources()) {
|
||||
return;
|
||||
}
|
||||
const VirtualSegments* found =
|
||||
@@ -1094,7 +1022,7 @@ void BufferCache<P>::ResolveMultiRangeStorage(Binding& binding,
|
||||
if (!found || found->size() < 2) {
|
||||
return;
|
||||
}
|
||||
const VirtualSegments& segments = *found;
|
||||
const VirtualSegments segments = *found;
|
||||
const u32 first = static_cast<u32>(pool.size());
|
||||
const bool prefer_sparse = runtime.PrefersSparseSources();
|
||||
for (const VirtualSegment& segment : segments) {
|
||||
@@ -1131,7 +1059,9 @@ bool BufferCache<P>::BindMultiRangeStorage(const Binding& binding, bool is_writt
|
||||
const MultiRangeSegment& segment = pool[binding.segment_first + index];
|
||||
Buffer& buffer = slot_buffers[segment.buffer_id];
|
||||
TouchBuffer(buffer, segment.buffer_id);
|
||||
SynchronizeBuffer(buffer, segment.device_addr, segment.size);
|
||||
if (SynchronizeBuffer(buffer, segment.device_addr, segment.size)) {
|
||||
runtime.InvalidateMultiRange(key);
|
||||
}
|
||||
const u32 offset = buffer.Offset(segment.device_addr);
|
||||
buffer.MarkUsage(offset, segment.size);
|
||||
if (is_written) {
|
||||
@@ -1222,8 +1152,8 @@ void BufferCache<P>::BindHostTransformFeedbackBuffers() {
|
||||
Buffer& buffer = slot_buffers[binding.buffer_id];
|
||||
TouchBuffer(buffer, binding.buffer_id);
|
||||
size = binding.size;
|
||||
SynchronizeBuffer(buffer, binding.device_addr, size, false);
|
||||
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size, true);
|
||||
SynchronizeBuffer(buffer, binding.device_addr, size);
|
||||
MarkWrittenBuffer(binding.buffer_id, binding.device_addr, size);
|
||||
offset = buffer.Offset(binding.device_addr);
|
||||
buffer.MarkUsage(offset, size);
|
||||
host_buffer = &buffer;
|
||||
@@ -1387,15 +1317,13 @@ void BufferCache<P>::UpdateIndexBuffer() {
|
||||
flags[Dirty::IndexBuffer] = false;
|
||||
if (!draw_state.inline_index_draw_indexes.empty()) [[unlikely]] {
|
||||
auto inline_index_size = static_cast<u32>(draw_state.inline_index_draw_indexes.size());
|
||||
if constexpr (!USE_MEMORY_MAPS_FOR_UPLOADS) {
|
||||
const u32 buffer_size = Common::AlignUp(inline_index_size, CACHING_PAGESIZE);
|
||||
if (inline_buffer_id == NULL_BUFFER_ID) [[unlikely]] {
|
||||
inline_buffer_id = CreateBuffer(0, buffer_size, false);
|
||||
}
|
||||
if (slot_buffers[inline_buffer_id].SizeBytes() < buffer_size) [[unlikely]] {
|
||||
DeleteBuffer(inline_buffer_id, true);
|
||||
inline_buffer_id = CreateBuffer(0, buffer_size, false);
|
||||
}
|
||||
u32 buffer_size = Common::AlignUp(inline_index_size, CACHING_PAGESIZE);
|
||||
if (inline_buffer_id == NULL_BUFFER_ID) [[unlikely]] {
|
||||
inline_buffer_id = CreateBuffer(0, buffer_size, false);
|
||||
}
|
||||
if (slot_buffers[inline_buffer_id].SizeBytes() < buffer_size) [[unlikely]] {
|
||||
slot_buffers.erase(inline_buffer_id);
|
||||
inline_buffer_id = CreateBuffer(0, buffer_size, false);
|
||||
}
|
||||
channel_state->index_buffer = Binding{
|
||||
.device_addr = 0,
|
||||
@@ -1509,12 +1437,10 @@ void BufferCache<P>::UpdateStorageBuffers(size_t stage) {
|
||||
ForEachEnabledBit(channel_state->enabled_storage_buffers[stage], [&](u32 index) {
|
||||
// Resolve buffer
|
||||
Binding& binding = channel_state->storage_buffers[stage][index];
|
||||
const BufferId buffer_id = FindBuffer(binding.device_addr, binding.size, false);
|
||||
binding.buffer_id = buffer_id;
|
||||
const bool is_written = ((channel_state->written_storage_buffers[stage] >> index) & 1) != 0;
|
||||
ResolveMultiRangeStorage(binding, graphics_segments, is_written);
|
||||
binding.buffer_id = NULL_BUFFER_ID;
|
||||
if (binding.segment_count == 0 || is_written) {
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false);
|
||||
}
|
||||
ResolveMultiRangeStorage(binding, is_written, graphics_segments);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1577,12 +1503,10 @@ void BufferCache<P>::UpdateComputeStorageBuffers() {
|
||||
ForEachEnabledBit(channel_state->enabled_compute_storage_buffers, [&](u32 index) {
|
||||
// Resolve buffer
|
||||
Binding& binding = channel_state->compute_storage_buffers[index];
|
||||
const bool is_written = ((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
|
||||
ResolveMultiRangeStorage(binding, compute_segments, is_written);
|
||||
binding.buffer_id = NULL_BUFFER_ID;
|
||||
if (binding.segment_count == 0 || is_written) {
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false);
|
||||
}
|
||||
binding.buffer_id = FindBuffer(binding.device_addr, binding.size, false);
|
||||
const bool is_written =
|
||||
((channel_state->written_compute_storage_buffers >> index) & 1) != 0;
|
||||
ResolveMultiRangeStorage(binding, is_written, compute_segments);
|
||||
});
|
||||
}
|
||||
|
||||
@@ -1595,15 +1519,10 @@ void BufferCache<P>::UpdateComputeTextureBuffers() {
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u32 size,
|
||||
bool feedback) {
|
||||
void BufferCache<P>::MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u32 size) {
|
||||
if constexpr (!IS_OPENGL) {
|
||||
Buffer& buffer = slot_buffers[buffer_id];
|
||||
buffer.setWriteTick(runtime.CurrentTick());
|
||||
buffer.MarkContentModified();
|
||||
if (recording_draw) {
|
||||
draw_writes.push_back({buffer_id, device_addr, size, feedback});
|
||||
}
|
||||
}
|
||||
memory_tracker.MarkRegionAsGpuModified(device_addr, size);
|
||||
gpu_modified_ranges.Add(device_addr, size);
|
||||
@@ -1761,11 +1680,6 @@ BufferId BufferCache<P>::CreateBuffer(DAddr device_addr, u32 wanted_size,
|
||||
wanted_size = static_cast<u32>(device_addr_end - device_addr);
|
||||
const OverlapResult overlap = ResolveOverlaps(device_addr, wanted_size);
|
||||
const u32 size = static_cast<u32>(overlap.end - overlap.begin);
|
||||
if constexpr (requires(Buffer& buffer) { buffer.IsSparseCompatible(); }) {
|
||||
for (const BufferId overlap_id : overlap.ids) {
|
||||
sparse_compatible |= slot_buffers[overlap_id].IsSparseCompatible();
|
||||
}
|
||||
}
|
||||
const BufferId new_buffer_id =
|
||||
slot_buffers.insert(runtime, overlap.begin, size, sparse_compatible);
|
||||
auto& new_buffer = slot_buffers[new_buffer_id];
|
||||
@@ -1823,12 +1737,7 @@ void BufferCache<P>::TouchBuffer(Buffer& buffer, BufferId buffer_id) noexcept {
|
||||
}
|
||||
|
||||
template <class P>
|
||||
bool BufferCache<P>::SynchronizeBuffer(Buffer& buffer, DAddr device_addr, u32 size,
|
||||
bool check_feedback) {
|
||||
if constexpr (!IS_OPENGL) {
|
||||
draw_hazard |=
|
||||
buffer.HasDrawHazard(draw_pass, draw_wfi, device_addr, size, check_feedback);
|
||||
}
|
||||
bool BufferCache<P>::SynchronizeBuffer(Buffer& buffer, DAddr device_addr, u32 size) {
|
||||
upload_copies.clear();
|
||||
u64 total_size_bytes = 0;
|
||||
u64 largest_copy = 0;
|
||||
@@ -1847,16 +1756,13 @@ bool BufferCache<P>::SynchronizeBuffer(Buffer& buffer, DAddr device_addr, u32 si
|
||||
}
|
||||
const std::span<BufferCopy> copies_span(upload_copies.data(), upload_copies.size());
|
||||
UploadMemory(buffer, total_size_bytes, largest_copy, copies_span);
|
||||
if constexpr (IS_OPENGL) {
|
||||
any_buffer_uploaded = true;
|
||||
}
|
||||
any_buffer_uploaded = true;
|
||||
return false;
|
||||
}
|
||||
|
||||
template <class P>
|
||||
void BufferCache<P>::UploadMemory(Buffer& buffer, u64 total_size_bytes, u64 largest_copy,
|
||||
std::span<BufferCopy> copies) {
|
||||
buffer.MarkContentModified();
|
||||
if constexpr (USE_MEMORY_MAPS_FOR_UPLOADS) {
|
||||
MappedUploadMemory(buffer, total_size_bytes, copies);
|
||||
} else {
|
||||
@@ -1898,20 +1804,6 @@ void BufferCache<P>::MappedUploadMemory([[maybe_unused]] Buffer& buffer,
|
||||
[[maybe_unused]] u64 total_size_bytes,
|
||||
[[maybe_unused]] std::span<BufferCopy> copies) {
|
||||
if constexpr (USE_MEMORY_MAPS) {
|
||||
if constexpr (requires { runtime.DirectUploadSpan(buffer, copies); }) {
|
||||
const std::span<u8> direct = runtime.DirectUploadSpan(buffer, copies);
|
||||
if (!direct.empty()) {
|
||||
for (const BufferCopy& copy : copies) {
|
||||
const DAddr device_addr = buffer.CpuAddr() + copy.dst_offset;
|
||||
if (Settings::values.enable_gpu_buffer_readback.GetValue()) {
|
||||
DownloadBufferMemory(buffer, device_addr, copy.size);
|
||||
}
|
||||
device_memory.ReadBlockUnsafe(device_addr, direct.data() + copy.dst_offset,
|
||||
copy.size);
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
auto upload_staging = runtime.UploadStagingBuffer(total_size_bytes);
|
||||
const std::span<u8> staging_pointer = upload_staging.mapped_span;
|
||||
for (BufferCopy& copy : copies) {
|
||||
@@ -1956,7 +1848,6 @@ void BufferCache<P>::InlineMemoryImplementation(DAddr dest_address, size_t copy_
|
||||
BufferId buffer_id = FindBuffer(dest_address, static_cast<u32>(copy_size), false);
|
||||
auto& buffer = slot_buffers[buffer_id];
|
||||
SynchronizeBuffer(buffer, dest_address, static_cast<u32>(copy_size));
|
||||
buffer.MarkContentModified();
|
||||
|
||||
if constexpr (USE_MEMORY_MAPS_FOR_UPLOADS) {
|
||||
auto upload_staging = runtime.UploadStagingBuffer(copy_size);
|
||||
@@ -2011,16 +1902,6 @@ void BufferCache<P>::DownloadBufferMemory(Buffer& buffer, DAddr device_addr, u64
|
||||
}
|
||||
|
||||
if constexpr (USE_MEMORY_MAPS) {
|
||||
if constexpr (requires { runtime.DirectDownloadSpan(buffer); }) {
|
||||
const std::span<const u8> direct = runtime.DirectDownloadSpan(buffer);
|
||||
if (!direct.empty()) {
|
||||
for (const BufferCopy& copy : copies) {
|
||||
device_memory.WriteBlockUnsafe(buffer.CpuAddr() + copy.src_offset,
|
||||
direct.data() + copy.src_offset, copy.size);
|
||||
}
|
||||
return;
|
||||
}
|
||||
}
|
||||
auto download_staging = runtime.DownloadStagingBuffer(total_size_bytes);
|
||||
const u8* const mapped_memory = download_staging.mapped_span.data();
|
||||
const std::span<BufferCopy> copies_span(copies.data(), copies.data() + copies.size());
|
||||
@@ -2089,10 +1970,6 @@ void BufferCache<P>::DeleteBuffer(BufferId buffer_id, bool do_not_mark) {
|
||||
memory_tracker.MarkRegionAsCpuModified(buffer.CpuAddr(), buffer.SizeBytes());
|
||||
}
|
||||
|
||||
if (inline_buffer_id == buffer_id) {
|
||||
inline_buffer_id = NULL_BUFFER_ID;
|
||||
}
|
||||
|
||||
Unregister(buffer_id);
|
||||
|
||||
#ifdef YUZU_LEGACY
|
||||
|
||||
@@ -31,7 +31,6 @@
|
||||
#include "video_core/buffer_cache/buffer_base.h"
|
||||
#include "video_core/buffer_cache/virtual_range_cache.h"
|
||||
#include "video_core/control/channel_state_cache.h"
|
||||
#include "video_core/cache_reclaim.h"
|
||||
#include "video_core/delayed_destruction_ring.h"
|
||||
#include "video_core/dirty_flags.h"
|
||||
#include "video_core/engines/maxwell_3d.h"
|
||||
@@ -200,8 +199,6 @@ class BufferCache : public VideoCommon::ChannelSetupCaches<BufferCacheChannelInf
|
||||
|
||||
static constexpr s64 DEFAULT_EXPECTED_MEMORY = 512_MiB;
|
||||
static constexpr s64 DEFAULT_CRITICAL_MEMORY = 1_GiB;
|
||||
static constexpr u64 HEAP_PRESSURE_HEADROOM = 512_MiB;
|
||||
static constexpr u64 INLINE_TICKS_TO_DESTROY = 240;
|
||||
|
||||
// Debug Flags.
|
||||
|
||||
@@ -231,12 +228,8 @@ public:
|
||||
bool BindMultiRangeStorage(const Binding& binding, bool is_written,
|
||||
std::span<const MultiRangeSegment> pool);
|
||||
|
||||
|
||||
|
||||
|
||||
|
||||
void ResolveMultiRangeStorage(Binding& binding, std::vector<MultiRangeSegment>& pool,
|
||||
bool is_written);
|
||||
void ResolveMultiRangeStorage(Binding& binding, bool is_written,
|
||||
std::vector<MultiRangeSegment>& pool);
|
||||
|
||||
void UnmapGPUMemory(size_t as_id, GPUVAddr gpu_addr, size_t size);
|
||||
|
||||
@@ -260,10 +253,6 @@ public:
|
||||
|
||||
void UpdateComputeBuffers();
|
||||
|
||||
[[nodiscard]] bool TakeDrawHazard() noexcept;
|
||||
|
||||
void CommitDrawWrites();
|
||||
|
||||
void BindHostGeometryBuffers(bool is_indexed);
|
||||
|
||||
void BindHostStageBuffers(size_t stage);
|
||||
@@ -389,8 +378,6 @@ private:
|
||||
|
||||
void RunGarbageCollector();
|
||||
|
||||
void ReclaimInline();
|
||||
|
||||
void BindHostIndexBuffer();
|
||||
|
||||
void BindHostVertexBuffers();
|
||||
@@ -443,8 +430,7 @@ private:
|
||||
|
||||
void UpdateComputeTextureBuffers();
|
||||
|
||||
void MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u32 size,
|
||||
bool feedback = false);
|
||||
void MarkWrittenBuffer(BufferId buffer_id, DAddr device_addr, u32 size);
|
||||
|
||||
[[nodiscard]] BufferId FindBuffer(DAddr device_addr, u32 size, bool sparse_compatible);
|
||||
|
||||
@@ -466,8 +452,7 @@ private:
|
||||
|
||||
void TouchBuffer(Buffer& buffer, BufferId buffer_id) noexcept;
|
||||
|
||||
bool SynchronizeBuffer(Buffer& buffer, DAddr device_addr, u32 size,
|
||||
bool check_feedback = true);
|
||||
bool SynchronizeBuffer(Buffer& buffer, DAddr device_addr, u32 size);
|
||||
|
||||
void UploadMemory(Buffer& buffer, u64 total_size_bytes, u64 largest_copy,
|
||||
std::span<BufferCopy> copies);
|
||||
@@ -526,18 +511,6 @@ private:
|
||||
|
||||
boost::container::small_vector<BufferCopy, 4> upload_copies;
|
||||
|
||||
struct DrawWrite {
|
||||
BufferId buffer_id;
|
||||
DAddr device_addr;
|
||||
u32 size;
|
||||
bool feedback;
|
||||
};
|
||||
boost::container::small_vector<DrawWrite, 8> draw_writes;
|
||||
u64 draw_pass = 0;
|
||||
u64 draw_wfi = 0;
|
||||
bool draw_hazard = false;
|
||||
bool recording_draw = false;
|
||||
|
||||
MemoryTracker memory_tracker;
|
||||
Common::RangeSet<DAddr> uncommitted_gpu_modified_ranges;
|
||||
Common::RangeSet<DAddr> gpu_modified_ranges;
|
||||
@@ -564,12 +537,8 @@ private:
|
||||
std::vector<MultiRangeSegment> compute_segments;
|
||||
u64 frame_tick = 0;
|
||||
u64 total_used_memory = 0;
|
||||
u64 device_local_memory = 0;
|
||||
u64 minimum_memory = 0;
|
||||
u64 expected_memory = 0;
|
||||
u64 critical_memory = 0;
|
||||
u64 heap_headroom = 0;
|
||||
bool heap_pressure = false;
|
||||
BufferId inline_buffer_id;
|
||||
#ifdef YUZU_LEGACY
|
||||
bool immediately_free = false;
|
||||
|
||||
Some files were not shown because too many files have changed in this diff Show More
Reference in New Issue
Block a user