mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-10-06 14:09:58 +00:00
Another take on the qcom/turnip missing handling for geometry stages
This commit is contained in:
@@ -4,6 +4,7 @@
|
||||
// SPDX-FileCopyrightText: Copyright 2021 yuzu Emulator Project
|
||||
// SPDX-License-Identifier: GPL-2.0-or-later
|
||||
|
||||
#include <algorithm>
|
||||
#include <span>
|
||||
#include <tuple>
|
||||
#include <type_traits>
|
||||
@@ -439,10 +440,13 @@ void SetupCapabilities(const Profile& profile, const Info& info, EmitContext& ct
|
||||
ctx.AddCapability(spv::Capability::DrawParameters);
|
||||
}
|
||||
if ((info.uses_subgroup_vote || info.uses_subgroup_invocation_id ||
|
||||
info.uses_subgroup_shuffles) &&
|
||||
info.uses_subgroup_shuffles || info.uses_subgroup_mask) &&
|
||||
profile.support_vote && profile.SupportsSubgroupStage(ctx.stage)) {
|
||||
ctx.AddCapability(spv::Capability::GroupNonUniformBallot);
|
||||
ctx.AddCapability(spv::Capability::GroupNonUniformShuffle);
|
||||
if (info.uses_subgroup_shuffles && profile.support_shuffle_relative) {
|
||||
ctx.AddCapability(spv::Capability::GroupNonUniformShuffleRelative);
|
||||
}
|
||||
if (!profile.warp_size_potentially_larger_than_guest) {
|
||||
// vote ops are only used when not taking the long path
|
||||
ctx.AddCapability(spv::Capability::GroupNonUniformVote);
|
||||
@@ -521,6 +525,29 @@ void PatchPhiNodes(IR::Program& program, EmitContext& ctx) {
|
||||
return { ctx.Def(phi->Arg(phi_arg)), parent };
|
||||
});
|
||||
}
|
||||
|
||||
void RewriteOpcodes(std::vector<u32>& code, std::span<const std::pair<u32, spv::Op>> rewrites) {
|
||||
if (rewrites.empty()) {
|
||||
return;
|
||||
}
|
||||
size_t offset = 5;
|
||||
while (offset + 2 < code.size()) {
|
||||
const u32 word_count{code[offset] >> 16};
|
||||
const auto opcode{static_cast<spv::Op>(code[offset] & 0xFFFFu)};
|
||||
if (word_count == 0) {
|
||||
return;
|
||||
}
|
||||
if (opcode == spv::Op::OpGroupNonUniformShuffleXor ||
|
||||
opcode == spv::Op::OpGroupNonUniformQuadBroadcast) {
|
||||
const auto it{std::ranges::find(rewrites, code[offset + 2],
|
||||
&std::pair<u32, spv::Op>::first)};
|
||||
if (it != rewrites.end()) {
|
||||
code[offset] = (code[offset] & 0xFFFF0000u) | static_cast<u32>(it->second);
|
||||
}
|
||||
}
|
||||
offset += word_count;
|
||||
}
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
std::vector<u32> EmitSPIRV(const Profile& profile, const RuntimeInfo& runtime_info, IR::Program& program, Bindings& bindings) {
|
||||
@@ -535,7 +562,9 @@ std::vector<u32> EmitSPIRV(const Profile& profile, const RuntimeInfo& runtime_in
|
||||
SetupCapabilities(profile, program.info, ctx);
|
||||
SetupTransformFeedbackCapabilities(ctx, main);
|
||||
PatchPhiNodes(program, ctx);
|
||||
return ctx.Assemble();
|
||||
std::vector<u32> code{ctx.Assemble()};
|
||||
RewriteOpcodes(code, ctx.opcode_rewrites);
|
||||
return code;
|
||||
}
|
||||
|
||||
Id EmitPhi(EmitContext& ctx, IR::Inst* inst) {
|
||||
|
||||
@@ -77,20 +77,57 @@ Id GetMaxThreadId(EmitContext& ctx, Id thread_id, Id clamp, Id segmentation_mask
|
||||
return ComputeMaxThreadId(ctx, min_thread_id, clamp, not_seg_mask);
|
||||
}
|
||||
|
||||
Id SelectValue(EmitContext& ctx, Id in_range, Id value, Id src_thread_id) {
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return value;
|
||||
Id HostThreadId(EmitContext& ctx, Id thread_id) {
|
||||
if (!ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
return thread_id;
|
||||
}
|
||||
return ctx.OpSelect(
|
||||
ctx.U32[1], in_range,
|
||||
ctx.OpGroupNonUniformShuffle(ctx.U32[1], SubgroupScope(ctx), value, src_thread_id), value);
|
||||
}
|
||||
|
||||
Id AddPartitionBase(EmitContext& ctx, Id thread_id) {
|
||||
const Id partition_idx{ctx.OpShiftRightLogical(ctx.U32[1], GetThreadId(ctx), ctx.Const(5u))};
|
||||
const Id partition_base{ctx.OpShiftLeftLogical(ctx.U32[1], partition_idx, ctx.Const(5u))};
|
||||
return ctx.OpIAdd(ctx.U32[1], thread_id, partition_base);
|
||||
}
|
||||
|
||||
Id GuestLane(EmitContext& ctx, Id index) {
|
||||
return ctx.OpBitwiseAnd(ctx.U32[1], index, ctx.Const(31U));
|
||||
}
|
||||
|
||||
Id ShuffleAbsolute(EmitContext& ctx, Id value, Id src_thread_id) {
|
||||
if (!ctx.profile.has_broken_spirv_subgroup_shuffle) {
|
||||
return ctx.OpGroupNonUniformShuffle(ctx.U32[1], SubgroupScope(ctx), value, src_thread_id);
|
||||
}
|
||||
Id result{ctx.u32_zero_value};
|
||||
for (u32 lane = 0; lane < ctx.profile.max_subgroup_size; ++lane) {
|
||||
const Id read{
|
||||
ctx.OpGroupNonUniformBroadcast(ctx.U32[1], SubgroupScope(ctx), value, ctx.Const(lane))};
|
||||
const Id matches{ctx.OpIEqual(ctx.U1, src_thread_id, ctx.Const(lane))};
|
||||
result = ctx.OpSelect(ctx.U32[1], matches, read, result);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
|
||||
Id ShuffleRelative(EmitContext& ctx, Id value, Id delta, Id src_thread_id, spv::Op op) {
|
||||
if (!ctx.profile.support_shuffle_relative) {
|
||||
return ShuffleAbsolute(ctx, value, HostThreadId(ctx, src_thread_id));
|
||||
}
|
||||
const Id result{ctx.OpGroupNonUniformShuffleXor(ctx.U32[1], SubgroupScope(ctx), value, delta)};
|
||||
ctx.opcode_rewrites.emplace_back(result.value, op);
|
||||
return result;
|
||||
}
|
||||
|
||||
Id BroadcastLane(EmitContext& ctx, Id value, u32 lane) {
|
||||
Id result{
|
||||
ctx.OpGroupNonUniformBroadcast(ctx.U32[1], SubgroupScope(ctx), value, ctx.Const(lane))};
|
||||
if (!ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
return result;
|
||||
}
|
||||
const Id partition_idx{ctx.OpShiftRightLogical(ctx.U32[1], GetThreadId(ctx), ctx.Const(5u))};
|
||||
for (u32 base = 32; base < ctx.profile.max_subgroup_size; base += 32) {
|
||||
const Id read{ctx.OpGroupNonUniformBroadcast(ctx.U32[1], SubgroupScope(ctx), value,
|
||||
ctx.Const(base + lane))};
|
||||
const Id matches{ctx.OpIEqual(ctx.U1, partition_idx, ctx.Const(base >> 5))};
|
||||
result = ctx.OpSelect(ctx.U32[1], matches, read, result);
|
||||
}
|
||||
return result;
|
||||
}
|
||||
} // Anonymous namespace
|
||||
|
||||
Id EmitLaneId(EmitContext& ctx) {
|
||||
@@ -203,61 +240,75 @@ Id EmitShuffleIndex(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id cla
|
||||
const Id min_thread_id{ComputeMinThreadId(ctx, thread_id, segmentation_mask)};
|
||||
const Id max_thread_id{ComputeMaxThreadId(ctx, min_thread_id, clamp, not_seg_mask)};
|
||||
|
||||
const Id lhs{ctx.OpBitwiseAnd(ctx.U32[1], index, not_seg_mask)};
|
||||
Id src_thread_id{ctx.OpBitwiseOr(ctx.U32[1], lhs, min_thread_id)};
|
||||
const Id lhs{ctx.OpBitwiseAnd(ctx.U32[1], GuestLane(ctx, index), not_seg_mask)};
|
||||
const Id src_thread_id{ctx.OpBitwiseOr(ctx.U32[1], lhs, min_thread_id)};
|
||||
const Id in_range{ctx.OpSLessThanEqual(ctx.U1, src_thread_id, max_thread_id)};
|
||||
|
||||
if (ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
src_thread_id = AddPartitionBase(ctx, src_thread_id);
|
||||
}
|
||||
|
||||
SetInBoundsFlag(inst, in_range);
|
||||
return SelectValue(ctx, in_range, value, src_thread_id);
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return value;
|
||||
}
|
||||
const IR::Value lane{inst->Arg(1).Resolve()};
|
||||
const IR::Value segment{inst->Arg(3).Resolve()};
|
||||
if (lane.IsImmediate() && segment.IsImmediate() && segment.U32() == 0) {
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, BroadcastLane(ctx, value, lane.U32() & 31),
|
||||
value);
|
||||
}
|
||||
const Id shuffled{ShuffleAbsolute(ctx, value, HostThreadId(ctx, src_thread_id))};
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, shuffled, value);
|
||||
}
|
||||
|
||||
Id EmitShuffleUp(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id delta{GuestLane(ctx, index)};
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id max_thread_id{GetMaxThreadId(ctx, thread_id, clamp, segmentation_mask)};
|
||||
Id src_thread_id{ctx.OpISub(ctx.U32[1], thread_id, index)};
|
||||
const Id src_thread_id{ctx.OpISub(ctx.U32[1], thread_id, delta)};
|
||||
const Id in_range{ctx.OpSGreaterThanEqual(ctx.U1, src_thread_id, max_thread_id)};
|
||||
|
||||
if (ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
src_thread_id = AddPartitionBase(ctx, src_thread_id);
|
||||
}
|
||||
|
||||
SetInBoundsFlag(inst, in_range);
|
||||
return SelectValue(ctx, in_range, value, src_thread_id);
|
||||
const Id shuffled{ShuffleRelative(ctx, value, delta, src_thread_id,
|
||||
spv::Op::OpGroupNonUniformShuffleUp)};
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, shuffled, value);
|
||||
}
|
||||
|
||||
Id EmitShuffleDown(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id delta{GuestLane(ctx, index)};
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id max_thread_id{GetMaxThreadId(ctx, thread_id, clamp, segmentation_mask)};
|
||||
Id src_thread_id{ctx.OpIAdd(ctx.U32[1], thread_id, index)};
|
||||
const Id src_thread_id{ctx.OpIAdd(ctx.U32[1], thread_id, delta)};
|
||||
const Id in_range{ctx.OpSLessThanEqual(ctx.U1, src_thread_id, max_thread_id)};
|
||||
|
||||
if (ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
src_thread_id = AddPartitionBase(ctx, src_thread_id);
|
||||
}
|
||||
|
||||
SetInBoundsFlag(inst, in_range);
|
||||
return SelectValue(ctx, in_range, value, src_thread_id);
|
||||
const Id shuffled{ShuffleRelative(ctx, value, delta, src_thread_id,
|
||||
spv::Op::OpGroupNonUniformShuffleDown)};
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, shuffled, value);
|
||||
}
|
||||
|
||||
Id EmitShuffleButterfly(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id mask{GuestLane(ctx, index)};
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id max_thread_id{GetMaxThreadId(ctx, thread_id, clamp, segmentation_mask)};
|
||||
Id src_thread_id{ctx.OpBitwiseXor(ctx.U32[1], thread_id, index)};
|
||||
const Id src_thread_id{ctx.OpBitwiseXor(ctx.U32[1], thread_id, mask)};
|
||||
const Id in_range{ctx.OpSLessThanEqual(ctx.U1, src_thread_id, max_thread_id)};
|
||||
|
||||
if (ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
src_thread_id = AddPartitionBase(ctx, src_thread_id);
|
||||
}
|
||||
|
||||
SetInBoundsFlag(inst, in_range);
|
||||
return SelectValue(ctx, in_range, value, src_thread_id);
|
||||
const Id shuffled{ctx.OpGroupNonUniformShuffleXor(ctx.U32[1], SubgroupScope(ctx), value, mask)};
|
||||
return ctx.OpSelect(ctx.U32[1], in_range, shuffled, value);
|
||||
}
|
||||
|
||||
Id EmitQuadBroadcast(EmitContext& ctx, Id value, Id lane) {
|
||||
@@ -267,10 +318,16 @@ Id EmitQuadBroadcast(EmitContext& ctx, Id value, Id lane) {
|
||||
const Id base{ctx.OpBitwiseAnd(ctx.U32[1], GetThreadId(ctx), ctx.Const(~3u))};
|
||||
const Id local_lane{ctx.OpBitwiseAnd(ctx.U32[1], lane, ctx.Const(3u))};
|
||||
const Id src_thread_id{ctx.OpBitwiseOr(ctx.U32[1], base, local_lane)};
|
||||
return ctx.OpGroupNonUniformShuffle(ctx.U32[1], SubgroupScope(ctx), value, src_thread_id);
|
||||
return ShuffleAbsolute(ctx, value, src_thread_id);
|
||||
}
|
||||
|
||||
Id EmitQuadSwap(EmitContext& ctx, Id value, Id direction) {
|
||||
if (ctx.profile.support_quad_shuffles) {
|
||||
const Id result{
|
||||
ctx.OpGroupNonUniformQuadBroadcast(ctx.U32[1], SubgroupScope(ctx), value, direction)};
|
||||
ctx.opcode_rewrites.emplace_back(result.value, spv::Op::OpGroupNonUniformQuadSwap);
|
||||
return result;
|
||||
}
|
||||
const Id xor_mask{ctx.OpIAdd(ctx.U32[1], direction, ctx.Const(1u))};
|
||||
return ctx.OpGroupNonUniformShuffleXor(ctx.U32[1], SubgroupScope(ctx), value, xor_mask);
|
||||
}
|
||||
|
||||
@@ -361,6 +361,7 @@ public:
|
||||
Id frag_depth{};
|
||||
|
||||
std::vector<Id> interfaces;
|
||||
std::vector<std::pair<u32, spv::Op>> opcode_rewrites;
|
||||
|
||||
Id load_const_func_u8{};
|
||||
Id load_const_func_u16{};
|
||||
|
||||
@@ -651,6 +651,29 @@ IR::Value GetThroughCast(IR::Value value, IR::Opcode expected_cast) {
|
||||
return value;
|
||||
}
|
||||
|
||||
u32 QuadButterflyMask(const IR::Inst& inst) {
|
||||
if (inst.GetOpcode() == IR::Opcode::QuadSwap) {
|
||||
const IR::Value direction{inst.Arg(1)};
|
||||
if (!direction.IsImmediate()) {
|
||||
return 0;
|
||||
}
|
||||
return direction.U32() + 1;
|
||||
}
|
||||
if (inst.GetOpcode() != IR::Opcode::ShuffleButterfly) {
|
||||
return 0;
|
||||
}
|
||||
const IR::Value index{inst.Arg(1)};
|
||||
const IR::Value clamp{inst.Arg(2)};
|
||||
const IR::Value segmentation_mask{inst.Arg(3)};
|
||||
if (!index.IsImmediate() || !clamp.IsImmediate() || !segmentation_mask.IsImmediate()) {
|
||||
return 0;
|
||||
}
|
||||
if (clamp.U32() != 3 || segmentation_mask.U32() != 28) {
|
||||
return 0;
|
||||
}
|
||||
return index.U32();
|
||||
}
|
||||
|
||||
void FoldFSwizzleAdd(IR::Block& block, IR::Inst& inst) {
|
||||
const IR::Value swizzle{inst.Arg(2)};
|
||||
if (!swizzle.IsImmediate()) {
|
||||
@@ -666,7 +689,8 @@ void FoldFSwizzleAdd(IR::Block& block, IR::Inst& inst) {
|
||||
return;
|
||||
}
|
||||
IR::Inst* const inst2{value_1.InstRecursive()};
|
||||
if (inst2->GetOpcode() != IR::Opcode::ShuffleButterfly) {
|
||||
const u32 lane_mask{QuadButterflyMask(*inst2)};
|
||||
if (lane_mask == 0) {
|
||||
return;
|
||||
}
|
||||
const IR::Value value_3{GetThroughCast(inst2->Arg(0).Resolve(), IR::Opcode::BitCastU32F32)};
|
||||
@@ -678,24 +702,15 @@ void FoldFSwizzleAdd(IR::Block& block, IR::Inst& inst) {
|
||||
return;
|
||||
}
|
||||
}
|
||||
const IR::Value index{inst2->Arg(1)};
|
||||
const IR::Value clamp{inst2->Arg(2)};
|
||||
const IR::Value segmentation_mask{inst2->Arg(3)};
|
||||
if (!index.IsImmediate() || !clamp.IsImmediate() || !segmentation_mask.IsImmediate()) {
|
||||
return;
|
||||
}
|
||||
if (clamp.U32() != 3 || segmentation_mask.U32() != 28) {
|
||||
return;
|
||||
}
|
||||
if (swizzle_value == 0x99) {
|
||||
// DPdxFine
|
||||
if (index.U32() == 1) {
|
||||
if (lane_mask == 1) {
|
||||
IR::IREmitter ir{block, IR::Block::InstructionList::s_iterator_to(inst)};
|
||||
inst.ReplaceUsesWith(ir.DPdxFine(IR::F32{inst.Arg(1)}));
|
||||
}
|
||||
} else if (swizzle_value == 0xA5) {
|
||||
// DPdyFine
|
||||
if (index.U32() == 2) {
|
||||
if (lane_mask == 2) {
|
||||
IR::IREmitter ir{block, IR::Block::InstructionList::s_iterator_to(inst)};
|
||||
inst.ReplaceUsesWith(ir.DPdyFine(IR::F32{inst.Arg(1)}));
|
||||
}
|
||||
@@ -709,6 +724,13 @@ bool FindGradient3DDerivatives(std::array<IR::Value, 3>& results, IR::Value coor
|
||||
const auto check_through_shuffle = [](IR::Value input, IR::Value& result) {
|
||||
const IR::Value value_1{GetThroughCast(input.Resolve(), IR::Opcode::BitCastF32U32)};
|
||||
IR::Inst* const inst2{value_1.InstRecursive()};
|
||||
if (inst2->GetOpcode() == IR::Opcode::QuadBroadcast) {
|
||||
if (!inst2->Arg(1).Resolve().IsImmediate()) {
|
||||
return false;
|
||||
}
|
||||
result = GetThroughCast(inst2->Arg(0).Resolve(), IR::Opcode::BitCastU32F32);
|
||||
return true;
|
||||
}
|
||||
if (inst2->GetOpcode() != IR::Opcode::ShuffleIndex) {
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -40,6 +40,7 @@ struct Profile {
|
||||
bool support_shader_quad_control{};
|
||||
bool support_quad_shuffles{};
|
||||
bool support_vote{};
|
||||
bool support_shuffle_relative{};
|
||||
u32 supported_subgroup_stages{0x7F};
|
||||
bool support_viewport_index_layer_non_geometry{};
|
||||
bool support_viewport_mask{};
|
||||
@@ -102,6 +103,8 @@ struct Profile {
|
||||
bool ignore_nan_fp_comparisons{};
|
||||
/// Some drivers have broken support for OpVectorExtractDynamic on subgroup mask inputs
|
||||
bool has_broken_spirv_subgroup_mask_vector_extract_dynamic{};
|
||||
bool has_broken_spirv_subgroup_shuffle{};
|
||||
u32 max_subgroup_size{};
|
||||
|
||||
u32 gl_max_compute_smem_size{};
|
||||
|
||||
|
||||
@@ -362,7 +362,7 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
const auto subgroup_stage_bit{[subgroup_stages](VkShaderStageFlags flag, Shader::Stage stage) {
|
||||
return (subgroup_stages & flag) != 0 ? (1u << static_cast<u32>(stage)) : 0u;
|
||||
}};
|
||||
const u32 supported_subgroup_stages{
|
||||
u32 supported_subgroup_stages{
|
||||
subgroup_stage_bit(VK_SHADER_STAGE_VERTEX_BIT, Shader::Stage::VertexA) |
|
||||
subgroup_stage_bit(VK_SHADER_STAGE_VERTEX_BIT, Shader::Stage::VertexB) |
|
||||
subgroup_stage_bit(VK_SHADER_STAGE_TESSELLATION_CONTROL_BIT,
|
||||
@@ -372,6 +372,9 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
subgroup_stage_bit(VK_SHADER_STAGE_GEOMETRY_BIT, Shader::Stage::Geometry) |
|
||||
subgroup_stage_bit(VK_SHADER_STAGE_FRAGMENT_BIT, Shader::Stage::Fragment) |
|
||||
subgroup_stage_bit(VK_SHADER_STAGE_COMPUTE_BIT, Shader::Stage::Compute)};
|
||||
if (driver_id == VK_DRIVER_ID_MESA_TURNIP) {
|
||||
supported_subgroup_stages &= ~(1u << static_cast<u32>(Shader::Stage::Geometry));
|
||||
}
|
||||
profile = Shader::Profile{
|
||||
.supported_spirv = device.SupportedSpirvVersion(),
|
||||
.unified_descriptor_binding = true,
|
||||
@@ -409,6 +412,8 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
.support_shader_quad_control = device.IsKhrShaderQuadControlSupported(),
|
||||
.support_quad_shuffles = device.IsSubgroupFeatureSupported(VK_SUBGROUP_FEATURE_QUAD_BIT),
|
||||
.support_vote = device.IsSubgroupFeatureSupported(VK_SUBGROUP_FEATURE_VOTE_BIT),
|
||||
.support_shuffle_relative =
|
||||
device.IsSubgroupFeatureSupported(VK_SUBGROUP_FEATURE_SHUFFLE_RELATIVE_BIT),
|
||||
.supported_subgroup_stages = supported_subgroup_stages,
|
||||
.support_viewport_index_layer_non_geometry =
|
||||
device.IsExtShaderViewportIndexLayerSupported(),
|
||||
@@ -444,13 +449,17 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_,
|
||||
driver_id == VK_DRIVER_ID_INTEL_OPEN_SOURCE_MESA,
|
||||
|
||||
.has_broken_spirv_clamp = driver_id == VK_DRIVER_ID_INTEL_PROPRIETARY_WINDOWS,
|
||||
.has_broken_spirv_position_input = driver_id == false,
|
||||
.has_broken_spirv_position_input = driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY,
|
||||
.has_broken_unsigned_image_offsets = false,
|
||||
.has_broken_signed_operations = false,
|
||||
.has_broken_fp16_float_controls = driver_id == VK_DRIVER_ID_NVIDIA_PROPRIETARY,
|
||||
.has_broken_fp32_denorm_flush = driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY,
|
||||
.ignore_nan_fp_comparisons = false,
|
||||
.has_broken_spirv_subgroup_mask_vector_extract_dynamic = false,
|
||||
.has_broken_spirv_subgroup_mask_vector_extract_dynamic =
|
||||
driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY &&
|
||||
device.GetDriverVersion() < VK_MAKE_VERSION(512, 672, 0),
|
||||
.has_broken_spirv_subgroup_shuffle = driver_id == VK_DRIVER_ID_QUALCOMM_PROPRIETARY,
|
||||
.max_subgroup_size = device.GetMaxSubgroupSize(),
|
||||
.has_broken_robust =
|
||||
device.IsNvidia() && device.GetNvidiaArch() <= NvidiaArchitecture::Arch_Pascal,
|
||||
.min_ssbo_alignment = device.GetStorageBufferAlignment(),
|
||||
|
||||
@@ -485,6 +485,11 @@ FN_MAX_LIMIT_LIST
|
||||
return properties.subgroup_properties.supportedStages;
|
||||
}
|
||||
|
||||
u32 GetMaxSubgroupSize() const {
|
||||
return (std::max)(properties.subgroup_properties.subgroupSize,
|
||||
properties.subgroup_size_control.maxSubgroupSize);
|
||||
}
|
||||
|
||||
/// Returns the maximum number of push descriptors.
|
||||
u32 MaxPushDescriptors() const {
|
||||
return properties.push_descriptor.maxPushDescriptors;
|
||||
|
||||
Reference in New Issue
Block a user