mirror of
https://git.eden-emu.dev/eden-emu/eden.git
synced 2026-08-31 10:36:56 +00:00
[vulkan, spirv] Fall back to scalar warp intrinsics in the geometry stage when unsupported
This commit is contained in:
@@ -13,6 +13,18 @@ Id SubgroupScope(EmitContext& ctx) {
|
||||
return ctx.Const(static_cast<u32>(spv::Scope::Subgroup));
|
||||
}
|
||||
|
||||
// Some mobile GPUs (e.g. Adreno/Turnip) only advertise subgroup ballot/shuffle support for the
|
||||
// fragment and compute stages (VkPhysicalDeviceSubgroupProperties::supportedStages), even though
|
||||
// they support these operations elsewhere. Guest shaders that use VOTE/SHFL in a geometry program
|
||||
// would otherwise emit GroupNonUniform* SPIR-V the driver never declared support for in that
|
||||
// stage. There is no barrier in the geometry stage, so a real cross-invocation emulation can't be
|
||||
// made correct; instead, treat the current invocation as if it were alone in its subgroup. This is
|
||||
// semantically wrong for guest code that relies on genuine cross-lane communication, but it is
|
||||
// well-defined, valid SPIR-V that doesn't depend on unsupported hardware capabilities.
|
||||
bool NeedsGeometrySubgroupFallback(EmitContext& ctx) {
|
||||
return ctx.stage == Stage::Geometry && !ctx.profile.support_subgroup_in_geometry_stage;
|
||||
}
|
||||
|
||||
bool StageSupportsSubgroups(EmitContext& ctx) {
|
||||
return ctx.profile.SupportsSubgroupStage(ctx.stage);
|
||||
}
|
||||
@@ -94,6 +106,9 @@ Id AddPartitionBase(EmitContext& ctx, Id thread_id) {
|
||||
} // Anonymous namespace
|
||||
|
||||
Id EmitLaneId(EmitContext& ctx) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
return ctx.u32_zero_value;
|
||||
}
|
||||
const Id id{GetThreadId(ctx)};
|
||||
if (!ctx.profile.warp_size_potentially_larger_than_guest) {
|
||||
return id;
|
||||
@@ -102,6 +117,9 @@ Id EmitLaneId(EmitContext& ctx) {
|
||||
}
|
||||
|
||||
Id EmitVoteAll(EmitContext& ctx, Id pred) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
return pred;
|
||||
}
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return pred;
|
||||
}
|
||||
@@ -118,6 +136,9 @@ Id EmitVoteAll(EmitContext& ctx, Id pred) {
|
||||
}
|
||||
|
||||
Id EmitVoteAny(EmitContext& ctx, Id pred) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
return pred;
|
||||
}
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return pred;
|
||||
}
|
||||
@@ -134,6 +155,9 @@ Id EmitVoteAny(EmitContext& ctx, Id pred) {
|
||||
}
|
||||
|
||||
Id EmitVoteEqual(EmitContext& ctx, Id pred) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
return ctx.true_value;
|
||||
}
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return ctx.true_value;
|
||||
}
|
||||
@@ -151,6 +175,16 @@ Id EmitVoteEqual(EmitContext& ctx, Id pred) {
|
||||
}
|
||||
|
||||
Id EmitSubgroupBallot(EmitContext& ctx, Id pred) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
// Reflect only this invocation's own predicate. There is no way to observe other
|
||||
// invocations' predicates without real subgroup hardware support in this stage, so this
|
||||
// is a best-effort approximation: it keeps any branch gated on "did anyone match" live
|
||||
// (rather than letting the SPIR-V optimizer prove it dead, which previously caused
|
||||
// indirect draws fed by this shader to see indexCount=instanceCount=0), but any downstream
|
||||
// math that assumes a real cross-lane population count (e.g. popcount-based compaction
|
||||
// offsets) will not be correct.
|
||||
return ctx.OpSelect(ctx.U32[1], pred, ctx.Const(1U), ctx.u32_zero_value);
|
||||
}
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return ctx.OpSelect(ctx.U32[1], pred, ctx.Const(1u), ctx.u32_zero_value);
|
||||
}
|
||||
@@ -162,6 +196,9 @@ Id EmitSubgroupBallot(EmitContext& ctx, Id pred) {
|
||||
}
|
||||
|
||||
Id EmitSubgroupEqMask(EmitContext& ctx) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
return ctx.Const(1U);
|
||||
}
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return ctx.Const(1u);
|
||||
}
|
||||
@@ -169,6 +206,9 @@ Id EmitSubgroupEqMask(EmitContext& ctx) {
|
||||
}
|
||||
|
||||
Id EmitSubgroupLtMask(EmitContext& ctx) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
return ctx.u32_zero_value;
|
||||
}
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return ctx.u32_zero_value;
|
||||
}
|
||||
@@ -176,6 +216,9 @@ Id EmitSubgroupLtMask(EmitContext& ctx) {
|
||||
}
|
||||
|
||||
Id EmitSubgroupLeMask(EmitContext& ctx) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
return ctx.Const(1U);
|
||||
}
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return ctx.Const(1u);
|
||||
}
|
||||
@@ -183,6 +226,9 @@ Id EmitSubgroupLeMask(EmitContext& ctx) {
|
||||
}
|
||||
|
||||
Id EmitSubgroupGtMask(EmitContext& ctx) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
return ctx.u32_zero_value;
|
||||
}
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return ctx.u32_zero_value;
|
||||
}
|
||||
@@ -190,6 +236,9 @@ Id EmitSubgroupGtMask(EmitContext& ctx) {
|
||||
}
|
||||
|
||||
Id EmitSubgroupGeMask(EmitContext& ctx) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
return ctx.Const(1U);
|
||||
}
|
||||
if (!StageSupportsSubgroups(ctx)) {
|
||||
return ctx.Const(1u);
|
||||
}
|
||||
@@ -198,6 +247,10 @@ Id EmitSubgroupGeMask(EmitContext& ctx) {
|
||||
|
||||
Id EmitShuffleIndex(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id not_seg_mask{ctx.OpNot(ctx.U32[1], segmentation_mask)};
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id min_thread_id{ComputeMinThreadId(ctx, thread_id, segmentation_mask)};
|
||||
@@ -217,6 +270,10 @@ Id EmitShuffleIndex(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id cla
|
||||
|
||||
Id EmitShuffleUp(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id max_thread_id{GetMaxThreadId(ctx, thread_id, clamp, segmentation_mask)};
|
||||
Id src_thread_id{ctx.OpISub(ctx.U32[1], thread_id, index)};
|
||||
@@ -232,6 +289,10 @@ Id EmitShuffleUp(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
|
||||
Id EmitShuffleDown(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id max_thread_id{GetMaxThreadId(ctx, thread_id, clamp, segmentation_mask)};
|
||||
Id src_thread_id{ctx.OpIAdd(ctx.U32[1], thread_id, index)};
|
||||
@@ -247,6 +308,10 @@ Id EmitShuffleDown(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clam
|
||||
|
||||
Id EmitShuffleButterfly(EmitContext& ctx, IR::Inst* inst, Id value, Id index, Id clamp,
|
||||
Id segmentation_mask) {
|
||||
if (NeedsGeometrySubgroupFallback(ctx)) {
|
||||
SetInBoundsFlag(inst, ctx.false_value);
|
||||
return value;
|
||||
}
|
||||
const Id thread_id{EmitLaneId(ctx)};
|
||||
const Id max_thread_id{GetMaxThreadId(ctx, thread_id, clamp, segmentation_mask)};
|
||||
Id src_thread_id{ctx.OpBitwiseXor(ctx.U32[1], thread_id, index)};
|
||||
|
||||
Reference in New Issue
Block a user