diff --git a/src/shader_recompiler/host_translate_info.h b/src/shader_recompiler/host_translate_info.h index 93ce881c2d..fbedbb098b 100644 --- a/src/shader_recompiler/host_translate_info.h +++ b/src/shader_recompiler/host_translate_info.h @@ -38,6 +38,8 @@ struct HostTranslateInfo { ///< passthrough shaders bool support_conditional_barrier{}; ///< True when the device supports barriers in conditional ///< control flow + bool support_attribute_texture_handle{}; ///< True when the host binds texture handles that + ///< arrive as vertex attributes void ApplyDescriptorLimitPolicy() noexcept { if (min_ssbo_alignment == 0) { diff --git a/src/shader_recompiler/ir_opt/texture_pass.cpp b/src/shader_recompiler/ir_opt/texture_pass.cpp index fe34a67555..e18365121b 100644 --- a/src/shader_recompiler/ir_opt/texture_pass.cpp +++ b/src/shader_recompiler/ir_opt/texture_pass.cpp @@ -268,6 +268,10 @@ static inline u32 ReadCbufCached(Environment& env, u32 index, u32 offset) { return v; } +static inline bool IsAttributeHandle(const ConstBufferAddr& cbuf) { + return cbuf.index == ATTRIBUTE_HANDLE_CBUF_INDEX; +} + static inline u32 GetTextureHandleCached(Environment& env, const ConstBufferAddr& cbuf) { // Must all be uniquely different variables // If has secondary, then it will be cbuf.secondary_{index|offset}, else its 0. @@ -287,14 +291,23 @@ static inline u32 GetTextureHandleCached(Environment& env, const ConstBufferAddr // Cached variants of existing helpers static inline TextureType ReadTextureTypeCached(Environment& env, const ConstBufferAddr& cbuf) { + if (IsAttributeHandle(cbuf)) { + return TextureType::Color2D; + } return env.ReadTextureType(GetTextureHandleCached(env, cbuf)); } static inline TexturePixelFormat ReadTexturePixelFormatCached(Environment& env, const ConstBufferAddr& cbuf) { + if (IsAttributeHandle(cbuf)) { + return TexturePixelFormat::A8B8G8R8_UNORM; + } return env.ReadTexturePixelFormat(GetTextureHandleCached(env, cbuf)); } static inline bool IsTexturePixelFormatIntegerCached(Environment& env, const ConstBufferAddr& cbuf) { + if (IsAttributeHandle(cbuf)) { + return false; + } return env.IsTexturePixelFormatInteger(GetTextureHandleCached(env, cbuf)); } @@ -466,15 +479,43 @@ std::optional TryGetConstBuffer(const IR::Inst* inst, Environme }; } +bool IsAttributeSourcedHandle(const IR::Value& handle) { + if (handle.IsImmediate()) { + return false; + } + const IR::Inst* inst{handle.InstRecursive()}; + if (inst->GetOpcode() == IR::Opcode::BitCastU32F32 && !inst->Arg(0).IsImmediate()) { + inst = inst->Arg(0).InstRecursive(); + } + return inst->GetOpcode() == IR::Opcode::GetAttribute || + inst->GetOpcode() == IR::Opcode::GetAttributeU32; +} + TextureInst MakeInst(Environment& env, IR::Block* block, IR::Inst& inst, const HostTranslateInfo& host_info) { ConstBufferAddr addr; if (IsBindless(inst)) { const std::optional track_addr{TrackCached(inst.Arg(0), env, host_info)}; - if (!track_addr) { - throw NotImplementedException("Failed to track bindless texture constant buffer"); - } else { + if (track_addr) { addr = *track_addr; + } else if (host_info.support_attribute_texture_handle && + IsAttributeSourcedHandle(inst.Arg(0))) { + // The handle travels with the vertices (sprite batches pick the texture per quad), so + // there is no constant buffer to read it from. The host binds the handle of each run + // of vertices that share one. + addr = ConstBufferAddr{ + .index = ATTRIBUTE_HANDLE_CBUF_INDEX, + .offset = 0, + .shift_left = 0, + .secondary_index = 0, + .secondary_offset = 0, + .secondary_shift_left = 0, + .dynamic_offset = {}, + .count = 1, + .has_secondary = false, + }; + } else { + throw NotImplementedException("Failed to track bindless texture constant buffer"); } } else { addr = ConstBufferAddr{ diff --git a/src/shader_recompiler/shader_info.h b/src/shader_recompiler/shader_info.h index 71309e040e..d60be1238d 100644 --- a/src/shader_recompiler/shader_info.h +++ b/src/shader_recompiler/shader_info.h @@ -38,6 +38,10 @@ enum class TextureType : u32 { }; constexpr u32 NUM_TEXTURE_TYPES = 9; +/// Constant buffer index marking a texture handle that the shader receives as a vertex attribute +/// instead of reading it from a constant buffer. The host supplies the handle of each draw run. +constexpr u32 ATTRIBUTE_HANDLE_CBUF_INDEX = 0xFFFF; + enum class TexturePixelFormat { A8B8G8R8_UNORM, A8B8G8R8_SNORM, diff --git a/src/video_core/renderer_vulkan/vk_graphics_pipeline.cpp b/src/video_core/renderer_vulkan/vk_graphics_pipeline.cpp index 8b5b0bc9c5..c9e4f939b1 100644 --- a/src/video_core/renderer_vulkan/vk_graphics_pipeline.cpp +++ b/src/video_core/renderer_vulkan/vk_graphics_pipeline.cpp @@ -276,6 +276,9 @@ GraphicsPipeline::GraphicsPipeline( num_textures += Shader::NumDescriptors(info->texture_descriptors); num_image_elements += Shader::NumDescriptors(info->texture_descriptors); num_image_elements += Shader::NumDescriptors(info->image_descriptors); + for (const auto& desc : info->texture_descriptors) { + uses_attribute_handle |= desc.cbuf_index == Shader::ATTRIBUTE_HANDLE_CBUF_INDEX; + } num_descriptor_entries += NumDescriptorEntries(*info); } fragment_has_color0_output = stage_infos[NUM_STAGES - 1].stores_frag_color[0]; @@ -375,6 +378,9 @@ bool GraphicsPipeline::ConfigureImpl(bool is_indexed) { } const auto& cbufs{maxwell3d->state.shader_stages[stage].const_buffers}; const auto read_handle{[&](const auto& desc, u32 index) { + if (desc.cbuf_index == Shader::ATTRIBUTE_HANDLE_CBUF_INDEX) { + return TexturePair(attribute_handle, via_header_index); + } ASSERT(cbufs[desc.cbuf_index].enabled); const u32 index_offset{index << desc.size_shift}; const u32 offset{desc.cbuf_offset + index_offset}; diff --git a/src/video_core/renderer_vulkan/vk_graphics_pipeline.h b/src/video_core/renderer_vulkan/vk_graphics_pipeline.h index b1ac5a3fcc..a6edab8bc0 100644 --- a/src/video_core/renderer_vulkan/vk_graphics_pipeline.h +++ b/src/video_core/renderer_vulkan/vk_graphics_pipeline.h @@ -110,6 +110,16 @@ public: return configure_func(this, is_indexed); } + /// True when a shader samples a texture whose handle arrives as a vertex attribute. + [[nodiscard]] bool UsesAttributeHandle() const noexcept { + return uses_attribute_handle; + } + + /// Handle bound to those textures by the next Configure. + void SetAttributeHandle(u32 handle) noexcept { + attribute_handle = handle; + } + [[nodiscard]] GraphicsPipeline* Next(const GraphicsPipelineCacheKey& current_key) noexcept { if (key == current_key) { return this; @@ -168,6 +178,8 @@ private: u32 num_descriptor_entries{}; size_t num_image_elements{}; u32 num_textures{}; + u32 attribute_handle{}; + bool uses_attribute_handle{}; bool fragment_has_color0_output{}; vk::DescriptorSetLayout descriptor_set_layout; diff --git a/src/video_core/renderer_vulkan/vk_pipeline_cache.cpp b/src/video_core/renderer_vulkan/vk_pipeline_cache.cpp index 642b9fa223..db54ad11cc 100644 --- a/src/video_core/renderer_vulkan/vk_pipeline_cache.cpp +++ b/src/video_core/renderer_vulkan/vk_pipeline_cache.cpp @@ -479,6 +479,7 @@ PipelineCache::PipelineCache(Tegra::MaxwellDeviceMemoryManager& device_memory_, .support_viewport_index_layer = device.IsExtShaderViewportIndexLayerSupported(), .support_geometry_shader_passthrough = device.IsNvGeometryShaderPassthroughSupported(), .support_conditional_barrier = device.SupportsConditionalBarriers(), + .support_attribute_texture_handle = true, }; host_info.ApplyDescriptorLimitPolicy(); diff --git a/src/video_core/renderer_vulkan/vk_rasterizer.cpp b/src/video_core/renderer_vulkan/vk_rasterizer.cpp index f43bea1220..c4008486e4 100644 --- a/src/video_core/renderer_vulkan/vk_rasterizer.cpp +++ b/src/video_core/renderer_vulkan/vk_rasterizer.cpp @@ -248,6 +248,15 @@ void RasterizerVulkan::PrepareDraw(bool is_indexed, Func&& draw_func) { std::scoped_lock lock{buffer_cache.mutex, texture_cache.mutex}; // update engine as channel may be different. pipeline->SetEngine(maxwell3d, gpu_memory); + attribute_handle_pipeline = nullptr; + attribute_handle_runs.clear(); + if (split_attribute_handles && pipeline->UsesAttributeHandle()) { + CollectAttributeHandleRuns(is_indexed); + if (!attribute_handle_runs.empty()) { + attribute_handle_pipeline = pipeline; + pipeline->SetAttributeHandle(attribute_handle_runs.front().handle); + } + } if (!pipeline->Configure(is_indexed)) return; @@ -259,22 +268,110 @@ void RasterizerVulkan::PrepareDraw(bool is_indexed, Func&& draw_func) { draw_func(); } +void RasterizerVulkan::CollectAttributeHandleRuns(bool is_indexed) { + using VertexAttribute = Maxwell::VertexAttribute; + const auto& regs = maxwell3d->regs; + const auto& draw_state = maxwell3d->draw_manager.draw_state; + const u32 total = is_indexed ? draw_state.index_buffer.count : draw_state.vertex_buffer.count; + if (total == 0) { + return; + } + // The handle is an integer attribute; sprite batches repeat it on every vertex of a quad. + const VertexAttribute* source{}; + for (const auto& attr : regs.vertex_attrib_format) { + if (attr.constant || attr.size == VertexAttribute::Size::Invalid) { + continue; + } + if (attr.type == VertexAttribute::Type::UInt || attr.type == VertexAttribute::Type::SInt) { + source = &attr; + break; + } + } + const auto& stream = regs.vertex_streams[source ? source->buffer.Value() : 0]; + if (!source || !stream.IsEnabled()) { + attribute_handle_runs.push_back({0, total, 0}); + return; + } + // Only plain triangle lists are split; anything else keeps the handle of its first vertex. + const u32 primitive_size = + draw_state.topology == Maxwell::PrimitiveTopology::Triangles ? 3 : total; + const GPUVAddr base_address = stream.Address() + source->offset; + const auto read_handle = [&](u32 position) { + u32 vertex; + if (is_indexed) { + const auto& index_buffer = draw_state.index_buffer; + const GPUVAddr address = + index_buffer.IndexStart() + size_t{position} * index_buffer.FormatSizeInBytes(); + u32 index{}; + gpu_memory->ReadBlockUnsafe(address, &index, index_buffer.FormatSizeInBytes()); + vertex = static_cast(draw_state.base_index + index); + } else { + vertex = draw_state.vertex_buffer.first + position; + } + return gpu_memory->Read(base_address + static_cast(vertex) * stream.stride); + }; + for (u32 position = 0; position + primitive_size <= total; position += primitive_size) { + const u32 handle = read_handle(position); + if (!attribute_handle_runs.empty() && attribute_handle_runs.back().handle == handle) { + attribute_handle_runs.back().count += primitive_size; + } else { + attribute_handle_runs.push_back({position, primitive_size, handle}); + } + } + if (attribute_handle_runs.empty()) { + attribute_handle_runs.push_back({0, total, read_handle(0)}); + } else { + const u32 covered = attribute_handle_runs.back().first + attribute_handle_runs.back().count; + if (covered < total) { + attribute_handle_runs.back().count += total - covered; + } + } +} + void RasterizerVulkan::Draw(bool is_indexed, u32 instance_count) { + split_attribute_handles = true; + SCOPE_EXIT { + split_attribute_handles = false; + }; PrepareDraw(is_indexed, [this, is_indexed, instance_count] { const auto& draw_state = maxwell3d->draw_manager.draw_state; const u32 num_instances{instance_count}; const DrawParams draw_params{MakeDrawParams(draw_state, num_instances, is_indexed)}; - scheduler.Record([draw_params](vk::CommandBuffer cmdbuf) { - if (draw_params.is_indexed) { - cmdbuf.DrawIndexed(draw_params.num_vertices, draw_params.num_instances, - draw_params.first_index, draw_params.base_vertex, - draw_params.base_instance); - } else { - cmdbuf.Draw(draw_params.num_vertices, draw_params.num_instances, - draw_params.base_vertex, draw_params.base_instance); + const auto record_draw = [this](const DrawParams& params) { + scheduler.Record([params](vk::CommandBuffer cmdbuf) { + if (params.is_indexed) { + cmdbuf.DrawIndexed(params.num_vertices, params.num_instances, + params.first_index, params.base_vertex, + params.base_instance); + } else { + cmdbuf.Draw(params.num_vertices, params.num_instances, params.base_vertex, + params.base_instance); + } + }); + }; + if (attribute_handle_pipeline && attribute_handle_runs.size() > 1) { + // The first run is already configured, each further one binds its own texture. + for (size_t i = 0; i < attribute_handle_runs.size(); ++i) { + const AttributeHandleRun& run = attribute_handle_runs[i]; + if (i != 0) { + attribute_handle_pipeline->SetAttributeHandle(run.handle); + if (!attribute_handle_pipeline->Configure(is_indexed)) { + continue; + } + } + DrawParams run_params = draw_params; + run_params.num_vertices = run.count; + if (is_indexed) { + run_params.first_index += run.first; + } else { + run_params.base_vertex += run.first; + } + record_draw(run_params); } - }); + } else { + record_draw(draw_params); + } // Log draw call if (GPU::Logging::IsActive() && diff --git a/src/video_core/renderer_vulkan/vk_rasterizer.h b/src/video_core/renderer_vulkan/vk_rasterizer.h index 7470aa4f14..fcef3b9149 100644 --- a/src/video_core/renderer_vulkan/vk_rasterizer.h +++ b/src/video_core/renderer_vulkan/vk_rasterizer.h @@ -157,6 +157,20 @@ private: template void PrepareDraw(bool is_indexed, Func&&); + /// A stretch of a draw whose vertices all carry the same texture handle. + struct AttributeHandleRun { + u32 first; + u32 count; + u32 handle; + }; + + /// Splits the current draw wherever the texture handle carried in the vertex data changes. + void CollectAttributeHandleRuns(bool is_indexed); + + std::vector attribute_handle_runs; + GraphicsPipeline* attribute_handle_pipeline{}; + bool split_attribute_handles{}; + void FlushWork(); void UpdateDynamicStates();