From ccf8fb66f57a0d9d1af46bdeafc065a55b829eb7 Mon Sep 17 00:00:00 2001 From: "Herman S." <429230+has207@users.noreply.github.com> Date: Tue, 23 Sep 2025 21:06:51 +0900 Subject: [PATCH] [Vulkan] Add thread pool for async shader compilation Uses priority queue with higher priority on shaders with visible render targets --- .../gpu/vulkan/vulkan_command_processor.cc | 29 +- src/xenia/gpu/vulkan/vulkan_pipeline_cache.cc | 264 +++++++++++++++--- src/xenia/gpu/vulkan/vulkan_pipeline_cache.h | 91 ++++-- 3 files changed, 324 insertions(+), 60 deletions(-) diff --git a/src/xenia/gpu/vulkan/vulkan_command_processor.cc b/src/xenia/gpu/vulkan/vulkan_command_processor.cc index e81536c68..875fc8ed1 100644 --- a/src/xenia/gpu/vulkan/vulkan_command_processor.cc +++ b/src/xenia/gpu/vulkan/vulkan_command_processor.cc @@ -2446,17 +2446,27 @@ bool VulkanCommandProcessor::IssueDraw(xenos::PrimitiveType prim_type, // Create the pipeline (for this, need the render pass from the render target // cache), translating the shaders - doing this now to obtain the used // textures. - VkPipeline pipeline; - const VulkanPipelineCache::PipelineLayoutProvider* pipeline_layout_provider; + VulkanPipelineCache::Pipeline* pipeline; if (!pipeline_cache_->ConfigurePipeline( vertex_shader_translation, pixel_shader_translation, primitive_processing_result, normalized_depth_control, normalized_color_mask, - render_target_cache_->last_update_render_pass_key(), pipeline, - pipeline_layout_provider)) { + render_target_cache_->last_update_render_pass_key(), &pipeline)) { return false; } + VkPipeline current_pipeline = + pipeline->pipeline.load(std::memory_order_acquire); + if (current_pipeline == VK_NULL_HANDLE) { + // Pipeline is not ready yet - wait for it to be created. + pipeline_cache_->EndSubmission(); + current_pipeline = pipeline->pipeline.load(std::memory_order_acquire); + if (current_pipeline == VK_NULL_HANDLE) { + // Still not ready - something is wrong. + return false; + } + } + // Update the textures before most other work in the submission because // samplers depend on this (and in case of sampler overflow in a submission, // submissions must be split) - may perform dispatches and copying. @@ -2470,14 +2480,17 @@ bool VulkanCommandProcessor::IssueDraw(xenos::PrimitiveType prim_type, // Update the graphics pipeline, and if the new graphics pipeline has a // different layout, invalidate incompatible descriptor sets before updating // current_guest_graphics_pipeline_layout_. - if (current_guest_graphics_pipeline_ != pipeline) { + // The pipeline may be not ready yet if created asynchronously. + // EndSubmission must be called before submitting the command buffer to + // await its creation. + if (current_guest_graphics_pipeline_ != current_pipeline) { deferred_command_buffer_.CmdVkBindPipeline(VK_PIPELINE_BIND_POINT_GRAPHICS, - pipeline); - current_guest_graphics_pipeline_ = pipeline; + current_pipeline); + current_guest_graphics_pipeline_ = current_pipeline; current_external_graphics_pipeline_ = VK_NULL_HANDLE; } auto pipeline_layout = - static_cast(pipeline_layout_provider); + static_cast(pipeline->pipeline_layout); if (current_guest_graphics_pipeline_layout_ != pipeline_layout) { if (current_guest_graphics_pipeline_layout_) { // Keep descriptor set layouts for which the new pipeline layout is diff --git a/src/xenia/gpu/vulkan/vulkan_pipeline_cache.cc b/src/xenia/gpu/vulkan/vulkan_pipeline_cache.cc index 4de93313f..78dcf3f05 100644 --- a/src/xenia/gpu/vulkan/vulkan_pipeline_cache.cc +++ b/src/xenia/gpu/vulkan/vulkan_pipeline_cache.cc @@ -45,6 +45,14 @@ namespace shaders { #include "xenia/gpu/shaders/bytecode/vulkan_spirv/tessellation_adaptive_vs.h" #include "xenia/gpu/shaders/bytecode/vulkan_spirv/tessellation_indexed_vs.h" } // namespace shaders + +DEFINE_int32( + vulkan_pipeline_creation_threads, -1, + "Number of threads used for graphics pipeline creation. -1 to calculate " + "automatically (75% of logical CPU cores), a positive number to specify " + "the number of threads explicitly (up to the number of logical CPU cores), " + "0 to disable multithreaded pipeline creation.", + "Vulkan"); namespace xe { namespace gpu { namespace vulkan { @@ -167,10 +175,52 @@ bool VulkanPipelineCache::Initialize() { vk_pipeline_cache_ = VK_NULL_HANDLE; } + uint32_t logical_processor_count = xe::threading::logical_processor_count(); + if (!logical_processor_count) { + // Pick some reasonable amount if couldn't determine the number of cores. + logical_processor_count = 6; + } + creation_completion_event_ = + xe::threading::Event::CreateManualResetEvent(true); + assert_not_null(creation_completion_event_); + if (cvars::vulkan_pipeline_creation_threads != 0) { + size_t creation_thread_count; + if (cvars::vulkan_pipeline_creation_threads < 0) { + creation_thread_count = + std::max(logical_processor_count * 3 / 4, uint32_t(1)); + } else { + creation_thread_count = + std::min(uint32_t(cvars::vulkan_pipeline_creation_threads), + logical_processor_count); + } + creation_threads_shutdown_ = false; + for (size_t i = 0; i < creation_thread_count; ++i) { + std::unique_ptr creation_thread = + xe::threading::Thread::Create({}, [this]() { CreationThread(); }); + assert_not_null(creation_thread); + creation_thread->set_name("Vulkan Pipelines"); + creation_threads_.push_back(std::move(creation_thread)); + } + } return true; } void VulkanPipelineCache::Shutdown() { + // Shut down all threads, before destroying the pipelines since they may be + // creating them. + if (!creation_threads_.empty()) { + { + std::lock_guard lock(creation_request_lock_); + creation_threads_shutdown_ = true; + } + creation_request_cond_.notify_all(); + for (size_t i = 0; i < creation_threads_.size(); ++i) { + xe::threading::Wait(creation_threads_[i].get(), false); + } + creation_threads_.clear(); + } + creation_completion_event_.reset(); + const ui::vulkan::VulkanDevice* const vulkan_device = command_processor_.GetVulkanDevice(); const ui::vulkan::VulkanDevice::Functions& dfn = vulkan_device->functions(); @@ -423,17 +473,11 @@ bool VulkanPipelineCache::ConfigurePipeline( reg::RB_DEPTHCONTROL normalized_depth_control, uint32_t normalized_color_mask, VulkanRenderTargetCache::RenderPassKey render_pass_key, - VkPipeline& pipeline_out, - const PipelineLayoutProvider*& pipeline_layout_out) { + VulkanPipelineCache::Pipeline** pipeline_out) { #if XE_GPU_FINE_GRAINED_DRAW_SCOPES SCOPE_profile_cpu_f("gpu"); #endif // XE_GPU_FINE_GRAINED_DRAW_SCOPES - // Ensure shaders are translated - needed now for GetCurrentStateDescription. - if (!EnsureShadersTranslated(vertex_shader, pixel_shader)) { - return false; - } - PipelineDescription description; if (!GetCurrentStateDescription( vertex_shader, pixel_shader, primitive_processing_result, @@ -442,15 +486,13 @@ bool VulkanPipelineCache::ConfigurePipeline( return false; } if (last_pipeline_ && last_pipeline_->first == description) { - pipeline_out = last_pipeline_->second.pipeline; - pipeline_layout_out = last_pipeline_->second.pipeline_layout; + *pipeline_out = &last_pipeline_->second; return true; } auto it = pipelines_.find(description); if (it != pipelines_.end()) { last_pipeline_ = &*it; - pipeline_out = it->second.pipeline; - pipeline_layout_out = it->second.pipeline_layout; + *pipeline_out = &it->second; return true; } @@ -476,19 +518,22 @@ bool VulkanPipelineCache::ConfigurePipeline( if (!pipeline_layout) { return false; } + VkShaderModule geometry_shader = VK_NULL_HANDLE; - GeometryShaderKey geometry_shader_key; - if (GetGeometryShaderKey( - description.geometry_shader, - SpirvShaderTranslator::Modification(vertex_shader->modification()), - SpirvShaderTranslator::Modification( - pixel_shader ? pixel_shader->modification() : 0), - geometry_shader_key)) { + if (description.geometry_shader != PipelineGeometryShader::kNone) { + GeometryShaderKey geometry_shader_key; + GetGeometryShaderKey( + description.geometry_shader, + SpirvShaderTranslator::Modification(vertex_shader->modification()), + SpirvShaderTranslator::Modification( + pixel_shader ? pixel_shader->modification() : 0), + geometry_shader_key); geometry_shader = GetGeometryShader(geometry_shader_key); if (geometry_shader == VK_NULL_HANDLE) { return false; } } + VkRenderPass render_pass = render_target_cache_.GetPath() == RenderTargetCache::Path::kPixelShaderInterlock @@ -498,9 +543,10 @@ bool VulkanPipelineCache::ConfigurePipeline( if (render_pass == VK_NULL_HANDLE) { return false; } - PipelineCreationArguments creation_arguments; - auto& pipeline = + + auto& pipeline_pair = *pipelines_.emplace(description, Pipeline(pipeline_layout)).first; + // Get tessellation shaders if needed. VkShaderModule tessellation_vertex_shader = VK_NULL_HANDLE; VkShaderModule tessellation_control_shader = VK_NULL_HANDLE; @@ -523,19 +569,169 @@ bool VulkanPipelineCache::ConfigurePipeline( } } - creation_arguments.pipeline = &pipeline; - creation_arguments.vertex_shader = vertex_shader; - creation_arguments.pixel_shader = pixel_shader; - creation_arguments.geometry_shader = geometry_shader; - creation_arguments.tessellation_vertex_shader = tessellation_vertex_shader; - creation_arguments.tessellation_control_shader = tessellation_control_shader; - creation_arguments.render_pass = render_pass; - if (!EnsurePipelineCreated(creation_arguments)) { + // Pipeline hot-swap: When async mode is enabled, we have creation threads and + // a pixel shader, create a placeholder pipeline immediately (fast compile) + // and queue the real pipeline creation in the background. This reduces + // stutter from pipeline compilation. + bool use_async = cvars::async_shader_compilation && + !creation_threads_.empty() && pixel_shader && + placeholder_pixel_shader_ != VK_NULL_HANDLE; + + if (use_async) { + // Create placeholder pipeline immediately (uses simple PS, fast compile). + // Set is_placeholder BEFORE creating the pipeline to avoid race condition + // with the creation thread checking this flag. + pipeline_pair.second.is_placeholder.store(true, std::memory_order_release); + + PipelineCreationArguments placeholder_args; + placeholder_args.pipeline = &pipeline_pair; + placeholder_args.vertex_shader = vertex_shader; + placeholder_args.pixel_shader = nullptr; // Will use placeholder PS + placeholder_args.geometry_shader = geometry_shader; + placeholder_args.tessellation_vertex_shader = tessellation_vertex_shader; + placeholder_args.tessellation_control_shader = tessellation_control_shader; + placeholder_args.render_pass = render_pass; + + if (EnsurePipelineCreatedWithPlaceholder(placeholder_args)) { + // Queue real pipeline creation in background. + // Calculate priority based on whether shader writes to visible RTs. + uint8_t priority = 0; + if (pixel_shader) { + // Get bound RT mask from normalized_color_mask (4 bits per RT). + uint32_t bound_rts = (((normalized_color_mask >> 0) & 0xF) ? 1 : 0) | + (((normalized_color_mask >> 4) & 0xF) ? 2 : 0) | + (((normalized_color_mask >> 8) & 0xF) ? 4 : 0) | + (((normalized_color_mask >> 12) & 0xF) ? 8 : 0); + uint32_t shader_writes = pixel_shader->shader().writes_color_targets(); + if (bound_rts & shader_writes) { + // Writes to at least one visible RT - high priority. + priority = 2; + // Extra priority if writing to RT0 (usually main color buffer). + if ((bound_rts & shader_writes) & 1) { + priority = 3; + } + } else if (pixel_shader->shader().writes_depth()) { + // Depth-only - medium priority. + priority = 1; + } + // else: writes to unbound RTs only - lowest priority (0). + } + + { + std::lock_guard lock(creation_request_lock_); + PipelineCreationArguments creation_arguments; + creation_arguments.pipeline = &pipeline_pair; + creation_arguments.vertex_shader = vertex_shader; + creation_arguments.pixel_shader = pixel_shader; + creation_arguments.geometry_shader = geometry_shader; + creation_arguments.tessellation_vertex_shader = + tessellation_vertex_shader; + creation_arguments.tessellation_control_shader = + tessellation_control_shader; + creation_arguments.render_pass = render_pass; + creation_arguments.priority = priority; + creation_queue_.push(creation_arguments); + } + creation_request_cond_.notify_one(); + } else { + // Placeholder creation failed, fall back to sync creation. + // Reset the flag we set earlier. + pipeline_pair.second.is_placeholder.store(false, + std::memory_order_release); + use_async = false; + } + } + + if (!use_async) { + // Sync mode or no creation threads: create synchronously. + PipelineCreationArguments creation_arguments; + creation_arguments.pipeline = &pipeline_pair; + creation_arguments.vertex_shader = vertex_shader; + creation_arguments.pixel_shader = pixel_shader; + creation_arguments.geometry_shader = geometry_shader; + creation_arguments.tessellation_vertex_shader = tessellation_vertex_shader; + creation_arguments.tessellation_control_shader = tessellation_control_shader; + creation_arguments.render_pass = render_pass; + if (!EnsurePipelineCreated(creation_arguments)) { + return false; + } + } + + last_pipeline_ = &pipeline_pair; + *pipeline_out = &pipeline_pair.second; + return true; +} + +void VulkanPipelineCache::EndSubmission() { + if (creation_threads_.empty()) { + return; + } + // Await creation of all queued pipelines. + bool await_creation_completion_event; + { + std::lock_guard lock(creation_request_lock_); + // Assuming the creation queue is already empty (because the processor + // thread also worked on creating the leftover pipelines), so only check + // if there are threads with pipelines currently being created. + await_creation_completion_event = + !creation_queue_.empty() || creation_threads_busy_ != 0; + if (await_creation_completion_event) { + creation_completion_event_->Reset(); + creation_completion_set_event_.store(true, std::memory_order_release); + } + } + if (await_creation_completion_event) { + creation_request_cond_.notify_one(); + xe::threading::Wait(creation_completion_event_.get(), false); + } +} + +bool VulkanPipelineCache::IsCreatingPipelines() { + if (creation_threads_.empty()) { return false; } - pipeline_out = pipeline.second.pipeline; - pipeline_layout_out = pipeline_layout; - return true; + std::lock_guard lock(creation_request_lock_); + return !creation_queue_.empty() || creation_threads_busy_ != 0; +} + +void VulkanPipelineCache::CreationThread() { + for (;;) { + PipelineCreationArguments creation_arguments; + { + std::unique_lock lock(creation_request_lock_); + creation_request_cond_.wait(lock, [this]() { + return !creation_queue_.empty() || creation_threads_shutdown_; + }); + if (creation_threads_shutdown_) { + break; + } + creation_arguments = creation_queue_.top(); + creation_queue_.pop(); + ++creation_threads_busy_; + } + + if (!EnsureShadersTranslated(creation_arguments.vertex_shader, + creation_arguments.pixel_shader)) { + // Mark pipeline as failed by keeping it as VK_NULL_HANDLE. + // The pipeline will remain VK_NULL_HANDLE, indicating failure. + XELOGE("Failed to translate shaders for pipeline creation"); + } else { + if (!EnsurePipelineCreated(creation_arguments)) { + // Pipeline creation failed - it will remain VK_NULL_HANDLE. + XELOGE("Failed to create Vulkan pipeline"); + } + } + + { + std::lock_guard lock(creation_request_lock_); + --creation_threads_busy_; + if (creation_completion_set_event_.load(std::memory_order_acquire) && + creation_threads_busy_ == 0 && creation_queue_.empty()) { + creation_completion_set_event_.store(false, std::memory_order_release); + creation_completion_event_->Set(); + } + } + } } bool VulkanPipelineCache::TranslateAnalyzedShader( @@ -2057,7 +2253,8 @@ VkShaderModule VulkanPipelineCache::GetTessellationVertexShader( bool VulkanPipelineCache::EnsurePipelineCreated( const PipelineCreationArguments& creation_arguments) { - if (creation_arguments.pipeline->second.pipeline != VK_NULL_HANDLE) { + if (creation_arguments.pipeline->second.pipeline.load( + std::memory_order_acquire) != VK_NULL_HANDLE) { return true; } @@ -2564,7 +2761,8 @@ bool VulkanPipelineCache::EnsurePipelineCreated( } return false; } - creation_arguments.pipeline->second.pipeline = pipeline; + creation_arguments.pipeline->second.pipeline.store(pipeline, + std::memory_order_release); return true; } diff --git a/src/xenia/gpu/vulkan/vulkan_pipeline_cache.h b/src/xenia/gpu/vulkan/vulkan_pipeline_cache.h index 2e1f28335..ebbea7425 100644 --- a/src/xenia/gpu/vulkan/vulkan_pipeline_cache.h +++ b/src/xenia/gpu/vulkan/vulkan_pipeline_cache.h @@ -10,15 +10,22 @@ #ifndef XENIA_GPU_VULKAN_VULKAN_PIPELINE_STATE_CACHE_H_ #define XENIA_GPU_VULKAN_VULKAN_PIPELINE_STATE_CACHE_H_ +#include +#include #include #include +#include #include #include +#include +#include #include #include +#include #include "xenia/base/hash.h" #include "xenia/base/platform.h" +#include "xenia/base/threading.h" #include "xenia/base/xxhash.h" #include "xenia/gpu/primitive_processor.h" #include "xenia/gpu/register_file.h" @@ -39,8 +46,6 @@ class VulkanCommandProcessor; // implementations. class VulkanPipelineCache { public: - static constexpr size_t kLayoutUIDEmpty = 0; - class PipelineLayoutProvider { public: virtual ~PipelineLayoutProvider() {} @@ -50,6 +55,34 @@ class VulkanPipelineCache { PipelineLayoutProvider() = default; }; + struct Pipeline { + std::atomic pipeline{VK_NULL_HANDLE}; + // The layouts are owned by the VulkanCommandProcessor, and must not be + // destroyed by it while the pipeline cache is active. + const PipelineLayoutProvider* pipeline_layout; + + Pipeline(const PipelineLayoutProvider* pipeline_layout_provider) + : pipeline_layout(pipeline_layout_provider) {} + + // Copy constructor needed for unordered_map + Pipeline(const Pipeline& other) + : pipeline(other.pipeline.load(std::memory_order_acquire)), + pipeline_layout(other.pipeline_layout) {} + + // Move constructor + Pipeline(Pipeline&& other) noexcept + : pipeline(other.pipeline.load(std::memory_order_acquire)), + pipeline_layout(other.pipeline_layout) {} + + // Deleted copy assignment to prevent accidental copying + Pipeline& operator=(const Pipeline&) = delete; + + // Deleted move assignment + Pipeline& operator=(Pipeline&&) = delete; + }; + + static constexpr size_t kLayoutUIDEmpty = 0; + VulkanPipelineCache(VulkanCommandProcessor& command_processor, const RegisterFile& register_file, VulkanRenderTargetCache& render_target_cache, @@ -59,6 +92,9 @@ class VulkanPipelineCache { bool Initialize(); void Shutdown(); + void EndSubmission(); + bool IsCreatingPipelines(); + VulkanShader* LoadShader(xenos::ShaderType shader_type, const uint32_t* host_address, uint32_t dword_count); // Analyze shader microcode on the translator thread. @@ -78,7 +114,6 @@ class VulkanPipelineCache { bool EnsureShadersTranslated(VulkanShader::VulkanTranslation* vertex_shader, VulkanShader::VulkanTranslation* pixel_shader); - // TODO(Triang3l): Return a deferred creation handle. bool ConfigurePipeline( VulkanShader::VulkanTranslation* vertex_shader, VulkanShader::VulkanTranslation* pixel_shader, @@ -86,8 +121,7 @@ class VulkanPipelineCache { reg::RB_DEPTHCONTROL normalized_depth_control, uint32_t normalized_color_mask, VulkanRenderTargetCache::RenderPassKey render_pass_key, - VkPipeline& pipeline_out, - const PipelineLayoutProvider*& pipeline_layout_out); + Pipeline** pipeline_out); private: enum class PipelineGeometryShader : uint32_t { @@ -218,26 +252,27 @@ class VulkanPipelineCache { }; }); - struct Pipeline { - VkPipeline pipeline = VK_NULL_HANDLE; - // The layouts are owned by the VulkanCommandProcessor, and must not be - // destroyed by it while the pipeline cache is active. - const PipelineLayoutProvider* pipeline_layout; - Pipeline(const PipelineLayoutProvider* pipeline_layout_provider) - : pipeline_layout(pipeline_layout_provider) {} - }; - - // Description that can be passed from the command processor thread to the // creation threads, with everything needed from caches pre-looked-up. struct PipelineCreationArguments { std::pair* pipeline; - const VulkanShader::VulkanTranslation* vertex_shader; - const VulkanShader::VulkanTranslation* pixel_shader; + VulkanShader::VulkanTranslation* vertex_shader; + VulkanShader::VulkanTranslation* pixel_shader; VkShaderModule geometry_shader; // Tessellation shaders (only used when tessellation is active). VkShaderModule tessellation_vertex_shader; // VS for passing data to TCS. VkShaderModule tessellation_control_shader; // TCS (hull shader). VkRenderPass render_pass; + // Priority for async compilation (higher = compiled sooner). + // Pipelines that write to visible render targets get higher priority. + uint8_t priority = 0; + }; + + // Comparator for priority queue - higher priority first. + struct PipelineCreationPriorityCompare { + bool operator()(const PipelineCreationArguments& a, + const PipelineCreationArguments& b) const { + return a.priority < b.priority; // max-heap: lower priority at bottom + } }; union GeometryShaderKey { @@ -374,8 +409,26 @@ class VulkanPipelineCache { pipelines_; // Previously used pipeline, to avoid lookups if the state wasn't changed. - const std::pair* last_pipeline_ = - nullptr; + std::pair* last_pipeline_ = nullptr; + + void CreationThread(); + + // For asynchronous creation. + std::vector> creation_threads_; + std::atomic creation_threads_shutdown_{false}; + std::atomic creation_threads_busy_{0}; + // Priority queue contains pointers to map entries. Pipelines are never + // evicted as games have a finite set that should all remain cached for + // performance. Higher priority pipelines (those writing to visible RTs) + // are compiled first. + std::priority_queue, + PipelineCreationPriorityCompare> + creation_queue_; + std::mutex creation_request_lock_; + std::condition_variable creation_request_cond_; + std::unique_ptr creation_completion_event_ = nullptr; + std::atomic creation_completion_set_event_{false}; }; } // namespace vulkan