[Vulkan] Add thread pool for async shader compilation

Uses priority queue with higher priority on shaders with visible
render targets
This commit is contained in:
Herman S.
2025-09-23 21:06:51 +09:00
parent 19c5401eda
commit ccf8fb66f5
3 changed files with 324 additions and 60 deletions

View File

@@ -2446,17 +2446,27 @@ bool VulkanCommandProcessor::IssueDraw(xenos::PrimitiveType prim_type,
// Create the pipeline (for this, need the render pass from the render target
// cache), translating the shaders - doing this now to obtain the used
// textures.
VkPipeline pipeline;
const VulkanPipelineCache::PipelineLayoutProvider* pipeline_layout_provider;
VulkanPipelineCache::Pipeline* pipeline;
if (!pipeline_cache_->ConfigurePipeline(
vertex_shader_translation, pixel_shader_translation,
primitive_processing_result, normalized_depth_control,
normalized_color_mask,
render_target_cache_->last_update_render_pass_key(), pipeline,
pipeline_layout_provider)) {
render_target_cache_->last_update_render_pass_key(), &pipeline)) {
return false;
}
VkPipeline current_pipeline =
pipeline->pipeline.load(std::memory_order_acquire);
if (current_pipeline == VK_NULL_HANDLE) {
// Pipeline is not ready yet - wait for it to be created.
pipeline_cache_->EndSubmission();
current_pipeline = pipeline->pipeline.load(std::memory_order_acquire);
if (current_pipeline == VK_NULL_HANDLE) {
// Still not ready - something is wrong.
return false;
}
}
// Update the textures before most other work in the submission because
// samplers depend on this (and in case of sampler overflow in a submission,
// submissions must be split) - may perform dispatches and copying.
@@ -2470,14 +2480,17 @@ bool VulkanCommandProcessor::IssueDraw(xenos::PrimitiveType prim_type,
// Update the graphics pipeline, and if the new graphics pipeline has a
// different layout, invalidate incompatible descriptor sets before updating
// current_guest_graphics_pipeline_layout_.
if (current_guest_graphics_pipeline_ != pipeline) {
// The pipeline may be not ready yet if created asynchronously.
// EndSubmission must be called before submitting the command buffer to
// await its creation.
if (current_guest_graphics_pipeline_ != current_pipeline) {
deferred_command_buffer_.CmdVkBindPipeline(VK_PIPELINE_BIND_POINT_GRAPHICS,
pipeline);
current_guest_graphics_pipeline_ = pipeline;
current_pipeline);
current_guest_graphics_pipeline_ = current_pipeline;
current_external_graphics_pipeline_ = VK_NULL_HANDLE;
}
auto pipeline_layout =
static_cast<const PipelineLayout*>(pipeline_layout_provider);
static_cast<const PipelineLayout*>(pipeline->pipeline_layout);
if (current_guest_graphics_pipeline_layout_ != pipeline_layout) {
if (current_guest_graphics_pipeline_layout_) {
// Keep descriptor set layouts for which the new pipeline layout is

View File

@@ -45,6 +45,14 @@ namespace shaders {
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/tessellation_adaptive_vs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/tessellation_indexed_vs.h"
} // namespace shaders
DEFINE_int32(
vulkan_pipeline_creation_threads, -1,
"Number of threads used for graphics pipeline creation. -1 to calculate "
"automatically (75% of logical CPU cores), a positive number to specify "
"the number of threads explicitly (up to the number of logical CPU cores), "
"0 to disable multithreaded pipeline creation.",
"Vulkan");
namespace xe {
namespace gpu {
namespace vulkan {
@@ -167,10 +175,52 @@ bool VulkanPipelineCache::Initialize() {
vk_pipeline_cache_ = VK_NULL_HANDLE;
}
uint32_t logical_processor_count = xe::threading::logical_processor_count();
if (!logical_processor_count) {
// Pick some reasonable amount if couldn't determine the number of cores.
logical_processor_count = 6;
}
creation_completion_event_ =
xe::threading::Event::CreateManualResetEvent(true);
assert_not_null(creation_completion_event_);
if (cvars::vulkan_pipeline_creation_threads != 0) {
size_t creation_thread_count;
if (cvars::vulkan_pipeline_creation_threads < 0) {
creation_thread_count =
std::max(logical_processor_count * 3 / 4, uint32_t(1));
} else {
creation_thread_count =
std::min(uint32_t(cvars::vulkan_pipeline_creation_threads),
logical_processor_count);
}
creation_threads_shutdown_ = false;
for (size_t i = 0; i < creation_thread_count; ++i) {
std::unique_ptr<xe::threading::Thread> creation_thread =
xe::threading::Thread::Create({}, [this]() { CreationThread(); });
assert_not_null(creation_thread);
creation_thread->set_name("Vulkan Pipelines");
creation_threads_.push_back(std::move(creation_thread));
}
}
return true;
}
void VulkanPipelineCache::Shutdown() {
// Shut down all threads, before destroying the pipelines since they may be
// creating them.
if (!creation_threads_.empty()) {
{
std::lock_guard<std::mutex> lock(creation_request_lock_);
creation_threads_shutdown_ = true;
}
creation_request_cond_.notify_all();
for (size_t i = 0; i < creation_threads_.size(); ++i) {
xe::threading::Wait(creation_threads_[i].get(), false);
}
creation_threads_.clear();
}
creation_completion_event_.reset();
const ui::vulkan::VulkanDevice* const vulkan_device =
command_processor_.GetVulkanDevice();
const ui::vulkan::VulkanDevice::Functions& dfn = vulkan_device->functions();
@@ -423,17 +473,11 @@ bool VulkanPipelineCache::ConfigurePipeline(
reg::RB_DEPTHCONTROL normalized_depth_control,
uint32_t normalized_color_mask,
VulkanRenderTargetCache::RenderPassKey render_pass_key,
VkPipeline& pipeline_out,
const PipelineLayoutProvider*& pipeline_layout_out) {
VulkanPipelineCache::Pipeline** pipeline_out) {
#if XE_GPU_FINE_GRAINED_DRAW_SCOPES
SCOPE_profile_cpu_f("gpu");
#endif // XE_GPU_FINE_GRAINED_DRAW_SCOPES
// Ensure shaders are translated - needed now for GetCurrentStateDescription.
if (!EnsureShadersTranslated(vertex_shader, pixel_shader)) {
return false;
}
PipelineDescription description;
if (!GetCurrentStateDescription(
vertex_shader, pixel_shader, primitive_processing_result,
@@ -442,15 +486,13 @@ bool VulkanPipelineCache::ConfigurePipeline(
return false;
}
if (last_pipeline_ && last_pipeline_->first == description) {
pipeline_out = last_pipeline_->second.pipeline;
pipeline_layout_out = last_pipeline_->second.pipeline_layout;
*pipeline_out = &last_pipeline_->second;
return true;
}
auto it = pipelines_.find(description);
if (it != pipelines_.end()) {
last_pipeline_ = &*it;
pipeline_out = it->second.pipeline;
pipeline_layout_out = it->second.pipeline_layout;
*pipeline_out = &it->second;
return true;
}
@@ -476,19 +518,22 @@ bool VulkanPipelineCache::ConfigurePipeline(
if (!pipeline_layout) {
return false;
}
VkShaderModule geometry_shader = VK_NULL_HANDLE;
GeometryShaderKey geometry_shader_key;
if (GetGeometryShaderKey(
description.geometry_shader,
SpirvShaderTranslator::Modification(vertex_shader->modification()),
SpirvShaderTranslator::Modification(
pixel_shader ? pixel_shader->modification() : 0),
geometry_shader_key)) {
if (description.geometry_shader != PipelineGeometryShader::kNone) {
GeometryShaderKey geometry_shader_key;
GetGeometryShaderKey(
description.geometry_shader,
SpirvShaderTranslator::Modification(vertex_shader->modification()),
SpirvShaderTranslator::Modification(
pixel_shader ? pixel_shader->modification() : 0),
geometry_shader_key);
geometry_shader = GetGeometryShader(geometry_shader_key);
if (geometry_shader == VK_NULL_HANDLE) {
return false;
}
}
VkRenderPass render_pass =
render_target_cache_.GetPath() ==
RenderTargetCache::Path::kPixelShaderInterlock
@@ -498,9 +543,10 @@ bool VulkanPipelineCache::ConfigurePipeline(
if (render_pass == VK_NULL_HANDLE) {
return false;
}
PipelineCreationArguments creation_arguments;
auto& pipeline =
auto& pipeline_pair =
*pipelines_.emplace(description, Pipeline(pipeline_layout)).first;
// Get tessellation shaders if needed.
VkShaderModule tessellation_vertex_shader = VK_NULL_HANDLE;
VkShaderModule tessellation_control_shader = VK_NULL_HANDLE;
@@ -523,19 +569,169 @@ bool VulkanPipelineCache::ConfigurePipeline(
}
}
creation_arguments.pipeline = &pipeline;
creation_arguments.vertex_shader = vertex_shader;
creation_arguments.pixel_shader = pixel_shader;
creation_arguments.geometry_shader = geometry_shader;
creation_arguments.tessellation_vertex_shader = tessellation_vertex_shader;
creation_arguments.tessellation_control_shader = tessellation_control_shader;
creation_arguments.render_pass = render_pass;
if (!EnsurePipelineCreated(creation_arguments)) {
// Pipeline hot-swap: When async mode is enabled, we have creation threads and
// a pixel shader, create a placeholder pipeline immediately (fast compile)
// and queue the real pipeline creation in the background. This reduces
// stutter from pipeline compilation.
bool use_async = cvars::async_shader_compilation &&
!creation_threads_.empty() && pixel_shader &&
placeholder_pixel_shader_ != VK_NULL_HANDLE;
if (use_async) {
// Create placeholder pipeline immediately (uses simple PS, fast compile).
// Set is_placeholder BEFORE creating the pipeline to avoid race condition
// with the creation thread checking this flag.
pipeline_pair.second.is_placeholder.store(true, std::memory_order_release);
PipelineCreationArguments placeholder_args;
placeholder_args.pipeline = &pipeline_pair;
placeholder_args.vertex_shader = vertex_shader;
placeholder_args.pixel_shader = nullptr; // Will use placeholder PS
placeholder_args.geometry_shader = geometry_shader;
placeholder_args.tessellation_vertex_shader = tessellation_vertex_shader;
placeholder_args.tessellation_control_shader = tessellation_control_shader;
placeholder_args.render_pass = render_pass;
if (EnsurePipelineCreatedWithPlaceholder(placeholder_args)) {
// Queue real pipeline creation in background.
// Calculate priority based on whether shader writes to visible RTs.
uint8_t priority = 0;
if (pixel_shader) {
// Get bound RT mask from normalized_color_mask (4 bits per RT).
uint32_t bound_rts = (((normalized_color_mask >> 0) & 0xF) ? 1 : 0) |
(((normalized_color_mask >> 4) & 0xF) ? 2 : 0) |
(((normalized_color_mask >> 8) & 0xF) ? 4 : 0) |
(((normalized_color_mask >> 12) & 0xF) ? 8 : 0);
uint32_t shader_writes = pixel_shader->shader().writes_color_targets();
if (bound_rts & shader_writes) {
// Writes to at least one visible RT - high priority.
priority = 2;
// Extra priority if writing to RT0 (usually main color buffer).
if ((bound_rts & shader_writes) & 1) {
priority = 3;
}
} else if (pixel_shader->shader().writes_depth()) {
// Depth-only - medium priority.
priority = 1;
}
// else: writes to unbound RTs only - lowest priority (0).
}
{
std::lock_guard<std::mutex> lock(creation_request_lock_);
PipelineCreationArguments creation_arguments;
creation_arguments.pipeline = &pipeline_pair;
creation_arguments.vertex_shader = vertex_shader;
creation_arguments.pixel_shader = pixel_shader;
creation_arguments.geometry_shader = geometry_shader;
creation_arguments.tessellation_vertex_shader =
tessellation_vertex_shader;
creation_arguments.tessellation_control_shader =
tessellation_control_shader;
creation_arguments.render_pass = render_pass;
creation_arguments.priority = priority;
creation_queue_.push(creation_arguments);
}
creation_request_cond_.notify_one();
} else {
// Placeholder creation failed, fall back to sync creation.
// Reset the flag we set earlier.
pipeline_pair.second.is_placeholder.store(false,
std::memory_order_release);
use_async = false;
}
}
if (!use_async) {
// Sync mode or no creation threads: create synchronously.
PipelineCreationArguments creation_arguments;
creation_arguments.pipeline = &pipeline_pair;
creation_arguments.vertex_shader = vertex_shader;
creation_arguments.pixel_shader = pixel_shader;
creation_arguments.geometry_shader = geometry_shader;
creation_arguments.tessellation_vertex_shader = tessellation_vertex_shader;
creation_arguments.tessellation_control_shader = tessellation_control_shader;
creation_arguments.render_pass = render_pass;
if (!EnsurePipelineCreated(creation_arguments)) {
return false;
}
}
last_pipeline_ = &pipeline_pair;
*pipeline_out = &pipeline_pair.second;
return true;
}
void VulkanPipelineCache::EndSubmission() {
if (creation_threads_.empty()) {
return;
}
// Await creation of all queued pipelines.
bool await_creation_completion_event;
{
std::lock_guard<std::mutex> lock(creation_request_lock_);
// Assuming the creation queue is already empty (because the processor
// thread also worked on creating the leftover pipelines), so only check
// if there are threads with pipelines currently being created.
await_creation_completion_event =
!creation_queue_.empty() || creation_threads_busy_ != 0;
if (await_creation_completion_event) {
creation_completion_event_->Reset();
creation_completion_set_event_.store(true, std::memory_order_release);
}
}
if (await_creation_completion_event) {
creation_request_cond_.notify_one();
xe::threading::Wait(creation_completion_event_.get(), false);
}
}
bool VulkanPipelineCache::IsCreatingPipelines() {
if (creation_threads_.empty()) {
return false;
}
pipeline_out = pipeline.second.pipeline;
pipeline_layout_out = pipeline_layout;
return true;
std::lock_guard<std::mutex> lock(creation_request_lock_);
return !creation_queue_.empty() || creation_threads_busy_ != 0;
}
void VulkanPipelineCache::CreationThread() {
for (;;) {
PipelineCreationArguments creation_arguments;
{
std::unique_lock<std::mutex> lock(creation_request_lock_);
creation_request_cond_.wait(lock, [this]() {
return !creation_queue_.empty() || creation_threads_shutdown_;
});
if (creation_threads_shutdown_) {
break;
}
creation_arguments = creation_queue_.top();
creation_queue_.pop();
++creation_threads_busy_;
}
if (!EnsureShadersTranslated(creation_arguments.vertex_shader,
creation_arguments.pixel_shader)) {
// Mark pipeline as failed by keeping it as VK_NULL_HANDLE.
// The pipeline will remain VK_NULL_HANDLE, indicating failure.
XELOGE("Failed to translate shaders for pipeline creation");
} else {
if (!EnsurePipelineCreated(creation_arguments)) {
// Pipeline creation failed - it will remain VK_NULL_HANDLE.
XELOGE("Failed to create Vulkan pipeline");
}
}
{
std::lock_guard<std::mutex> lock(creation_request_lock_);
--creation_threads_busy_;
if (creation_completion_set_event_.load(std::memory_order_acquire) &&
creation_threads_busy_ == 0 && creation_queue_.empty()) {
creation_completion_set_event_.store(false, std::memory_order_release);
creation_completion_event_->Set();
}
}
}
}
bool VulkanPipelineCache::TranslateAnalyzedShader(
@@ -2057,7 +2253,8 @@ VkShaderModule VulkanPipelineCache::GetTessellationVertexShader(
bool VulkanPipelineCache::EnsurePipelineCreated(
const PipelineCreationArguments& creation_arguments) {
if (creation_arguments.pipeline->second.pipeline != VK_NULL_HANDLE) {
if (creation_arguments.pipeline->second.pipeline.load(
std::memory_order_acquire) != VK_NULL_HANDLE) {
return true;
}
@@ -2564,7 +2761,8 @@ bool VulkanPipelineCache::EnsurePipelineCreated(
}
return false;
}
creation_arguments.pipeline->second.pipeline = pipeline;
creation_arguments.pipeline->second.pipeline.store(pipeline,
std::memory_order_release);
return true;
}

View File

@@ -10,15 +10,22 @@
#ifndef XENIA_GPU_VULKAN_VULKAN_PIPELINE_STATE_CACHE_H_
#define XENIA_GPU_VULKAN_VULKAN_PIPELINE_STATE_CACHE_H_
#include <atomic>
#include <condition_variable>
#include <cstddef>
#include <cstring>
#include <deque>
#include <functional>
#include <memory>
#include <mutex>
#include <queue>
#include <unordered_map>
#include <utility>
#include <vector>
#include "xenia/base/hash.h"
#include "xenia/base/platform.h"
#include "xenia/base/threading.h"
#include "xenia/base/xxhash.h"
#include "xenia/gpu/primitive_processor.h"
#include "xenia/gpu/register_file.h"
@@ -39,8 +46,6 @@ class VulkanCommandProcessor;
// implementations.
class VulkanPipelineCache {
public:
static constexpr size_t kLayoutUIDEmpty = 0;
class PipelineLayoutProvider {
public:
virtual ~PipelineLayoutProvider() {}
@@ -50,6 +55,34 @@ class VulkanPipelineCache {
PipelineLayoutProvider() = default;
};
struct Pipeline {
std::atomic<VkPipeline> pipeline{VK_NULL_HANDLE};
// The layouts are owned by the VulkanCommandProcessor, and must not be
// destroyed by it while the pipeline cache is active.
const PipelineLayoutProvider* pipeline_layout;
Pipeline(const PipelineLayoutProvider* pipeline_layout_provider)
: pipeline_layout(pipeline_layout_provider) {}
// Copy constructor needed for unordered_map
Pipeline(const Pipeline& other)
: pipeline(other.pipeline.load(std::memory_order_acquire)),
pipeline_layout(other.pipeline_layout) {}
// Move constructor
Pipeline(Pipeline&& other) noexcept
: pipeline(other.pipeline.load(std::memory_order_acquire)),
pipeline_layout(other.pipeline_layout) {}
// Deleted copy assignment to prevent accidental copying
Pipeline& operator=(const Pipeline&) = delete;
// Deleted move assignment
Pipeline& operator=(Pipeline&&) = delete;
};
static constexpr size_t kLayoutUIDEmpty = 0;
VulkanPipelineCache(VulkanCommandProcessor& command_processor,
const RegisterFile& register_file,
VulkanRenderTargetCache& render_target_cache,
@@ -59,6 +92,9 @@ class VulkanPipelineCache {
bool Initialize();
void Shutdown();
void EndSubmission();
bool IsCreatingPipelines();
VulkanShader* LoadShader(xenos::ShaderType shader_type,
const uint32_t* host_address, uint32_t dword_count);
// Analyze shader microcode on the translator thread.
@@ -78,7 +114,6 @@ class VulkanPipelineCache {
bool EnsureShadersTranslated(VulkanShader::VulkanTranslation* vertex_shader,
VulkanShader::VulkanTranslation* pixel_shader);
// TODO(Triang3l): Return a deferred creation handle.
bool ConfigurePipeline(
VulkanShader::VulkanTranslation* vertex_shader,
VulkanShader::VulkanTranslation* pixel_shader,
@@ -86,8 +121,7 @@ class VulkanPipelineCache {
reg::RB_DEPTHCONTROL normalized_depth_control,
uint32_t normalized_color_mask,
VulkanRenderTargetCache::RenderPassKey render_pass_key,
VkPipeline& pipeline_out,
const PipelineLayoutProvider*& pipeline_layout_out);
Pipeline** pipeline_out);
private:
enum class PipelineGeometryShader : uint32_t {
@@ -218,26 +252,27 @@ class VulkanPipelineCache {
};
});
struct Pipeline {
VkPipeline pipeline = VK_NULL_HANDLE;
// The layouts are owned by the VulkanCommandProcessor, and must not be
// destroyed by it while the pipeline cache is active.
const PipelineLayoutProvider* pipeline_layout;
Pipeline(const PipelineLayoutProvider* pipeline_layout_provider)
: pipeline_layout(pipeline_layout_provider) {}
};
// Description that can be passed from the command processor thread to the
// creation threads, with everything needed from caches pre-looked-up.
struct PipelineCreationArguments {
std::pair<const PipelineDescription, Pipeline>* pipeline;
const VulkanShader::VulkanTranslation* vertex_shader;
const VulkanShader::VulkanTranslation* pixel_shader;
VulkanShader::VulkanTranslation* vertex_shader;
VulkanShader::VulkanTranslation* pixel_shader;
VkShaderModule geometry_shader;
// Tessellation shaders (only used when tessellation is active).
VkShaderModule tessellation_vertex_shader; // VS for passing data to TCS.
VkShaderModule tessellation_control_shader; // TCS (hull shader).
VkRenderPass render_pass;
// Priority for async compilation (higher = compiled sooner).
// Pipelines that write to visible render targets get higher priority.
uint8_t priority = 0;
};
// Comparator for priority queue - higher priority first.
struct PipelineCreationPriorityCompare {
bool operator()(const PipelineCreationArguments& a,
const PipelineCreationArguments& b) const {
return a.priority < b.priority; // max-heap: lower priority at bottom
}
};
union GeometryShaderKey {
@@ -374,8 +409,26 @@ class VulkanPipelineCache {
pipelines_;
// Previously used pipeline, to avoid lookups if the state wasn't changed.
const std::pair<const PipelineDescription, Pipeline>* last_pipeline_ =
nullptr;
std::pair<const PipelineDescription, Pipeline>* last_pipeline_ = nullptr;
void CreationThread();
// For asynchronous creation.
std::vector<std::unique_ptr<xe::threading::Thread>> creation_threads_;
std::atomic<bool> creation_threads_shutdown_{false};
std::atomic<size_t> creation_threads_busy_{0};
// Priority queue contains pointers to map entries. Pipelines are never
// evicted as games have a finite set that should all remain cached for
// performance. Higher priority pipelines (those writing to visible RTs)
// are compiled first.
std::priority_queue<PipelineCreationArguments,
std::vector<PipelineCreationArguments>,
PipelineCreationPriorityCompare>
creation_queue_;
std::mutex creation_request_lock_;
std::condition_variable creation_request_cond_;
std::unique_ptr<xe::threading::Event> creation_completion_event_ = nullptr;
std::atomic<bool> creation_completion_set_event_{false};
};
} // namespace vulkan