Files
Xenia-Canary/src/xenia/gpu/vulkan/vulkan_render_target_cache.cc
2025-12-10 16:17:20 +01:00

6222 lines
296 KiB
C++

/**
******************************************************************************
* Xenia : Xbox 360 Emulator Research Project *
******************************************************************************
* Copyright 2022 Ben Vanik. All rights reserved. *
* Released under the BSD license - see LICENSE in the root for more details. *
******************************************************************************
*/
#include "xenia/gpu/vulkan/vulkan_render_target_cache.h"
#include <cstddef>
#include <cstdint>
#include <cstring>
#include "third_party/glslang/SPIRV/GLSL.std.450.h"
#include "xenia/base/assert.h"
#include "xenia/base/cvar.h"
#include "xenia/base/logging.h"
#include "xenia/base/math.h"
#include "xenia/gpu/draw_util.h"
#include "xenia/gpu/registers.h"
#include "xenia/gpu/spirv_builder.h"
#include "xenia/gpu/spirv_shader_translator.h"
#include "xenia/gpu/texture_cache.h"
#include "xenia/gpu/vulkan/deferred_command_buffer.h"
#include "xenia/gpu/vulkan/vulkan_command_processor.h"
#include "xenia/gpu/xenos.h"
#include "xenia/ui/vulkan/vulkan_util.h"
DEFINE_string(
render_target_path_vulkan, "",
"Render target emulation path to use on Vulkan.\n"
"Use: [any, fbo, fsi]\n"
" fbo:\n"
" Host framebuffers and fixed-function blending and depth / stencil "
"testing, copying between render targets when needed.\n"
" Lower accuracy (limited pixel format support).\n"
" Performance limited primarily by render target layout changes requiring "
"copying, but generally higher.\n"
" fsi:\n"
" Manual pixel packing, blending and depth / stencil testing, with free "
"render target layout changes.\n"
" Requires a GPU supporting fragment shader interlock.\n"
" Highest accuracy (all pixel formats handled in software).\n"
" Performance limited primarily by overdraw.\n"
" Any other value:\n"
" Choose what is considered the most optimal for the system (currently "
"always FB because the FSI path is much slower now).",
"GPU");
namespace xe {
namespace gpu {
namespace vulkan {
// Generated with `xb buildshaders`.
namespace shaders {
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/host_depth_store_1xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/host_depth_store_2xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/host_depth_store_4xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/passthrough_position_xy_vs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_clear_32bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_clear_32bpp_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_clear_64bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_clear_64bpp_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_32bpp_1x2xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_32bpp_1x2xmsaa_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_32bpp_4xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_32bpp_4xmsaa_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_64bpp_1x2xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_64bpp_1x2xmsaa_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_64bpp_4xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_64bpp_4xmsaa_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_128bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_128bpp_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_16bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_16bpp_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_32bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_32bpp_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_64bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_64bpp_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_8bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_8bpp_scaled_cs.h"
} // namespace shaders
const VulkanRenderTargetCache::ResolveCopyShaderCode
VulkanRenderTargetCache::kResolveCopyShaders[size_t(
draw_util::ResolveCopyShaderIndex::kCount)] = {
{shaders::resolve_fast_32bpp_1x2xmsaa_cs,
sizeof(shaders::resolve_fast_32bpp_1x2xmsaa_cs),
shaders::resolve_fast_32bpp_1x2xmsaa_scaled_cs,
sizeof(shaders::resolve_fast_32bpp_1x2xmsaa_scaled_cs)},
{shaders::resolve_fast_32bpp_4xmsaa_cs,
sizeof(shaders::resolve_fast_32bpp_4xmsaa_cs),
shaders::resolve_fast_32bpp_4xmsaa_scaled_cs,
sizeof(shaders::resolve_fast_32bpp_4xmsaa_scaled_cs)},
{shaders::resolve_fast_64bpp_1x2xmsaa_cs,
sizeof(shaders::resolve_fast_64bpp_1x2xmsaa_cs),
shaders::resolve_fast_64bpp_1x2xmsaa_scaled_cs,
sizeof(shaders::resolve_fast_64bpp_1x2xmsaa_scaled_cs)},
{shaders::resolve_fast_64bpp_4xmsaa_cs,
sizeof(shaders::resolve_fast_64bpp_4xmsaa_cs),
shaders::resolve_fast_64bpp_4xmsaa_scaled_cs,
sizeof(shaders::resolve_fast_64bpp_4xmsaa_scaled_cs)},
{shaders::resolve_full_8bpp_cs, sizeof(shaders::resolve_full_8bpp_cs),
shaders::resolve_full_8bpp_scaled_cs,
sizeof(shaders::resolve_full_8bpp_scaled_cs)},
{shaders::resolve_full_16bpp_cs, sizeof(shaders::resolve_full_16bpp_cs),
shaders::resolve_full_16bpp_scaled_cs,
sizeof(shaders::resolve_full_16bpp_scaled_cs)},
{shaders::resolve_full_32bpp_cs, sizeof(shaders::resolve_full_32bpp_cs),
shaders::resolve_full_32bpp_scaled_cs,
sizeof(shaders::resolve_full_32bpp_scaled_cs)},
{shaders::resolve_full_64bpp_cs, sizeof(shaders::resolve_full_64bpp_cs),
shaders::resolve_full_64bpp_scaled_cs,
sizeof(shaders::resolve_full_64bpp_scaled_cs)},
{shaders::resolve_full_128bpp_cs,
sizeof(shaders::resolve_full_128bpp_cs),
shaders::resolve_full_128bpp_scaled_cs,
sizeof(shaders::resolve_full_128bpp_scaled_cs)},
};
const VulkanRenderTargetCache::TransferPipelineLayoutInfo
VulkanRenderTargetCache::kTransferPipelineLayoutInfos[size_t(
TransferPipelineLayoutIndex::kCount)] = {
// kColor
{kTransferUsedDescriptorSetColorTextureBit,
kTransferUsedPushConstantDwordAddressBit},
// kDepth
{kTransferUsedDescriptorSetDepthStencilTexturesBit,
kTransferUsedPushConstantDwordAddressBit},
// kColorToStencilBit
{kTransferUsedDescriptorSetColorTextureBit,
kTransferUsedPushConstantDwordAddressBit |
kTransferUsedPushConstantDwordStencilMaskBit},
// kDepthToStencilBit
{kTransferUsedDescriptorSetDepthStencilTexturesBit,
kTransferUsedPushConstantDwordAddressBit |
kTransferUsedPushConstantDwordStencilMaskBit},
// kColorAndHostDepthTexture
{kTransferUsedDescriptorSetHostDepthStencilTexturesBit |
kTransferUsedDescriptorSetColorTextureBit,
kTransferUsedPushConstantDwordHostDepthAddressBit |
kTransferUsedPushConstantDwordAddressBit},
// kColorAndHostDepthBuffer
{kTransferUsedDescriptorSetHostDepthBufferBit |
kTransferUsedDescriptorSetColorTextureBit,
kTransferUsedPushConstantDwordHostDepthAddressBit |
kTransferUsedPushConstantDwordAddressBit},
// kDepthAndHostDepthTexture
{kTransferUsedDescriptorSetHostDepthStencilTexturesBit |
kTransferUsedDescriptorSetDepthStencilTexturesBit,
kTransferUsedPushConstantDwordHostDepthAddressBit |
kTransferUsedPushConstantDwordAddressBit},
// kDepthAndHostDepthBuffer
{kTransferUsedDescriptorSetHostDepthBufferBit |
kTransferUsedDescriptorSetDepthStencilTexturesBit,
kTransferUsedPushConstantDwordHostDepthAddressBit |
kTransferUsedPushConstantDwordAddressBit},
};
const VulkanRenderTargetCache::TransferModeInfo
VulkanRenderTargetCache::kTransferModes[size_t(TransferMode::kCount)] = {
// kColorToDepth
{TransferOutput::kDepth, TransferPipelineLayoutIndex::kColor},
// kColorToColor
{TransferOutput::kColor, TransferPipelineLayoutIndex::kColor},
// kDepthToDepth
{TransferOutput::kDepth, TransferPipelineLayoutIndex::kDepth},
// kDepthToColor
{TransferOutput::kColor, TransferPipelineLayoutIndex::kDepth},
// kColorToStencilBit
{TransferOutput::kStencilBit,
TransferPipelineLayoutIndex::kColorToStencilBit},
// kDepthToStencilBit
{TransferOutput::kStencilBit,
TransferPipelineLayoutIndex::kDepthToStencilBit},
// kColorAndHostDepthToDepth
{TransferOutput::kDepth,
TransferPipelineLayoutIndex::kColorAndHostDepthTexture},
// kDepthAndHostDepthToDepth
{TransferOutput::kDepth,
TransferPipelineLayoutIndex::kDepthAndHostDepthTexture},
// kColorAndHostDepthCopyToDepth
{TransferOutput::kDepth,
TransferPipelineLayoutIndex::kColorAndHostDepthBuffer},
// kDepthAndHostDepthCopyToDepth
{TransferOutput::kDepth,
TransferPipelineLayoutIndex::kDepthAndHostDepthBuffer},
};
VulkanRenderTargetCache::VulkanRenderTargetCache(
const RegisterFile& register_file, const Memory& memory,
TraceWriter& trace_writer, uint32_t draw_resolution_scale_x,
uint32_t draw_resolution_scale_y, VulkanCommandProcessor& command_processor)
: RenderTargetCache(register_file, memory, &trace_writer,
draw_resolution_scale_x, draw_resolution_scale_y),
command_processor_(command_processor),
trace_writer_(trace_writer) {}
VulkanRenderTargetCache::~VulkanRenderTargetCache() { Shutdown(true); }
bool VulkanRenderTargetCache::Initialize(uint32_t shared_memory_binding_count) {
const ui::vulkan::VulkanDevice* const vulkan_device =
command_processor_.GetVulkanDevice();
const ui::vulkan::VulkanInstance::Functions& ifn =
vulkan_device->vulkan_instance()->functions();
const VkPhysicalDevice physical_device = vulkan_device->physical_device();
const ui::vulkan::VulkanDevice::Functions& dfn = vulkan_device->functions();
const VkDevice device = vulkan_device->device();
const ui::vulkan::VulkanDevice::Properties& device_properties =
vulkan_device->properties();
if (cvars::render_target_path_vulkan == "fsi") {
path_ = Path::kPixelShaderInterlock;
} else {
path_ = Path::kHostRenderTargets;
}
// Fragment shader interlock is a feature implemented by pretty advanced GPUs,
// closer to Direct3D 11 / OpenGL ES 3.2 level mainly, not Direct3D 10 /
// OpenGL ES 3.1. Thus, it's fine to demand a wide range of other optional
// features for the fragment shader interlock backend to work.
if (path_ == Path::kPixelShaderInterlock) {
// Interlocking between fragments with common sample coverage is enough, but
// interlocking more is acceptable too (fragmentShaderShadingRateInterlock
// would be okay too, but it's unlikely that an implementation would
// advertise only it and not any other ones, as it's a very specific feature
// interacting with another optional feature that is variable shading rate,
// so there's no need to overcomplicate the checks and the shader execution
// mode setting).
// Sample-rate shading is required by certain SPIR-V revisions to access the
// sample mask fragment shader input.
// Stanard sample locations are needed for calculating the depth at the
// samples.
// It's unlikely that a device exposing fragment shader interlock won't have
// a large enough storage buffer range and a sufficient SSBO slot count for
// all the shared memory buffers and the EDRAM buffer - an in a conflict
// between, for instance, the ability to vfetch and memexport in fragment
// shaders, and the usage of fragment shader interlock, prefer the former
// for simplicity.
if (!(device_properties.fragmentShaderSampleInterlock ||
device_properties.fragmentShaderPixelInterlock) ||
!device_properties.fragmentStoresAndAtomics ||
!device_properties.sampleRateShading ||
!device_properties.standardSampleLocations ||
shared_memory_binding_count >=
device_properties.maxPerStageDescriptorStorageBuffers) {
path_ = Path::kHostRenderTargets;
}
}
// Format support.
constexpr VkFormatFeatureFlags kUsedDepthFormatFeatures =
VK_FORMAT_FEATURE_SAMPLED_IMAGE_BIT |
VK_FORMAT_FEATURE_DEPTH_STENCIL_ATTACHMENT_BIT;
VkFormatProperties depth_unorm24_properties;
ifn.vkGetPhysicalDeviceFormatProperties(
physical_device, VK_FORMAT_D24_UNORM_S8_UINT, &depth_unorm24_properties);
depth_unorm24_vulkan_format_supported_ =
(depth_unorm24_properties.optimalTilingFeatures &
kUsedDepthFormatFeatures) == kUsedDepthFormatFeatures;
// 2x MSAA support.
// TODO(Triang3l): Handle sampledImageIntegerSampleCounts 4 not supported in
// transfers.
if (cvars::native_2x_msaa) {
// Multisampled integer sampled images are optional in Vulkan and in Xenia.
msaa_2x_attachments_supported_ =
(device_properties.framebufferColorSampleCounts &
device_properties.framebufferDepthSampleCounts &
device_properties.framebufferStencilSampleCounts &
device_properties.sampledImageColorSampleCounts &
device_properties.sampledImageDepthSampleCounts &
device_properties.sampledImageStencilSampleCounts &
VK_SAMPLE_COUNT_2_BIT) &&
(device_properties.sampledImageIntegerSampleCounts &
(VK_SAMPLE_COUNT_2_BIT | VK_SAMPLE_COUNT_4_BIT)) !=
VK_SAMPLE_COUNT_4_BIT;
msaa_2x_no_attachments_supported_ =
(device_properties.framebufferNoAttachmentsSampleCounts &
VK_SAMPLE_COUNT_2_BIT) != 0;
} else {
msaa_2x_attachments_supported_ = false;
msaa_2x_no_attachments_supported_ = false;
}
// Descriptor set layouts.
VkDescriptorSetLayoutBinding descriptor_set_layout_bindings[2];
descriptor_set_layout_bindings[0].binding = 0;
descriptor_set_layout_bindings[0].descriptorType =
VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
descriptor_set_layout_bindings[0].descriptorCount = 1;
descriptor_set_layout_bindings[0].stageFlags =
VK_SHADER_STAGE_FRAGMENT_BIT | VK_SHADER_STAGE_COMPUTE_BIT;
descriptor_set_layout_bindings[0].pImmutableSamplers = nullptr;
VkDescriptorSetLayoutCreateInfo descriptor_set_layout_create_info;
descriptor_set_layout_create_info.sType =
VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO;
descriptor_set_layout_create_info.pNext = nullptr;
descriptor_set_layout_create_info.flags = 0;
descriptor_set_layout_create_info.bindingCount = 1;
descriptor_set_layout_create_info.pBindings = descriptor_set_layout_bindings;
if (dfn.vkCreateDescriptorSetLayout(
device, &descriptor_set_layout_create_info, nullptr,
&descriptor_set_layout_storage_buffer_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the descriptor set layout "
"with one storage buffer");
Shutdown();
return false;
}
descriptor_set_layout_bindings[0].descriptorType =
VK_DESCRIPTOR_TYPE_SAMPLED_IMAGE;
if (dfn.vkCreateDescriptorSetLayout(
device, &descriptor_set_layout_create_info, nullptr,
&descriptor_set_layout_sampled_image_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the descriptor set layout "
"with one sampled image");
Shutdown();
return false;
}
descriptor_set_layout_bindings[1].binding = 1;
descriptor_set_layout_bindings[1].descriptorType =
VK_DESCRIPTOR_TYPE_SAMPLED_IMAGE;
descriptor_set_layout_bindings[1].descriptorCount = 1;
descriptor_set_layout_bindings[1].stageFlags =
descriptor_set_layout_bindings[0].stageFlags;
descriptor_set_layout_bindings[1].pImmutableSamplers = nullptr;
descriptor_set_layout_create_info.bindingCount = 2;
if (dfn.vkCreateDescriptorSetLayout(
device, &descriptor_set_layout_create_info, nullptr,
&descriptor_set_layout_sampled_image_x2_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the descriptor set layout "
"with two sampled images");
Shutdown();
return false;
}
// Descriptor set pools.
// The pool sizes were chosen without a specific reason.
VkDescriptorPoolSize descriptor_set_layout_size;
descriptor_set_layout_size.type = VK_DESCRIPTOR_TYPE_SAMPLED_IMAGE;
descriptor_set_layout_size.descriptorCount = 1;
descriptor_set_pool_sampled_image_ =
std::make_unique<ui::vulkan::SingleLayoutDescriptorSetPool>(
vulkan_device, 256, 1, &descriptor_set_layout_size,
descriptor_set_layout_sampled_image_);
descriptor_set_layout_size.descriptorCount = 2;
descriptor_set_pool_sampled_image_x2_ =
std::make_unique<ui::vulkan::SingleLayoutDescriptorSetPool>(
vulkan_device, 256, 1, &descriptor_set_layout_size,
descriptor_set_layout_sampled_image_x2_);
// EDRAM contents reinterpretation buffer.
// 90 MB with 9x resolution scaling - within the minimum
// maxStorageBufferRange.
if (!ui::vulkan::util::CreateDedicatedAllocationBuffer(
vulkan_device,
VkDeviceSize(xenos::kEdramSizeBytes *
(draw_resolution_scale_x() * draw_resolution_scale_y())),
VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT |
VK_BUFFER_USAGE_STORAGE_BUFFER_BIT,
ui::vulkan::util::MemoryPurpose::kDeviceLocal, edram_buffer_,
edram_buffer_memory_)) {
XELOGE("VulkanRenderTargetCache: Failed to create the EDRAM buffer");
Shutdown();
return false;
}
if (GetPath() == Path::kPixelShaderInterlock) {
// The first operation will likely be drawing.
edram_buffer_usage_ = EdramBufferUsage::kFragmentReadWrite;
} else {
// The first operation will likely be depth self-comparison.
edram_buffer_usage_ = EdramBufferUsage::kFragmentRead;
}
edram_buffer_modification_status_ =
EdramBufferModificationStatus::kUnmodified;
VkDescriptorPoolSize edram_storage_buffer_descriptor_pool_size;
edram_storage_buffer_descriptor_pool_size.type =
VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
edram_storage_buffer_descriptor_pool_size.descriptorCount = 1;
VkDescriptorPoolCreateInfo edram_storage_buffer_descriptor_pool_create_info;
edram_storage_buffer_descriptor_pool_create_info.sType =
VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO;
edram_storage_buffer_descriptor_pool_create_info.pNext = nullptr;
edram_storage_buffer_descriptor_pool_create_info.flags = 0;
edram_storage_buffer_descriptor_pool_create_info.maxSets = 1;
edram_storage_buffer_descriptor_pool_create_info.poolSizeCount = 1;
edram_storage_buffer_descriptor_pool_create_info.pPoolSizes =
&edram_storage_buffer_descriptor_pool_size;
if (dfn.vkCreateDescriptorPool(
device, &edram_storage_buffer_descriptor_pool_create_info, nullptr,
&edram_storage_buffer_descriptor_pool_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the EDRAM buffer storage "
"buffer descriptor pool");
Shutdown();
return false;
}
VkDescriptorSetAllocateInfo edram_storage_buffer_descriptor_set_allocate_info;
edram_storage_buffer_descriptor_set_allocate_info.sType =
VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO;
edram_storage_buffer_descriptor_set_allocate_info.pNext = nullptr;
edram_storage_buffer_descriptor_set_allocate_info.descriptorPool =
edram_storage_buffer_descriptor_pool_;
edram_storage_buffer_descriptor_set_allocate_info.descriptorSetCount = 1;
edram_storage_buffer_descriptor_set_allocate_info.pSetLayouts =
&descriptor_set_layout_storage_buffer_;
if (dfn.vkAllocateDescriptorSets(
device, &edram_storage_buffer_descriptor_set_allocate_info,
&edram_storage_buffer_descriptor_set_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to allocate the EDRAM buffer storage "
"buffer descriptor set");
Shutdown();
return false;
}
VkDescriptorBufferInfo edram_storage_buffer_descriptor_buffer_info;
edram_storage_buffer_descriptor_buffer_info.buffer = edram_buffer_;
edram_storage_buffer_descriptor_buffer_info.offset = 0;
edram_storage_buffer_descriptor_buffer_info.range = VK_WHOLE_SIZE;
VkWriteDescriptorSet edram_storage_buffer_descriptor_write;
edram_storage_buffer_descriptor_write.sType =
VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
edram_storage_buffer_descriptor_write.pNext = nullptr;
edram_storage_buffer_descriptor_write.dstSet =
edram_storage_buffer_descriptor_set_;
edram_storage_buffer_descriptor_write.dstBinding = 0;
edram_storage_buffer_descriptor_write.dstArrayElement = 0;
edram_storage_buffer_descriptor_write.descriptorCount = 1;
edram_storage_buffer_descriptor_write.descriptorType =
VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
edram_storage_buffer_descriptor_write.pImageInfo = nullptr;
edram_storage_buffer_descriptor_write.pBufferInfo =
&edram_storage_buffer_descriptor_buffer_info;
edram_storage_buffer_descriptor_write.pTexelBufferView = nullptr;
dfn.vkUpdateDescriptorSets(device, 1, &edram_storage_buffer_descriptor_write,
0, nullptr);
bool draw_resolution_scaled = IsDrawResolutionScaled();
// Resolve copy pipeline layout.
VkDescriptorSetLayout
resolve_copy_descriptor_set_layouts[kResolveCopyDescriptorSetCount] = {};
resolve_copy_descriptor_set_layouts[kResolveCopyDescriptorSetEdram] =
descriptor_set_layout_storage_buffer_;
resolve_copy_descriptor_set_layouts[kResolveCopyDescriptorSetDest] =
command_processor_.GetSingleTransientDescriptorLayout(
VulkanCommandProcessor::SingleTransientDescriptorLayout ::
kStorageBufferCompute);
VkPushConstantRange resolve_copy_push_constant_range;
resolve_copy_push_constant_range.stageFlags = VK_SHADER_STAGE_COMPUTE_BIT;
resolve_copy_push_constant_range.offset = 0;
// Potentially binding all of the shared memory at 1x resolution, but only
// portions with scaled resolution.
resolve_copy_push_constant_range.size =
draw_resolution_scaled
? sizeof(draw_util::ResolveCopyShaderConstants::DestRelative)
: sizeof(draw_util::ResolveCopyShaderConstants);
VkPipelineLayoutCreateInfo resolve_copy_pipeline_layout_create_info;
resolve_copy_pipeline_layout_create_info.sType =
VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO;
resolve_copy_pipeline_layout_create_info.pNext = nullptr;
resolve_copy_pipeline_layout_create_info.flags = 0;
resolve_copy_pipeline_layout_create_info.setLayoutCount =
kResolveCopyDescriptorSetCount;
resolve_copy_pipeline_layout_create_info.pSetLayouts =
resolve_copy_descriptor_set_layouts;
resolve_copy_pipeline_layout_create_info.pushConstantRangeCount = 1;
resolve_copy_pipeline_layout_create_info.pPushConstantRanges =
&resolve_copy_push_constant_range;
if (dfn.vkCreatePipelineLayout(
device, &resolve_copy_pipeline_layout_create_info, nullptr,
&resolve_copy_pipeline_layout_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the resolve copy pipeline "
"layout");
Shutdown();
return false;
}
// Resolve copy pipelines.
for (size_t i = 0; i < size_t(draw_util::ResolveCopyShaderIndex::kCount);
++i) {
const draw_util::ResolveCopyShaderInfo& resolve_copy_shader_info =
draw_util::resolve_copy_shader_info[i];
const ResolveCopyShaderCode& resolve_copy_shader_code =
kResolveCopyShaders[i];
// Somewhat verification whether resolve_copy_shaders_ is up to date.
assert_true(resolve_copy_shader_code.unscaled &&
resolve_copy_shader_code.unscaled_size_bytes &&
resolve_copy_shader_code.scaled &&
resolve_copy_shader_code.scaled_size_bytes);
VkPipeline resolve_copy_pipeline = ui::vulkan::util::CreateComputePipeline(
vulkan_device, resolve_copy_pipeline_layout_,
draw_resolution_scaled ? resolve_copy_shader_code.scaled
: resolve_copy_shader_code.unscaled,
draw_resolution_scaled ? resolve_copy_shader_code.scaled_size_bytes
: resolve_copy_shader_code.unscaled_size_bytes);
if (resolve_copy_pipeline == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the resolve copy "
"pipeline {}",
resolve_copy_shader_info.debug_name);
Shutdown();
return false;
}
vulkan_device->SetObjectName(VK_OBJECT_TYPE_PIPELINE, resolve_copy_pipeline,
resolve_copy_shader_info.debug_name);
resolve_copy_pipelines_[i] = resolve_copy_pipeline;
}
// TODO(Triang3l): All paths (FSI).
if (path_ == Path::kHostRenderTargets) {
// Host render targets.
depth_float24_round_ = cvars::depth_float24_round;
// Host depth storing pipeline layout.
VkDescriptorSetLayout host_depth_store_descriptor_set_layouts[] = {
// Destination EDRAM storage buffer.
descriptor_set_layout_storage_buffer_,
// Source depth / stencil texture (only depth is used).
descriptor_set_layout_sampled_image_x2_,
};
VkPushConstantRange host_depth_store_push_constant_range;
host_depth_store_push_constant_range.stageFlags =
VK_SHADER_STAGE_COMPUTE_BIT;
host_depth_store_push_constant_range.offset = 0;
host_depth_store_push_constant_range.size = sizeof(HostDepthStoreConstants);
VkPipelineLayoutCreateInfo host_depth_store_pipeline_layout_create_info;
host_depth_store_pipeline_layout_create_info.sType =
VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO;
host_depth_store_pipeline_layout_create_info.pNext = nullptr;
host_depth_store_pipeline_layout_create_info.flags = 0;
host_depth_store_pipeline_layout_create_info.setLayoutCount =
uint32_t(xe::countof(host_depth_store_descriptor_set_layouts));
host_depth_store_pipeline_layout_create_info.pSetLayouts =
host_depth_store_descriptor_set_layouts;
host_depth_store_pipeline_layout_create_info.pushConstantRangeCount = 1;
host_depth_store_pipeline_layout_create_info.pPushConstantRanges =
&host_depth_store_push_constant_range;
if (dfn.vkCreatePipelineLayout(
device, &host_depth_store_pipeline_layout_create_info, nullptr,
&host_depth_store_pipeline_layout_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the host depth storing "
"pipeline layout");
Shutdown();
return false;
}
constexpr std::pair<const uint32_t*, size_t> host_depth_store_shaders[] = {
{shaders::host_depth_store_1xmsaa_cs,
sizeof(shaders::host_depth_store_1xmsaa_cs)},
{shaders::host_depth_store_2xmsaa_cs,
sizeof(shaders::host_depth_store_2xmsaa_cs)},
{shaders::host_depth_store_4xmsaa_cs,
sizeof(shaders::host_depth_store_4xmsaa_cs)},
};
for (size_t i = 0; i < xe::countof(host_depth_store_shaders); ++i) {
const std::pair<const uint32_t*, size_t> host_depth_store_shader =
host_depth_store_shaders[i];
VkPipeline host_depth_store_pipeline =
ui::vulkan::util::CreateComputePipeline(
vulkan_device, host_depth_store_pipeline_layout_,
host_depth_store_shader.first, host_depth_store_shader.second);
if (host_depth_store_pipeline == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the {}-sample host "
"depth storing pipeline",
uint32_t(1) << i);
Shutdown();
return false;
}
host_depth_store_pipelines_[i] = host_depth_store_pipeline;
}
// Transfer and clear vertex buffer, for quads of up to tile granularity.
transfer_vertex_buffer_pool_ =
std::make_unique<ui::vulkan::VulkanUploadBufferPool>(
vulkan_device, VK_BUFFER_USAGE_VERTEX_BUFFER_BIT,
std::max(ui::vulkan::VulkanUploadBufferPool::kDefaultPageSize,
sizeof(float) * 2 * 6 *
Transfer::kMaxCutoutBorderRectangles *
xenos::kEdramTileCount));
// Transfer vertex shader.
transfer_passthrough_vertex_shader_ = ui::vulkan::util::CreateShaderModule(
vulkan_device, shaders::passthrough_position_xy_vs,
sizeof(shaders::passthrough_position_xy_vs));
if (transfer_passthrough_vertex_shader_ == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the render target "
"ownership transfer vertex shader");
Shutdown();
return false;
}
// Transfer pipeline layouts.
VkDescriptorSetLayout transfer_pipeline_layout_descriptor_set_layouts
[kTransferUsedDescriptorSetCount];
VkPushConstantRange transfer_pipeline_layout_push_constant_range;
transfer_pipeline_layout_push_constant_range.stageFlags =
VK_SHADER_STAGE_FRAGMENT_BIT;
transfer_pipeline_layout_push_constant_range.offset = 0;
VkPipelineLayoutCreateInfo transfer_pipeline_layout_create_info;
transfer_pipeline_layout_create_info.sType =
VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO;
transfer_pipeline_layout_create_info.pNext = nullptr;
transfer_pipeline_layout_create_info.flags = 0;
transfer_pipeline_layout_create_info.pSetLayouts =
transfer_pipeline_layout_descriptor_set_layouts;
transfer_pipeline_layout_create_info.pPushConstantRanges =
&transfer_pipeline_layout_push_constant_range;
for (size_t i = 0; i < size_t(TransferPipelineLayoutIndex::kCount); ++i) {
const TransferPipelineLayoutInfo& transfer_pipeline_layout_info =
kTransferPipelineLayoutInfos[i];
transfer_pipeline_layout_create_info.setLayoutCount = 0;
uint32_t transfer_pipeline_layout_descriptor_sets_remaining =
transfer_pipeline_layout_info.used_descriptor_sets;
uint32_t transfer_pipeline_layout_descriptor_set_index;
while (xe::bit_scan_forward(
transfer_pipeline_layout_descriptor_sets_remaining,
&transfer_pipeline_layout_descriptor_set_index)) {
transfer_pipeline_layout_descriptor_sets_remaining &=
~(uint32_t(1) << transfer_pipeline_layout_descriptor_set_index);
VkDescriptorSetLayout transfer_pipeline_layout_descriptor_set_layout =
VK_NULL_HANDLE;
switch (TransferUsedDescriptorSet(
transfer_pipeline_layout_descriptor_set_index)) {
case kTransferUsedDescriptorSetHostDepthBuffer:
transfer_pipeline_layout_descriptor_set_layout =
descriptor_set_layout_storage_buffer_;
break;
case kTransferUsedDescriptorSetHostDepthStencilTextures:
case kTransferUsedDescriptorSetDepthStencilTextures:
transfer_pipeline_layout_descriptor_set_layout =
descriptor_set_layout_sampled_image_x2_;
break;
case kTransferUsedDescriptorSetColorTexture:
transfer_pipeline_layout_descriptor_set_layout =
descriptor_set_layout_sampled_image_;
break;
default:
assert_unhandled_case(TransferUsedDescriptorSet(
transfer_pipeline_layout_descriptor_set_index));
}
transfer_pipeline_layout_descriptor_set_layouts
[transfer_pipeline_layout_create_info.setLayoutCount++] =
transfer_pipeline_layout_descriptor_set_layout;
}
transfer_pipeline_layout_push_constant_range.size = uint32_t(
sizeof(uint32_t) *
xe::bit_count(
transfer_pipeline_layout_info.used_push_constant_dwords));
transfer_pipeline_layout_create_info.pushConstantRangeCount =
transfer_pipeline_layout_info.used_push_constant_dwords ? 1 : 0;
if (dfn.vkCreatePipelineLayout(
device, &transfer_pipeline_layout_create_info, nullptr,
&transfer_pipeline_layouts_[i]) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the render target "
"ownership transfer pipeline layout {}",
i);
Shutdown();
return false;
}
}
// Dump pipeline layouts.
VkDescriptorSetLayout
dump_pipeline_layout_descriptor_set_layouts[kDumpDescriptorSetCount];
dump_pipeline_layout_descriptor_set_layouts[kDumpDescriptorSetEdram] =
descriptor_set_layout_storage_buffer_;
dump_pipeline_layout_descriptor_set_layouts[kDumpDescriptorSetSource] =
descriptor_set_layout_sampled_image_;
VkPushConstantRange dump_pipeline_layout_push_constant_range;
dump_pipeline_layout_push_constant_range.stageFlags =
VK_SHADER_STAGE_COMPUTE_BIT;
dump_pipeline_layout_push_constant_range.offset = 0;
dump_pipeline_layout_push_constant_range.size =
sizeof(uint32_t) * kDumpPushConstantCount;
VkPipelineLayoutCreateInfo dump_pipeline_layout_create_info;
dump_pipeline_layout_create_info.sType =
VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO;
dump_pipeline_layout_create_info.pNext = nullptr;
dump_pipeline_layout_create_info.flags = 0;
dump_pipeline_layout_create_info.setLayoutCount =
uint32_t(xe::countof(dump_pipeline_layout_descriptor_set_layouts));
dump_pipeline_layout_create_info.pSetLayouts =
dump_pipeline_layout_descriptor_set_layouts;
dump_pipeline_layout_create_info.pushConstantRangeCount = 1;
dump_pipeline_layout_create_info.pPushConstantRanges =
&dump_pipeline_layout_push_constant_range;
if (dfn.vkCreatePipelineLayout(device, &dump_pipeline_layout_create_info,
nullptr, &dump_pipeline_layout_color_) !=
VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the color render target "
"dumping pipeline layout");
Shutdown();
return false;
}
dump_pipeline_layout_descriptor_set_layouts[kDumpDescriptorSetSource] =
descriptor_set_layout_sampled_image_x2_;
if (dfn.vkCreatePipelineLayout(device, &dump_pipeline_layout_create_info,
nullptr, &dump_pipeline_layout_depth_) !=
VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the depth render target "
"dumping pipeline layout");
Shutdown();
return false;
}
} else if (path_ == Path::kPixelShaderInterlock) {
// Pixel (fragment) shader interlock.
// Blending is done in linear space directly in shaders.
gamma_render_target_as_srgb_ = false;
// Always true float24 depth rounded to the nearest even.
depth_float24_round_ = true;
// The pipeline layout and the pipelines for clearing the EDRAM buffer in
// resolves.
VkPushConstantRange resolve_fsi_clear_push_constant_range;
resolve_fsi_clear_push_constant_range.stageFlags =
VK_SHADER_STAGE_COMPUTE_BIT;
resolve_fsi_clear_push_constant_range.offset = 0;
resolve_fsi_clear_push_constant_range.size =
sizeof(draw_util::ResolveClearShaderConstants);
VkPipelineLayoutCreateInfo resolve_fsi_clear_pipeline_layout_create_info;
resolve_fsi_clear_pipeline_layout_create_info.sType =
VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO;
resolve_fsi_clear_pipeline_layout_create_info.pNext = nullptr;
resolve_fsi_clear_pipeline_layout_create_info.flags = 0;
resolve_fsi_clear_pipeline_layout_create_info.setLayoutCount = 1;
resolve_fsi_clear_pipeline_layout_create_info.pSetLayouts =
&descriptor_set_layout_storage_buffer_;
resolve_fsi_clear_pipeline_layout_create_info.pushConstantRangeCount = 1;
resolve_fsi_clear_pipeline_layout_create_info.pPushConstantRanges =
&resolve_fsi_clear_push_constant_range;
if (dfn.vkCreatePipelineLayout(
device, &resolve_fsi_clear_pipeline_layout_create_info, nullptr,
&resolve_fsi_clear_pipeline_layout_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the resolve EDRAM buffer "
"clear pipeline layout");
Shutdown();
return false;
}
resolve_fsi_clear_32bpp_pipeline_ = ui::vulkan::util::CreateComputePipeline(
vulkan_device, resolve_fsi_clear_pipeline_layout_,
draw_resolution_scaled ? shaders::resolve_clear_32bpp_scaled_cs
: shaders::resolve_clear_32bpp_cs,
draw_resolution_scaled ? sizeof(shaders::resolve_clear_32bpp_scaled_cs)
: sizeof(shaders::resolve_clear_32bpp_cs));
if (resolve_fsi_clear_32bpp_pipeline_ == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the 32bpp resolve EDRAM "
"buffer clear pipeline");
Shutdown();
return false;
}
resolve_fsi_clear_64bpp_pipeline_ = ui::vulkan::util::CreateComputePipeline(
vulkan_device, resolve_fsi_clear_pipeline_layout_,
draw_resolution_scaled ? shaders::resolve_clear_64bpp_scaled_cs
: shaders::resolve_clear_64bpp_cs,
draw_resolution_scaled ? sizeof(shaders::resolve_clear_64bpp_scaled_cs)
: sizeof(shaders::resolve_clear_64bpp_cs));
if (resolve_fsi_clear_64bpp_pipeline_ == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the 64bpp resolve EDRAM "
"buffer clear pipeline");
Shutdown();
return false;
}
// Common render pass.
VkSubpassDescription fsi_subpass = {};
fsi_subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS;
// Fragment shader interlock provides synchronization and ordering within a
// subpass, create an external by-region dependency to maintain interlocking
// between passes. Framebuffer-global dependencies will be made with
// explicit barriers when the addressing of the EDRAM buffer relatively to
// the fragment coordinates is changed.
VkSubpassDependency fsi_subpass_dependencies[2];
fsi_subpass_dependencies[0].srcSubpass = VK_SUBPASS_EXTERNAL;
fsi_subpass_dependencies[0].dstSubpass = 0;
fsi_subpass_dependencies[0].srcStageMask =
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT;
fsi_subpass_dependencies[0].dstStageMask =
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT;
fsi_subpass_dependencies[0].srcAccessMask =
VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT;
fsi_subpass_dependencies[0].dstAccessMask =
VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT;
fsi_subpass_dependencies[0].dependencyFlags = VK_DEPENDENCY_BY_REGION_BIT;
fsi_subpass_dependencies[1] = fsi_subpass_dependencies[0];
std::swap(fsi_subpass_dependencies[1].srcSubpass,
fsi_subpass_dependencies[1].dstSubpass);
VkRenderPassCreateInfo fsi_render_pass_create_info;
fsi_render_pass_create_info.sType =
VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO;
fsi_render_pass_create_info.pNext = nullptr;
fsi_render_pass_create_info.flags = 0;
fsi_render_pass_create_info.attachmentCount = 0;
fsi_render_pass_create_info.pAttachments = nullptr;
fsi_render_pass_create_info.subpassCount = 1;
fsi_render_pass_create_info.pSubpasses = &fsi_subpass;
fsi_render_pass_create_info.dependencyCount =
uint32_t(xe::countof(fsi_subpass_dependencies));
fsi_render_pass_create_info.pDependencies = fsi_subpass_dependencies;
if (dfn.vkCreateRenderPass(device, &fsi_render_pass_create_info, nullptr,
&fsi_render_pass_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the fragment shader "
"interlock render backend render pass");
Shutdown();
return false;
}
// Common framebuffer.
VkFramebufferCreateInfo fsi_framebuffer_create_info;
fsi_framebuffer_create_info.sType =
VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO;
fsi_framebuffer_create_info.pNext = nullptr;
fsi_framebuffer_create_info.flags = 0;
fsi_framebuffer_create_info.renderPass = fsi_render_pass_;
fsi_framebuffer_create_info.attachmentCount = 0;
fsi_framebuffer_create_info.pAttachments = nullptr;
fsi_framebuffer_create_info.width = std::min(
xenos::kTexture2DCubeMaxWidthHeight * draw_resolution_scale_x(),
device_properties.maxFramebufferWidth);
fsi_framebuffer_create_info.height = std::min(
xenos::kTexture2DCubeMaxWidthHeight * draw_resolution_scale_y(),
device_properties.maxFramebufferHeight);
fsi_framebuffer_create_info.layers = 1;
if (dfn.vkCreateFramebuffer(device, &fsi_framebuffer_create_info, nullptr,
&fsi_framebuffer_.framebuffer) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the fragment shader "
"interlock render backend framebuffer");
Shutdown();
return false;
}
fsi_framebuffer_.host_extent.width = fsi_framebuffer_create_info.width;
fsi_framebuffer_.host_extent.height = fsi_framebuffer_create_info.height;
} else {
assert_unhandled_case(path_);
Shutdown();
return false;
}
// Reset the last update structures, to keep the defaults consistent between
// paths regardless of whether the update for the path actually modifies them.
last_update_render_pass_key_ = RenderPassKey();
last_update_render_pass_ = VK_NULL_HANDLE;
last_update_framebuffer_pitch_tiles_at_32bpp_ = 0;
std::memset(last_update_framebuffer_attachments_, 0,
sizeof(last_update_framebuffer_attachments_));
last_update_framebuffer_ = VK_NULL_HANDLE;
InitializeCommon();
return true;
}
void VulkanRenderTargetCache::Shutdown(bool from_destructor) {
const ui::vulkan::VulkanDevice* const vulkan_device =
command_processor_.GetVulkanDevice();
const ui::vulkan::VulkanDevice::Functions& dfn = vulkan_device->functions();
const VkDevice device = vulkan_device->device();
// Destroy all render targets before the descriptor set pool is destroyed -
// may happen if shutting down the VulkanRenderTargetCache by destroying it,
// so ShutdownCommon is called by the RenderTargetCache destructor, when it's
// already too late.
DestroyAllRenderTargets(true);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipeline, device,
resolve_fsi_clear_64bpp_pipeline_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipeline, device,
resolve_fsi_clear_32bpp_pipeline_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipelineLayout, device,
resolve_fsi_clear_pipeline_layout_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyFramebuffer, device,
fsi_framebuffer_.framebuffer);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyRenderPass, device,
fsi_render_pass_);
for (const auto& dump_pipeline_pair : dump_pipelines_) {
// May be null to prevent recreation attempts.
if (dump_pipeline_pair.second != VK_NULL_HANDLE) {
dfn.vkDestroyPipeline(device, dump_pipeline_pair.second, nullptr);
}
}
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipelineLayout, device,
dump_pipeline_layout_depth_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipelineLayout, device,
dump_pipeline_layout_color_);
for (const auto& transfer_pipeline_array_pair : transfer_pipelines_) {
for (VkPipeline transfer_pipeline : transfer_pipeline_array_pair.second) {
// May be null to prevent recreation attempts.
if (transfer_pipeline != VK_NULL_HANDLE) {
dfn.vkDestroyPipeline(device, transfer_pipeline, nullptr);
}
}
}
transfer_pipelines_.clear();
for (const auto& transfer_shader_pair : transfer_shaders_) {
if (transfer_shader_pair.second != VK_NULL_HANDLE) {
dfn.vkDestroyShaderModule(device, transfer_shader_pair.second, nullptr);
}
}
transfer_shaders_.clear();
for (size_t i = 0; i < size_t(TransferPipelineLayoutIndex::kCount); ++i) {
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipelineLayout, device,
transfer_pipeline_layouts_[i]);
}
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyShaderModule, device,
transfer_passthrough_vertex_shader_);
transfer_vertex_buffer_pool_.reset();
for (size_t i = 0; i < xe::countof(host_depth_store_pipelines_); ++i) {
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipeline, device,
host_depth_store_pipelines_[i]);
}
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipelineLayout, device,
host_depth_store_pipeline_layout_);
last_update_framebuffer_ = VK_NULL_HANDLE;
for (const auto& framebuffer_pair : framebuffers_) {
dfn.vkDestroyFramebuffer(device, framebuffer_pair.second.framebuffer,
nullptr);
}
framebuffers_.clear();
last_update_render_pass_ = VK_NULL_HANDLE;
for (const auto& render_pass_pair : render_passes_) {
if (render_pass_pair.second != VK_NULL_HANDLE) {
dfn.vkDestroyRenderPass(device, render_pass_pair.second, nullptr);
}
}
render_passes_.clear();
for (VkPipeline& resolve_copy_pipeline : resolve_copy_pipelines_) {
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipeline, device,
resolve_copy_pipeline);
}
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipelineLayout, device,
resolve_copy_pipeline_layout_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyDescriptorPool, device,
edram_storage_buffer_descriptor_pool_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyBuffer, device,
edram_buffer_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkFreeMemory, device,
edram_buffer_memory_);
descriptor_set_pool_sampled_image_x2_.reset();
descriptor_set_pool_sampled_image_.reset();
ui::vulkan::util::DestroyAndNullHandle(
dfn.vkDestroyDescriptorSetLayout, device,
descriptor_set_layout_sampled_image_x2_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyDescriptorSetLayout,
device,
descriptor_set_layout_sampled_image_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyDescriptorSetLayout,
device,
descriptor_set_layout_storage_buffer_);
if (!from_destructor) {
ShutdownCommon();
}
}
void VulkanRenderTargetCache::ClearCache() {
const ui::vulkan::VulkanDevice* const vulkan_device =
command_processor_.GetVulkanDevice();
const ui::vulkan::VulkanDevice::Functions& dfn = vulkan_device->functions();
const VkDevice device = vulkan_device->device();
// Framebuffer objects must be destroyed because they reference views of
// attachment images, which may be removed by the common ClearCache.
last_update_framebuffer_ = VK_NULL_HANDLE;
for (const auto& framebuffer_pair : framebuffers_) {
dfn.vkDestroyFramebuffer(device, framebuffer_pair.second.framebuffer,
nullptr);
}
framebuffers_.clear();
last_update_render_pass_ = VK_NULL_HANDLE;
for (const auto& render_pass_pair : render_passes_) {
dfn.vkDestroyRenderPass(device, render_pass_pair.second, nullptr);
}
render_passes_.clear();
RenderTargetCache::ClearCache();
}
void VulkanRenderTargetCache::CompletedSubmissionUpdated() {
if (transfer_vertex_buffer_pool_) {
transfer_vertex_buffer_pool_->Reclaim(
command_processor_.GetCompletedSubmission());
}
}
void VulkanRenderTargetCache::EndSubmission() {
if (transfer_vertex_buffer_pool_) {
transfer_vertex_buffer_pool_->FlushWrites();
}
}
bool VulkanRenderTargetCache::Resolve(const Memory& memory,
VulkanSharedMemory& shared_memory,
VulkanTextureCache& texture_cache,
uint32_t& written_address_out,
uint32_t& written_length_out) {
written_address_out = 0;
written_length_out = 0;
bool draw_resolution_scaled = IsDrawResolutionScaled();
draw_util::ResolveInfo resolve_info;
if (!draw_util::GetResolveInfo(
register_file(), memory, trace_writer_, draw_resolution_scale_x(),
draw_resolution_scale_y(), IsFixedRG16TruncatedToMinus1To1(),
IsFixedRGBA16TruncatedToMinus1To1(), resolve_info)) {
return false;
}
// Nothing to copy/clear.
if (!resolve_info.coordinate_info.width_div_8 || !resolve_info.height_div_8) {
return true;
}
const ui::vulkan::VulkanDevice* const vulkan_device =
command_processor_.GetVulkanDevice();
const ui::vulkan::VulkanDevice::Functions& dfn = vulkan_device->functions();
const VkDevice device = vulkan_device->device();
DeferredCommandBuffer& command_buffer =
command_processor_.deferred_command_buffer();
// Copying.
bool copied = false;
if (resolve_info.copy_dest_extent_length) {
if (GetPath() == Path::kHostRenderTargets) {
// Dump the current contents of the render targets owning the affected
// range to edram_buffer_.
// TODO(Triang3l): Direct host render target -> shared memory resolve
// shaders for non-converting cases.
uint32_t dump_base;
uint32_t dump_row_length_used;
uint32_t dump_rows;
uint32_t dump_pitch;
resolve_info.GetCopyEdramTileSpan(dump_base, dump_row_length_used,
dump_rows, dump_pitch);
// Scale tile parameters for resolution scaling to match resolve shader
// expectations
if (IsDrawResolutionScaled()) {
dump_row_length_used *= draw_resolution_scale_x();
dump_rows *= draw_resolution_scale_y();
dump_pitch *= draw_resolution_scale_x();
}
DumpRenderTargets(dump_base, dump_row_length_used, dump_rows, dump_pitch);
}
draw_util::ResolveCopyShaderConstants copy_shader_constants;
uint32_t copy_group_count_x, copy_group_count_y;
draw_util::ResolveCopyShaderIndex copy_shader = resolve_info.GetCopyShader(
draw_resolution_scale_x(), draw_resolution_scale_y(),
copy_shader_constants, copy_group_count_x, copy_group_count_y);
assert_true(copy_group_count_x && copy_group_count_y);
if (copy_shader != draw_util::ResolveCopyShaderIndex::kUnknown) {
const draw_util::ResolveCopyShaderInfo& copy_shader_info =
draw_util::resolve_copy_shader_info[size_t(copy_shader)];
// Make sure there is memory to write to.
bool copy_dest_committed;
// TODO(Triang3l): Resolution-scaled buffer committing.
copy_dest_committed =
shared_memory.RequestRange(resolve_info.copy_dest_extent_start,
resolve_info.copy_dest_extent_length);
if (!copy_dest_committed) {
XELOGE(
"VulkanRenderTargetCache: Failed to obtain the resolve destination "
"memory region");
} else {
// TODO(Triang3l): Switching between descriptors if exceeding
// maxStorageBufferRange.
// TODO(Triang3l): Use a single 512 MB shared memory binding if
// possible.
VkDescriptorSet descriptor_set_dest =
command_processor_.AllocateSingleTransientDescriptor(
VulkanCommandProcessor::SingleTransientDescriptorLayout ::
kStorageBufferCompute);
if (descriptor_set_dest != VK_NULL_HANDLE) {
// Write the destination descriptor.
VkDescriptorBufferInfo write_descriptor_set_dest_buffer_info;
bool scaled_buffer_ready = false;
if (draw_resolution_scaled) {
// For scaled resolve, ensure the scaled buffer exists and bind to
// it
uint32_t dest_address = resolve_info.copy_dest_base;
uint32_t dest_length = resolve_info.copy_dest_extent_start -
resolve_info.copy_dest_base +
resolve_info.copy_dest_extent_length;
// Ensure scaled resolve memory is committed
scaled_buffer_ready = true;
if (!texture_cache.EnsureScaledResolveMemoryCommittedPublic(
dest_address, dest_length)) {
XELOGE(
"Failed to commit scaled resolve memory for resolve dest at "
"0x{:08X}",
dest_address);
scaled_buffer_ready = false;
}
// Make the range current to get the buffer
if (scaled_buffer_ready &&
!texture_cache.MakeScaledResolveRangeCurrent(dest_address,
dest_length)) {
XELOGE(
"Failed to make scaled resolve range current for resolve "
"dest at 0x{:08X}",
dest_address);
scaled_buffer_ready = false;
}
// Get the current scaled buffer
VkBuffer scaled_buffer = VK_NULL_HANDLE;
if (scaled_buffer_ready) {
scaled_buffer = texture_cache.GetCurrentScaledResolveBuffer();
if (scaled_buffer == VK_NULL_HANDLE) {
XELOGE(
"No current scaled resolve buffer for resolve dest at "
"0x{:08X}",
dest_address);
scaled_buffer_ready = false;
}
}
if (scaled_buffer_ready) {
// Calculate offset within the scaled buffer
uint32_t draw_resolution_scale_area =
draw_resolution_scale_x() * draw_resolution_scale_y();
uint64_t scaled_offset =
uint64_t(dest_address) * draw_resolution_scale_area;
// Get the buffer's base offset to calculate relative offset
uint64_t buffer_relative_offset = 0;
size_t buffer_index =
texture_cache.GetScaledResolveCurrentBufferIndex();
auto* buffer_info =
texture_cache.GetScaledResolveBufferInfo(buffer_index);
if (buffer_info) {
buffer_relative_offset =
scaled_offset - buffer_info->range_start_scaled;
}
write_descriptor_set_dest_buffer_info.buffer = scaled_buffer;
write_descriptor_set_dest_buffer_info.offset =
buffer_relative_offset;
write_descriptor_set_dest_buffer_info.range =
dest_length * draw_resolution_scale_area;
}
}
if (!scaled_buffer_ready) {
// Regular unscaled resolve - write to shared memory
if (draw_resolution_scaled) {
XELOGW(
"Falling back to unscaled resolve at 0x{:08X} - scaled "
"buffer not available",
resolve_info.copy_dest_base);
}
write_descriptor_set_dest_buffer_info.buffer =
shared_memory.buffer();
write_descriptor_set_dest_buffer_info.offset =
resolve_info.copy_dest_base;
write_descriptor_set_dest_buffer_info.range =
resolve_info.copy_dest_extent_start -
resolve_info.copy_dest_base +
resolve_info.copy_dest_extent_length;
}
VkWriteDescriptorSet write_descriptor_set_dest;
write_descriptor_set_dest.sType =
VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
write_descriptor_set_dest.pNext = nullptr;
write_descriptor_set_dest.dstSet = descriptor_set_dest;
write_descriptor_set_dest.dstBinding = 0;
write_descriptor_set_dest.dstArrayElement = 0;
write_descriptor_set_dest.descriptorCount = 1;
write_descriptor_set_dest.descriptorType =
VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
write_descriptor_set_dest.pImageInfo = nullptr;
write_descriptor_set_dest.pBufferInfo =
&write_descriptor_set_dest_buffer_info;
write_descriptor_set_dest.pTexelBufferView = nullptr;
dfn.vkUpdateDescriptorSets(device, 1, &write_descriptor_set_dest, 0,
nullptr);
// Submit the resolve.
if (!scaled_buffer_ready) {
// Regular unscaled - transition shared memory for write
shared_memory.Use(VulkanSharedMemory::Usage::kComputeWrite,
std::pair<uint32_t, uint32_t>(
resolve_info.copy_dest_extent_start,
resolve_info.copy_dest_extent_length));
} else {
// Scaled - add barrier for the scaled resolve buffer
// The buffer transitions from compute shader read (texture loading)
// to compute shader write
VkBuffer scaled_buffer =
texture_cache.GetCurrentScaledResolveBuffer();
if (scaled_buffer != VK_NULL_HANDLE) {
VkBufferMemoryBarrier buffer_barrier = {};
buffer_barrier.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER;
// More specific: previous compute shader reads to compute shader
// write
buffer_barrier.srcAccessMask = VK_ACCESS_SHADER_READ_BIT;
buffer_barrier.dstAccessMask = VK_ACCESS_SHADER_WRITE_BIT;
buffer_barrier.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
buffer_barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
buffer_barrier.buffer = scaled_buffer;
buffer_barrier.offset = 0;
buffer_barrier.size = VK_WHOLE_SIZE;
command_buffer.CmdVkPipelineBarrier(
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, // From compute shader
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, // To compute shader
0, 0, nullptr, 1, &buffer_barrier, 0, nullptr);
}
}
UseEdramBuffer(EdramBufferUsage::kComputeRead);
command_processor_.BindExternalComputePipeline(
resolve_copy_pipelines_[size_t(copy_shader)]);
VkDescriptorSet descriptor_sets[kResolveCopyDescriptorSetCount] = {};
descriptor_sets[kResolveCopyDescriptorSetEdram] =
edram_storage_buffer_descriptor_set_;
descriptor_sets[kResolveCopyDescriptorSetDest] = descriptor_set_dest;
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_COMPUTE, resolve_copy_pipeline_layout_, 0,
uint32_t(xe::countof(descriptor_sets)), descriptor_sets, 0,
nullptr);
if (draw_resolution_scaled) {
command_buffer.CmdVkPushConstants(
resolve_copy_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT, 0,
sizeof(copy_shader_constants.dest_relative),
&copy_shader_constants.dest_relative);
} else {
// TODO(Triang3l): Proper dest_base in case of one 512 MB shared
// memory binding, or multiple shared memory bindings in case of
// splitting due to maxStorageBufferRange overflow.
copy_shader_constants.dest_base -=
uint32_t(write_descriptor_set_dest_buffer_info.offset);
command_buffer.CmdVkPushConstants(
resolve_copy_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT, 0,
sizeof(copy_shader_constants), &copy_shader_constants);
}
command_processor_.SubmitBarriers(true);
command_buffer.CmdVkDispatch(copy_group_count_x, copy_group_count_y,
1);
// Add barrier after writing to scaled resolve buffer
if (scaled_buffer_ready) {
VkBuffer scaled_buffer =
texture_cache.GetCurrentScaledResolveBuffer();
if (scaled_buffer != VK_NULL_HANDLE) {
VkBufferMemoryBarrier buffer_barrier = {};
buffer_barrier.sType = VK_STRUCTURE_TYPE_BUFFER_MEMORY_BARRIER;
buffer_barrier.srcAccessMask = VK_ACCESS_SHADER_WRITE_BIT;
buffer_barrier.dstAccessMask = VK_ACCESS_SHADER_READ_BIT;
buffer_barrier.srcQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
buffer_barrier.dstQueueFamilyIndex = VK_QUEUE_FAMILY_IGNORED;
buffer_barrier.buffer = scaled_buffer;
buffer_barrier.offset = 0;
buffer_barrier.size = VK_WHOLE_SIZE;
command_buffer.CmdVkPipelineBarrier(
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT, 0, 0, nullptr, 1,
&buffer_barrier, 0, nullptr);
}
}
// Invalidate textures and mark the range as scaled if needed.
texture_cache.MarkRangeAsResolved(
resolve_info.copy_dest_extent_start,
resolve_info.copy_dest_extent_length);
written_address_out = resolve_info.copy_dest_extent_start;
written_length_out = resolve_info.copy_dest_extent_length;
copied = true;
}
}
}
} else {
copied = true;
}
// Clearing.
bool cleared = false;
bool clear_depth = resolve_info.IsClearingDepth();
bool clear_color = resolve_info.IsClearingColor();
if (clear_depth || clear_color) {
switch (GetPath()) {
case Path::kHostRenderTargets: {
Transfer::Rectangle clear_rectangle;
RenderTarget* clear_render_targets[2];
// If PrepareHostRenderTargetsResolveClear returns false, may be just an
// empty region (success) or an error - don't care.
if (PrepareHostRenderTargetsResolveClear(
resolve_info, clear_rectangle, clear_render_targets[0],
clear_transfers_[0], clear_render_targets[1],
clear_transfers_[1])) {
uint64_t clear_values[2];
clear_values[0] = resolve_info.rb_depth_clear;
clear_values[1] = resolve_info.rb_color_clear |
(uint64_t(resolve_info.rb_color_clear_lo) << 32);
PerformTransfersAndResolveClears(2, clear_render_targets,
clear_transfers_, clear_values,
&clear_rectangle);
}
cleared = true;
} break;
case Path::kPixelShaderInterlock: {
UseEdramBuffer(EdramBufferUsage::kComputeWrite);
// Should be safe to only commit once (if was accessed as unordered or
// with fragment shader interlock previously - if there was nothing to
// copy, only to clear, for some reason, for instance), overlap of the
// depth and the color ranges is highly unlikely.
CommitEdramBufferShaderWrites();
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_COMPUTE, resolve_fsi_clear_pipeline_layout_,
0, 1, &edram_storage_buffer_descriptor_set_, 0, nullptr);
std::pair<uint32_t, uint32_t> clear_group_count =
resolve_info.GetClearShaderGroupCount(draw_resolution_scale_x(),
draw_resolution_scale_y());
assert_true(clear_group_count.first && clear_group_count.second);
if (clear_depth) {
command_processor_.BindExternalComputePipeline(
resolve_fsi_clear_32bpp_pipeline_);
draw_util::ResolveClearShaderConstants depth_clear_constants;
resolve_info.GetDepthClearShaderConstants(depth_clear_constants);
command_buffer.CmdVkPushConstants(
resolve_fsi_clear_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT,
0, sizeof(depth_clear_constants), &depth_clear_constants);
command_processor_.SubmitBarriers(true);
command_buffer.CmdVkDispatch(clear_group_count.first,
clear_group_count.second, 1);
}
if (clear_color) {
command_processor_.BindExternalComputePipeline(
resolve_info.color_edram_info.format_is_64bpp
? resolve_fsi_clear_64bpp_pipeline_
: resolve_fsi_clear_32bpp_pipeline_);
draw_util::ResolveClearShaderConstants color_clear_constants;
resolve_info.GetColorClearShaderConstants(color_clear_constants);
if (clear_depth) {
// Non-RT-specific constants have already been set.
command_buffer.CmdVkPushConstants(
resolve_fsi_clear_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT,
uint32_t(offsetof(draw_util::ResolveClearShaderConstants,
rt_specific)),
sizeof(color_clear_constants.rt_specific),
&color_clear_constants.rt_specific);
} else {
command_buffer.CmdVkPushConstants(
resolve_fsi_clear_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT,
0, sizeof(color_clear_constants), &color_clear_constants);
}
command_processor_.SubmitBarriers(true);
command_buffer.CmdVkDispatch(clear_group_count.first,
clear_group_count.second, 1);
}
MarkEdramBufferModified();
cleared = true;
} break;
default:
assert_unhandled_case(GetPath());
}
} else {
cleared = true;
}
return copied && cleared;
}
bool VulkanRenderTargetCache::Update(
bool is_rasterization_done, reg::RB_DEPTHCONTROL normalized_depth_control,
uint32_t normalized_color_mask, const Shader& vertex_shader) {
if (!RenderTargetCache::Update(is_rasterization_done,
normalized_depth_control,
normalized_color_mask, vertex_shader)) {
return false;
}
auto rb_surface_info = register_file().Get<reg::RB_SURFACE_INFO>();
RenderPassKey render_pass_key;
// Needed even with the fragment shader interlock render backend for passing
// the sample count to the pipeline cache.
render_pass_key.msaa_samples = rb_surface_info.msaa_samples;
switch (GetPath()) {
case Path::kHostRenderTargets: {
RenderTarget* const* depth_and_color_render_targets =
last_update_accumulated_render_targets();
PerformTransfersAndResolveClears(1 + xenos::kMaxColorRenderTargets,
depth_and_color_render_targets,
last_update_transfers());
uint32_t render_targets_are_srgb =
gamma_render_target_as_srgb_
? last_update_accumulated_color_targets_are_gamma()
: 0;
if (depth_and_color_render_targets[0]) {
render_pass_key.depth_and_color_used |= 1 << 0;
render_pass_key.depth_format =
depth_and_color_render_targets[0]->key().GetDepthFormat();
}
if (depth_and_color_render_targets[1]) {
render_pass_key.depth_and_color_used |= 1 << 1;
render_pass_key.color_0_view_format =
(render_targets_are_srgb & (1 << 0))
? xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA
: depth_and_color_render_targets[1]->key().GetColorFormat();
}
if (depth_and_color_render_targets[2]) {
render_pass_key.depth_and_color_used |= 1 << 2;
render_pass_key.color_1_view_format =
(render_targets_are_srgb & (1 << 1))
? xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA
: depth_and_color_render_targets[2]->key().GetColorFormat();
}
if (depth_and_color_render_targets[3]) {
render_pass_key.depth_and_color_used |= 1 << 3;
render_pass_key.color_2_view_format =
(render_targets_are_srgb & (1 << 2))
? xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA
: depth_and_color_render_targets[3]->key().GetColorFormat();
}
if (depth_and_color_render_targets[4]) {
render_pass_key.depth_and_color_used |= 1 << 4;
render_pass_key.color_3_view_format =
(render_targets_are_srgb & (1 << 3))
? xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA
: depth_and_color_render_targets[4]->key().GetColorFormat();
}
const Framebuffer* framebuffer = last_update_framebuffer_;
VkRenderPass render_pass = last_update_render_pass_key_ == render_pass_key
? last_update_render_pass_
: VK_NULL_HANDLE;
if (render_pass == VK_NULL_HANDLE) {
render_pass = GetHostRenderTargetsRenderPass(render_pass_key);
if (render_pass == VK_NULL_HANDLE) {
return false;
}
// Framebuffer for a different render pass needed now.
framebuffer = nullptr;
}
uint32_t pitch_tiles_at_32bpp =
((rb_surface_info.surface_pitch << uint32_t(
rb_surface_info.msaa_samples >= xenos::MsaaSamples::k4X)) +
(xenos::kEdramTileWidthSamples - 1)) /
xenos::kEdramTileWidthSamples;
if (framebuffer) {
if (last_update_framebuffer_pitch_tiles_at_32bpp_ !=
pitch_tiles_at_32bpp ||
std::memcmp(last_update_framebuffer_attachments_,
depth_and_color_render_targets,
sizeof(last_update_framebuffer_attachments_))) {
framebuffer = nullptr;
}
}
if (!framebuffer) {
framebuffer = GetHostRenderTargetsFramebuffer(
render_pass_key, pitch_tiles_at_32bpp,
depth_and_color_render_targets);
if (!framebuffer) {
return false;
}
}
// Successful update - write the new configuration.
last_update_render_pass_key_ = render_pass_key;
last_update_render_pass_ = render_pass;
last_update_framebuffer_pitch_tiles_at_32bpp_ = pitch_tiles_at_32bpp;
std::memcpy(last_update_framebuffer_attachments_,
depth_and_color_render_targets,
sizeof(last_update_framebuffer_attachments_));
last_update_framebuffer_ = framebuffer;
// Transition the used render targets.
for (uint32_t i = 0; i < 1 + xenos::kMaxColorRenderTargets; ++i) {
RenderTarget* rt = depth_and_color_render_targets[i];
if (!rt) {
continue;
}
auto& vulkan_rt = *static_cast<VulkanRenderTarget*>(rt);
VkPipelineStageFlags rt_dst_stage_mask;
VkAccessFlags rt_dst_access_mask;
VkImageLayout rt_new_layout;
VulkanRenderTarget::GetDrawUsage(i == 0, &rt_dst_stage_mask,
&rt_dst_access_mask, &rt_new_layout);
command_processor_.PushImageMemoryBarrier(
vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
i ? VK_IMAGE_ASPECT_COLOR_BIT
: (VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)),
vulkan_rt.current_stage_mask(), rt_dst_stage_mask,
vulkan_rt.current_access_mask(), rt_dst_access_mask,
vulkan_rt.current_layout(), rt_new_layout);
vulkan_rt.SetUsage(rt_dst_stage_mask, rt_dst_access_mask,
rt_new_layout);
}
} break;
case Path::kPixelShaderInterlock: {
// For FSI, only the barrier is needed - already scheduled if required.
// But the buffer will be used for FSI drawing now.
UseEdramBuffer(EdramBufferUsage::kFragmentReadWrite);
// Commit preceding unordered (but not FSI) writes like clears as they
// aren't synchronized with FSI accesses.
CommitEdramBufferShaderWrites(
EdramBufferModificationStatus::kViaUnordered);
// TODO(Triang3l): Check if this draw call modifies color or depth /
// stencil, at least coarsely, to prevent useless barriers.
MarkEdramBufferModified(
EdramBufferModificationStatus::kViaFragmentShaderInterlock);
last_update_render_pass_key_ = render_pass_key;
last_update_render_pass_ = fsi_render_pass_;
last_update_framebuffer_ = &fsi_framebuffer_;
} break;
default:
assert_unhandled_case(GetPath());
return false;
}
return true;
}
VkRenderPass VulkanRenderTargetCache::GetHostRenderTargetsRenderPass(
RenderPassKey key) {
assert_true(GetPath() == Path::kHostRenderTargets);
auto it = render_passes_.find(key);
if (it != render_passes_.end()) {
return it->second;
}
VkSampleCountFlagBits samples;
switch (key.msaa_samples) {
case xenos::MsaaSamples::k1X:
samples = VK_SAMPLE_COUNT_1_BIT;
break;
case xenos::MsaaSamples::k2X:
samples = IsMsaa2xSupported(key.depth_and_color_used != 0)
? VK_SAMPLE_COUNT_2_BIT
: VK_SAMPLE_COUNT_4_BIT;
break;
case xenos::MsaaSamples::k4X:
samples = VK_SAMPLE_COUNT_4_BIT;
break;
default:
return VK_NULL_HANDLE;
}
VkAttachmentDescription attachments[1 + xenos::kMaxColorRenderTargets];
if (key.depth_and_color_used & 0b1) {
VkAttachmentDescription& attachment = attachments[0];
attachment.flags = 0;
attachment.format = GetDepthVulkanFormat(key.depth_format);
attachment.samples = samples;
attachment.loadOp = VK_ATTACHMENT_LOAD_OP_LOAD;
attachment.storeOp = VK_ATTACHMENT_STORE_OP_STORE;
attachment.stencilLoadOp = VK_ATTACHMENT_LOAD_OP_LOAD;
attachment.stencilStoreOp = VK_ATTACHMENT_STORE_OP_STORE;
attachment.initialLayout = VulkanRenderTarget::kDepthDrawLayout;
attachment.finalLayout = VulkanRenderTarget::kDepthDrawLayout;
}
VkAttachmentReference color_attachments[xenos::kMaxColorRenderTargets];
xenos::ColorRenderTargetFormat color_formats[] = {
key.color_0_view_format,
key.color_1_view_format,
key.color_2_view_format,
key.color_3_view_format,
};
for (uint32_t i = 0; i < xenos::kMaxColorRenderTargets; ++i) {
VkAttachmentReference& color_attachment = color_attachments[i];
color_attachment.layout = VulkanRenderTarget::kColorDrawLayout;
uint32_t attachment_bit = uint32_t(1) << (1 + i);
if (!(key.depth_and_color_used & attachment_bit)) {
color_attachment.attachment = VK_ATTACHMENT_UNUSED;
continue;
}
uint32_t attachment_index =
xe::bit_count(key.depth_and_color_used & (attachment_bit - 1));
color_attachment.attachment = attachment_index;
VkAttachmentDescription& attachment = attachments[attachment_index];
attachment.flags = 0;
xenos::ColorRenderTargetFormat color_format = color_formats[i];
attachment.format =
key.color_rts_use_transfer_formats
? GetColorOwnershipTransferVulkanFormat(color_format)
: GetColorVulkanFormat(color_format);
attachment.samples = samples;
attachment.loadOp = VK_ATTACHMENT_LOAD_OP_LOAD;
attachment.storeOp = VK_ATTACHMENT_STORE_OP_STORE;
attachment.stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE;
attachment.stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE;
attachment.initialLayout = VulkanRenderTarget::kColorDrawLayout;
attachment.finalLayout = VulkanRenderTarget::kColorDrawLayout;
}
VkAttachmentReference depth_stencil_attachment;
depth_stencil_attachment.attachment =
(key.depth_and_color_used & 0b1) ? 0 : VK_ATTACHMENT_UNUSED;
depth_stencil_attachment.layout = VulkanRenderTarget::kDepthDrawLayout;
VkSubpassDescription subpass;
subpass.flags = 0;
subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS;
subpass.inputAttachmentCount = 0;
subpass.pInputAttachments = nullptr;
subpass.colorAttachmentCount =
32 - xe::lzcnt(uint32_t(key.depth_and_color_used >> 1));
subpass.pColorAttachments = color_attachments;
subpass.pResolveAttachments = nullptr;
subpass.pDepthStencilAttachment =
(key.depth_and_color_used & 0b1) ? &depth_stencil_attachment : nullptr;
subpass.preserveAttachmentCount = 0;
subpass.pPreserveAttachments = nullptr;
VkPipelineStageFlags dependency_stage_mask = 0;
VkAccessFlags dependency_access_mask = 0;
if (key.depth_and_color_used & 0b1) {
dependency_stage_mask |= VulkanRenderTarget::kDepthDrawStageMask;
dependency_access_mask |= VulkanRenderTarget::kDepthDrawAccessMask;
}
if (key.depth_and_color_used >> 1) {
dependency_stage_mask |= VulkanRenderTarget::kColorDrawStageMask;
dependency_access_mask |= VulkanRenderTarget::kColorDrawAccessMask;
}
VkSubpassDependency subpass_dependencies[2];
subpass_dependencies[0].srcSubpass = VK_SUBPASS_EXTERNAL;
subpass_dependencies[0].dstSubpass = 0;
subpass_dependencies[0].srcStageMask = dependency_stage_mask;
subpass_dependencies[0].dstStageMask = dependency_stage_mask;
subpass_dependencies[0].srcAccessMask = dependency_access_mask;
subpass_dependencies[0].dstAccessMask = dependency_access_mask;
subpass_dependencies[0].dependencyFlags = VK_DEPENDENCY_BY_REGION_BIT;
subpass_dependencies[1].srcSubpass = 0;
subpass_dependencies[1].dstSubpass = VK_SUBPASS_EXTERNAL;
subpass_dependencies[1].srcStageMask = dependency_stage_mask;
subpass_dependencies[1].dstStageMask = dependency_stage_mask;
subpass_dependencies[1].srcAccessMask = dependency_access_mask;
subpass_dependencies[1].dstAccessMask = dependency_access_mask;
subpass_dependencies[1].dependencyFlags = VK_DEPENDENCY_BY_REGION_BIT;
VkRenderPassCreateInfo render_pass_create_info;
render_pass_create_info.sType = VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO;
render_pass_create_info.pNext = nullptr;
render_pass_create_info.flags = 0;
render_pass_create_info.attachmentCount =
xe::bit_count(key.depth_and_color_used);
render_pass_create_info.pAttachments = attachments;
render_pass_create_info.subpassCount = 1;
render_pass_create_info.pSubpasses = &subpass;
render_pass_create_info.dependencyCount =
key.depth_and_color_used ? uint32_t(xe::countof(subpass_dependencies))
: 0;
render_pass_create_info.pDependencies = subpass_dependencies;
const ui::vulkan::VulkanDevice* const vulkan_device =
command_processor_.GetVulkanDevice();
const ui::vulkan::VulkanDevice::Functions& dfn = vulkan_device->functions();
const VkDevice device = vulkan_device->device();
VkRenderPass render_pass;
if (dfn.vkCreateRenderPass(device, &render_pass_create_info, nullptr,
&render_pass) != VK_SUCCESS) {
XELOGE("VulkanRenderTargetCache: Failed to create a render pass");
render_passes_.emplace(key, VK_NULL_HANDLE);
return VK_NULL_HANDLE;
}
render_passes_.emplace(key, render_pass);
return render_pass;
}
VkFormat VulkanRenderTargetCache::GetDepthVulkanFormat(
xenos::DepthRenderTargetFormat format) const {
if (format == xenos::DepthRenderTargetFormat::kD24S8 &&
depth_unorm24_vulkan_format_supported()) {
return VK_FORMAT_D24_UNORM_S8_UINT;
}
return VK_FORMAT_D32_SFLOAT_S8_UINT;
}
VkFormat VulkanRenderTargetCache::GetColorVulkanFormat(
xenos::ColorRenderTargetFormat format) const {
switch (format) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
return VK_FORMAT_R8G8B8A8_UNORM;
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA:
return gamma_render_target_as_srgb_ ? VK_FORMAT_R8G8B8A8_SRGB
: VK_FORMAT_R8G8B8A8_UNORM;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10:
return VK_FORMAT_A8B8G8R8_UNORM_PACK32;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT_AS_16_16_16_16:
return VK_FORMAT_R16G16B16A16_SFLOAT;
case xenos::ColorRenderTargetFormat::k_16_16:
// TODO(Triang3l): Fallback to float16 (disregarding clearing correctness
// likely) - possibly on render target gathering, treating them entirely
// as float16.
return VK_FORMAT_R16G16_SNORM;
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
// TODO(Triang3l): Fallback to float16 (disregarding clearing correctness
// likely) - possibly on render target gathering, treating them entirely
// as float16.
return VK_FORMAT_R16G16B16A16_SNORM;
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT:
return VK_FORMAT_R16G16_SFLOAT;
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT:
return VK_FORMAT_R16G16B16A16_SFLOAT;
case xenos::ColorRenderTargetFormat::k_32_FLOAT:
return VK_FORMAT_R32_SFLOAT;
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT:
return VK_FORMAT_R32G32_SFLOAT;
default:
assert_unhandled_case(format);
return VK_FORMAT_UNDEFINED;
}
}
VkFormat VulkanRenderTargetCache::GetColorOwnershipTransferVulkanFormat(
xenos::ColorRenderTargetFormat format, bool* is_integer_out) const {
if (is_integer_out) {
*is_integer_out = true;
}
// Floating-point numbers have NaNs that need to be propagated without
// modifications to the bit representation, and SNORM has two representations
// of -1.
switch (format) {
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT:
return VK_FORMAT_R16G16_UINT;
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT:
return VK_FORMAT_R16G16B16A16_UINT;
case xenos::ColorRenderTargetFormat::k_32_FLOAT:
return VK_FORMAT_R32_UINT;
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT:
return VK_FORMAT_R32G32_UINT;
default:
if (is_integer_out) {
*is_integer_out = false;
}
return GetColorVulkanFormat(format);
}
}
VulkanRenderTargetCache::VulkanRenderTarget::~VulkanRenderTarget() {
const ui::vulkan::VulkanDevice* const vulkan_device =
render_target_cache_.command_processor_.GetVulkanDevice();
const ui::vulkan::VulkanDevice::Functions& dfn = vulkan_device->functions();
const VkDevice device = vulkan_device->device();
ui::vulkan::SingleLayoutDescriptorSetPool& descriptor_set_pool =
key().is_depth
? *render_target_cache_.descriptor_set_pool_sampled_image_x2_
: *render_target_cache_.descriptor_set_pool_sampled_image_;
descriptor_set_pool.Free(descriptor_set_index_transfer_source_);
if (view_color_transfer_separate_ != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_color_transfer_separate_, nullptr);
}
if (view_srgb_ != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_srgb_, nullptr);
}
if (view_stencil_ != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_stencil_, nullptr);
}
if (view_depth_stencil_ != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_depth_stencil_, nullptr);
}
dfn.vkDestroyImageView(device, view_depth_color_, nullptr);
dfn.vkDestroyImage(device, image_, nullptr);
dfn.vkFreeMemory(device, memory_, nullptr);
}
uint32_t VulkanRenderTargetCache::GetMaxRenderTargetWidth() const {
const ui::vulkan::VulkanDevice::Properties& device_properties =
command_processor_.GetVulkanDevice()->properties();
return std::min(device_properties.maxFramebufferWidth,
device_properties.maxImageDimension2D);
}
uint32_t VulkanRenderTargetCache::GetMaxRenderTargetHeight() const {
const ui::vulkan::VulkanDevice::Properties& device_properties =
command_processor_.GetVulkanDevice()->properties();
return std::min(device_properties.maxFramebufferHeight,
device_properties.maxImageDimension2D);
}
RenderTargetCache::RenderTarget* VulkanRenderTargetCache::CreateRenderTarget(
RenderTargetKey key) {
const ui::vulkan::VulkanDevice* const vulkan_device =
command_processor_.GetVulkanDevice();
const ui::vulkan::VulkanDevice::Functions& dfn = vulkan_device->functions();
const VkDevice device = vulkan_device->device();
// Create the image.
VkImageCreateInfo image_create_info;
image_create_info.sType = VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO;
image_create_info.pNext = nullptr;
image_create_info.flags = 0;
image_create_info.imageType = VK_IMAGE_TYPE_2D;
image_create_info.extent.width = key.GetWidth() * draw_resolution_scale_x();
image_create_info.extent.height =
GetRenderTargetHeight(key.pitch_tiles_at_32bpp, key.msaa_samples) *
draw_resolution_scale_y();
image_create_info.extent.depth = 1;
image_create_info.mipLevels = 1;
image_create_info.arrayLayers = 1;
if (key.msaa_samples == xenos::MsaaSamples::k2X &&
!msaa_2x_attachments_supported_) {
image_create_info.samples = VK_SAMPLE_COUNT_4_BIT;
} else {
image_create_info.samples =
VkSampleCountFlagBits(uint32_t(1) << uint32_t(key.msaa_samples));
}
image_create_info.tiling = VK_IMAGE_TILING_OPTIMAL;
image_create_info.usage = VK_IMAGE_USAGE_SAMPLED_BIT;
image_create_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
image_create_info.queueFamilyIndexCount = 0;
image_create_info.pQueueFamilyIndices = nullptr;
image_create_info.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
VkFormat transfer_format;
bool is_srgb_view_needed = false;
if (key.is_depth) {
image_create_info.format = GetDepthVulkanFormat(key.GetDepthFormat());
transfer_format = image_create_info.format;
image_create_info.usage |= VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT;
} else {
xenos::ColorRenderTargetFormat color_format = key.GetColorFormat();
image_create_info.format = GetColorVulkanFormat(color_format);
transfer_format = GetColorOwnershipTransferVulkanFormat(color_format);
is_srgb_view_needed =
gamma_render_target_as_srgb_ &&
(color_format == xenos::ColorRenderTargetFormat::k_8_8_8_8 ||
color_format == xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA);
if (image_create_info.format != transfer_format || is_srgb_view_needed) {
image_create_info.flags |= VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT;
}
image_create_info.usage |= VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT;
}
if (image_create_info.format == VK_FORMAT_UNDEFINED) {
XELOGE("VulkanRenderTargetCache: Unknown {} render target format {}",
key.is_depth ? "depth" : "color", key.resource_format);
return nullptr;
}
VkImage image;
VkDeviceMemory memory;
if (!ui::vulkan::util::CreateDedicatedAllocationImage(
vulkan_device, image_create_info,
ui::vulkan::util::MemoryPurpose::kDeviceLocal, image, memory)) {
XELOGE(
"VulkanRenderTarget: Failed to create a {}x{} {}xMSAA {} render target "
"image",
image_create_info.extent.width, image_create_info.extent.height,
uint32_t(1) << uint32_t(key.msaa_samples), key.GetFormatName());
return nullptr;
}
// Create the image views.
VkImageViewCreateInfo view_create_info;
view_create_info.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO;
view_create_info.pNext = nullptr;
view_create_info.flags = 0;
view_create_info.image = image;
view_create_info.viewType = VK_IMAGE_VIEW_TYPE_2D;
view_create_info.format = image_create_info.format;
view_create_info.components.r = VK_COMPONENT_SWIZZLE_IDENTITY;
view_create_info.components.g = VK_COMPONENT_SWIZZLE_IDENTITY;
view_create_info.components.b = VK_COMPONENT_SWIZZLE_IDENTITY;
view_create_info.components.a = VK_COMPONENT_SWIZZLE_IDENTITY;
view_create_info.subresourceRange =
ui::vulkan::util::InitializeSubresourceRange(
key.is_depth ? VK_IMAGE_ASPECT_DEPTH_BIT : VK_IMAGE_ASPECT_COLOR_BIT);
VkImageView view_depth_color;
if (dfn.vkCreateImageView(device, &view_create_info, nullptr,
&view_depth_color) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTarget: Failed to create a {} view for a {}x{} {}xMSAA {} "
"render target",
key.is_depth ? "depth" : "color", image_create_info.extent.width,
image_create_info.extent.height,
uint32_t(1) << uint32_t(key.msaa_samples), key.GetFormatName());
dfn.vkDestroyImage(device, image, nullptr);
dfn.vkFreeMemory(device, memory, nullptr);
return nullptr;
}
VkImageView view_depth_stencil = VK_NULL_HANDLE;
VkImageView view_stencil = VK_NULL_HANDLE;
VkImageView view_srgb = VK_NULL_HANDLE;
VkImageView view_color_transfer_separate = VK_NULL_HANDLE;
if (key.is_depth) {
view_create_info.subresourceRange.aspectMask =
VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT;
if (dfn.vkCreateImageView(device, &view_create_info, nullptr,
&view_depth_stencil) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTarget: Failed to create a depth / stencil view for a "
"{}x{} {}xMSAA {} render target",
image_create_info.extent.width, image_create_info.extent.height,
uint32_t(1) << uint32_t(key.msaa_samples),
xenos::GetDepthRenderTargetFormatName(key.GetDepthFormat()));
dfn.vkDestroyImageView(device, view_depth_color, nullptr);
dfn.vkDestroyImage(device, image, nullptr);
dfn.vkFreeMemory(device, memory, nullptr);
return nullptr;
}
view_create_info.subresourceRange.aspectMask = VK_IMAGE_ASPECT_STENCIL_BIT;
if (dfn.vkCreateImageView(device, &view_create_info, nullptr,
&view_stencil) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTarget: Failed to create a stencil view for a {}x{} "
"{}xMSAA render target",
image_create_info.extent.width, image_create_info.extent.height,
uint32_t(1) << uint32_t(key.msaa_samples),
xenos::GetDepthRenderTargetFormatName(key.GetDepthFormat()));
dfn.vkDestroyImageView(device, view_depth_stencil, nullptr);
dfn.vkDestroyImageView(device, view_depth_color, nullptr);
dfn.vkDestroyImage(device, image, nullptr);
dfn.vkFreeMemory(device, memory, nullptr);
return nullptr;
}
} else {
if (is_srgb_view_needed) {
view_create_info.format = VK_FORMAT_R8G8B8A8_SRGB;
if (dfn.vkCreateImageView(device, &view_create_info, nullptr,
&view_srgb) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTarget: Failed to create an sRGB view for a {}x{} "
"{}xMSAA render target",
image_create_info.extent.width, image_create_info.extent.height,
uint32_t(1) << uint32_t(key.msaa_samples),
xenos::GetColorRenderTargetFormatName(key.GetColorFormat()));
dfn.vkDestroyImageView(device, view_depth_color, nullptr);
dfn.vkDestroyImage(device, image, nullptr);
dfn.vkFreeMemory(device, memory, nullptr);
return nullptr;
}
}
if (transfer_format != image_create_info.format) {
view_create_info.format = transfer_format;
if (dfn.vkCreateImageView(device, &view_create_info, nullptr,
&view_color_transfer_separate) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTarget: Failed to create a transfer view for a {}x{} "
"{}xMSAA {} render target",
image_create_info.extent.width, image_create_info.extent.height,
uint32_t(1) << uint32_t(key.msaa_samples), key.GetFormatName());
if (view_srgb != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_srgb, nullptr);
}
dfn.vkDestroyImageView(device, view_depth_color, nullptr);
dfn.vkDestroyImage(device, image, nullptr);
dfn.vkFreeMemory(device, memory, nullptr);
return nullptr;
}
}
}
ui::vulkan::SingleLayoutDescriptorSetPool& descriptor_set_pool =
key.is_depth ? *descriptor_set_pool_sampled_image_x2_
: *descriptor_set_pool_sampled_image_;
size_t descriptor_set_index_transfer_source = descriptor_set_pool.Allocate();
if (descriptor_set_index_transfer_source == SIZE_MAX) {
XELOGE(
"VulkanRenderTargetCache: Failed to allocate sampled image descriptors "
"for a {} render target",
key.is_depth ? "depth/stencil" : "color");
if (view_color_transfer_separate != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_color_transfer_separate, nullptr);
}
if (view_srgb != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_srgb, nullptr);
}
dfn.vkDestroyImageView(device, view_depth_color, nullptr);
dfn.vkDestroyImage(device, image, nullptr);
dfn.vkFreeMemory(device, memory, nullptr);
return nullptr;
}
VkDescriptorSet descriptor_set_transfer_source =
descriptor_set_pool.Get(descriptor_set_index_transfer_source);
VkWriteDescriptorSet descriptor_set_write[2];
VkDescriptorImageInfo descriptor_set_write_depth_color;
descriptor_set_write_depth_color.sampler = VK_NULL_HANDLE;
descriptor_set_write_depth_color.imageView =
view_color_transfer_separate != VK_NULL_HANDLE
? view_color_transfer_separate
: view_depth_color;
descriptor_set_write_depth_color.imageLayout =
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
descriptor_set_write[0].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
descriptor_set_write[0].pNext = nullptr;
descriptor_set_write[0].dstSet = descriptor_set_transfer_source;
descriptor_set_write[0].dstBinding = 0;
descriptor_set_write[0].dstArrayElement = 0;
descriptor_set_write[0].descriptorCount = 1;
descriptor_set_write[0].descriptorType = VK_DESCRIPTOR_TYPE_SAMPLED_IMAGE;
descriptor_set_write[0].pImageInfo = &descriptor_set_write_depth_color;
descriptor_set_write[0].pBufferInfo = nullptr;
descriptor_set_write[0].pTexelBufferView = nullptr;
VkDescriptorImageInfo descriptor_set_write_stencil;
if (key.is_depth) {
descriptor_set_write_stencil.sampler = VK_NULL_HANDLE;
descriptor_set_write_stencil.imageView = view_stencil;
descriptor_set_write_stencil.imageLayout =
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
descriptor_set_write[1].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
descriptor_set_write[1].pNext = nullptr;
descriptor_set_write[1].dstSet = descriptor_set_transfer_source;
descriptor_set_write[1].dstBinding = 1;
descriptor_set_write[1].dstArrayElement = 0;
descriptor_set_write[1].descriptorCount = 1;
descriptor_set_write[1].descriptorType = VK_DESCRIPTOR_TYPE_SAMPLED_IMAGE;
descriptor_set_write[1].pImageInfo = &descriptor_set_write_stencil;
descriptor_set_write[1].pBufferInfo = nullptr;
descriptor_set_write[1].pTexelBufferView = nullptr;
}
dfn.vkUpdateDescriptorSets(device, key.is_depth ? 2 : 1, descriptor_set_write,
0, nullptr);
return new VulkanRenderTarget(key, *this, image, memory, view_depth_color,
view_depth_stencil, view_stencil, view_srgb,
view_color_transfer_separate,
descriptor_set_index_transfer_source);
}
bool VulkanRenderTargetCache::IsHostDepthEncodingDifferent(
xenos::DepthRenderTargetFormat format) const {
// TODO(Triang3l): Conversion directly in shaders.
switch (format) {
case xenos::DepthRenderTargetFormat::kD24S8:
return !depth_unorm24_vulkan_format_supported();
case xenos::DepthRenderTargetFormat::kD24FS8:
return true;
}
return false;
}
void VulkanRenderTargetCache::RequestPixelShaderInterlockBarrier() {
if (edram_buffer_usage_ == EdramBufferUsage::kFragmentReadWrite) {
CommitEdramBufferShaderWrites();
}
}
void VulkanRenderTargetCache::GetEdramBufferUsageMasks(
EdramBufferUsage usage, VkPipelineStageFlags& stage_mask_out,
VkAccessFlags& access_mask_out) {
switch (usage) {
case EdramBufferUsage::kFragmentRead:
stage_mask_out = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT;
access_mask_out = VK_ACCESS_SHADER_READ_BIT;
break;
case EdramBufferUsage::kFragmentReadWrite:
stage_mask_out = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT;
access_mask_out = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT;
break;
case EdramBufferUsage::kComputeRead:
stage_mask_out = VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT;
access_mask_out = VK_ACCESS_SHADER_READ_BIT;
break;
case EdramBufferUsage::kComputeWrite:
stage_mask_out = VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT;
access_mask_out = VK_ACCESS_SHADER_WRITE_BIT;
break;
case EdramBufferUsage::kTransferRead:
stage_mask_out = VK_PIPELINE_STAGE_TRANSFER_BIT;
access_mask_out = VK_ACCESS_TRANSFER_READ_BIT;
break;
case EdramBufferUsage::kTransferWrite:
stage_mask_out = VK_PIPELINE_STAGE_TRANSFER_BIT;
access_mask_out = VK_ACCESS_TRANSFER_WRITE_BIT;
break;
default:
assert_unhandled_case(usage);
}
}
void VulkanRenderTargetCache::UseEdramBuffer(EdramBufferUsage new_usage) {
if (edram_buffer_usage_ == new_usage) {
return;
}
VkPipelineStageFlags src_stage_mask, dst_stage_mask;
VkAccessFlags src_access_mask, dst_access_mask;
GetEdramBufferUsageMasks(edram_buffer_usage_, src_stage_mask,
src_access_mask);
GetEdramBufferUsageMasks(new_usage, dst_stage_mask, dst_access_mask);
if (command_processor_.PushBufferMemoryBarrier(
edram_buffer_, 0, VK_WHOLE_SIZE, src_stage_mask, dst_stage_mask,
src_access_mask, dst_access_mask)) {
// Resetting edram_buffer_modification_status_ only if the barrier has been
// truly inserted.
edram_buffer_modification_status_ =
EdramBufferModificationStatus::kUnmodified;
}
edram_buffer_usage_ = new_usage;
}
void VulkanRenderTargetCache::MarkEdramBufferModified(
EdramBufferModificationStatus modification_status) {
assert_true(modification_status !=
EdramBufferModificationStatus::kUnmodified);
switch (edram_buffer_usage_) {
case EdramBufferUsage::kFragmentReadWrite:
// max because being modified via unordered access requires stricter
// synchronization than via fragment shader interlocks.
edram_buffer_modification_status_ =
std::max(edram_buffer_modification_status_, modification_status);
break;
case EdramBufferUsage::kComputeWrite:
assert_true(modification_status ==
EdramBufferModificationStatus::kViaUnordered);
modification_status = EdramBufferModificationStatus::kViaUnordered;
break;
default:
assert_always(
"While changing the usage of the EDRAM buffer before marking it as "
"modified is handled safely (but will cause spurious marking as "
"modified after the changes have been implicitly committed by the "
"usage switch), normally that shouldn't be done and is an "
"indication of architectural mistakes. Alternatively, this may "
"indicate that the usage switch has been forgotten before writing, "
"which is a clearly invalid situation.");
}
}
void VulkanRenderTargetCache::CommitEdramBufferShaderWrites(
EdramBufferModificationStatus commit_status) {
assert_true(commit_status != EdramBufferModificationStatus::kUnmodified);
if (edram_buffer_modification_status_ < commit_status) {
return;
}
VkPipelineStageFlags stage_mask;
VkAccessFlags access_mask;
GetEdramBufferUsageMasks(edram_buffer_usage_, stage_mask, access_mask);
assert_not_zero(access_mask & VK_ACCESS_SHADER_WRITE_BIT);
command_processor_.PushBufferMemoryBarrier(
edram_buffer_, 0, VK_WHOLE_SIZE, stage_mask, stage_mask, access_mask,
access_mask, VK_QUEUE_FAMILY_IGNORED, VK_QUEUE_FAMILY_IGNORED, false);
edram_buffer_modification_status_ =
EdramBufferModificationStatus::kUnmodified;
PixelShaderInterlockFullEdramBarrierPlaced();
}
const VulkanRenderTargetCache::Framebuffer*
VulkanRenderTargetCache::GetHostRenderTargetsFramebuffer(
RenderPassKey render_pass_key, uint32_t pitch_tiles_at_32bpp,
const RenderTarget* const* depth_and_color_render_targets) {
FramebufferKey key;
key.render_pass_key = render_pass_key;
key.pitch_tiles_at_32bpp = pitch_tiles_at_32bpp;
if (render_pass_key.depth_and_color_used & (1 << 0)) {
key.depth_base_tiles = depth_and_color_render_targets[0]->key().base_tiles;
}
if (render_pass_key.depth_and_color_used & (1 << 1)) {
key.color_0_base_tiles =
depth_and_color_render_targets[1]->key().base_tiles;
}
if (render_pass_key.depth_and_color_used & (1 << 2)) {
key.color_1_base_tiles =
depth_and_color_render_targets[2]->key().base_tiles;
}
if (render_pass_key.depth_and_color_used & (1 << 3)) {
key.color_2_base_tiles =
depth_and_color_render_targets[3]->key().base_tiles;
}
if (render_pass_key.depth_and_color_used & (1 << 4)) {
key.color_3_base_tiles =
depth_and_color_render_targets[4]->key().base_tiles;
}
auto it = framebuffers_.find(key);
if (it != framebuffers_.end()) {
return &it->second;
}
const ui::vulkan::VulkanDevice* const vulkan_device =
command_processor_.GetVulkanDevice();
const ui::vulkan::VulkanDevice::Functions& dfn = vulkan_device->functions();
const VkDevice device = vulkan_device->device();
const ui::vulkan::VulkanDevice::Properties& device_properties =
vulkan_device->properties();
VkRenderPass render_pass = GetHostRenderTargetsRenderPass(render_pass_key);
if (render_pass == VK_NULL_HANDLE) {
return nullptr;
}
VkImageView attachments[1 + xenos::kMaxColorRenderTargets];
uint32_t attachment_count = 0;
uint32_t depth_and_color_rts_remaining = render_pass_key.depth_and_color_used;
uint32_t rt_index;
while (xe::bit_scan_forward(depth_and_color_rts_remaining, &rt_index)) {
depth_and_color_rts_remaining &= ~(uint32_t(1) << rt_index);
const auto& vulkan_rt = *static_cast<const VulkanRenderTarget*>(
depth_and_color_render_targets[rt_index]);
VkImageView attachment;
if (rt_index) {
attachment = render_pass_key.color_rts_use_transfer_formats
? vulkan_rt.view_color_transfer()
: vulkan_rt.view_depth_color();
} else {
attachment = vulkan_rt.view_depth_stencil();
}
attachments[attachment_count++] = attachment;
}
VkFramebufferCreateInfo framebuffer_create_info;
framebuffer_create_info.sType = VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO;
framebuffer_create_info.pNext = nullptr;
framebuffer_create_info.flags = 0;
framebuffer_create_info.renderPass = render_pass;
framebuffer_create_info.attachmentCount = attachment_count;
framebuffer_create_info.pAttachments = attachments;
VkExtent2D host_extent;
if (pitch_tiles_at_32bpp) {
host_extent.width = RenderTargetKey::GetWidth(pitch_tiles_at_32bpp,
render_pass_key.msaa_samples);
host_extent.height = GetRenderTargetHeight(pitch_tiles_at_32bpp,
render_pass_key.msaa_samples);
} else {
assert_zero(render_pass_key.depth_and_color_used);
// Still needed for occlusion queries.
host_extent.width = xenos::kTexture2DCubeMaxWidthHeight;
host_extent.height = xenos::kTexture2DCubeMaxWidthHeight;
}
// Limiting to the device limit for the case of no attachments, for which
// there's no limit imposed by the sizes of the attachments that have been
// created successfully.
host_extent.width = std::min(host_extent.width * draw_resolution_scale_x(),
device_properties.maxFramebufferWidth);
host_extent.height = std::min(host_extent.height * draw_resolution_scale_y(),
device_properties.maxFramebufferHeight);
framebuffer_create_info.width = host_extent.width;
framebuffer_create_info.height = host_extent.height;
framebuffer_create_info.layers = 1;
VkFramebuffer framebuffer;
if (dfn.vkCreateFramebuffer(device, &framebuffer_create_info, nullptr,
&framebuffer) != VK_SUCCESS) {
return nullptr;
}
// Creates at a persistent location - safe to use pointers.
return &framebuffers_
.emplace(std::piecewise_construct, std::forward_as_tuple(key),
std::forward_as_tuple(framebuffer, host_extent))
.first->second;
}
VkShaderModule VulkanRenderTargetCache::GetTransferShader(
TransferShaderKey key) {
auto shader_it = transfer_shaders_.find(key);
if (shader_it != transfer_shaders_.end()) {
return shader_it->second;
}
const ui::vulkan::VulkanDevice* const vulkan_device =
command_processor_.GetVulkanDevice();
const ui::vulkan::VulkanDevice::Properties& device_properties =
vulkan_device->properties();
std::vector<spv::Id> id_vector_temp;
std::vector<unsigned int> uint_vector_temp;
SpirvBuilder builder(spv::Spv_1_0,
(SpirvShaderTranslator::kSpirvMagicToolId << 16) | 1,
nullptr);
spv::Id ext_inst_glsl_std_450 = builder.import("GLSL.std.450");
builder.addCapability(spv::CapabilityShader);
builder.setMemoryModel(spv::AddressingModelLogical, spv::MemoryModelGLSL450);
builder.setSource(spv::SourceLanguageUnknown, 0);
spv::Id type_void = builder.makeVoidType();
spv::Id type_bool = builder.makeBoolType();
spv::Id type_int = builder.makeIntType(32);
spv::Id type_int2 = builder.makeVectorType(type_int, 2);
spv::Id type_uint = builder.makeUintType(32);
spv::Id type_uint2 = builder.makeVectorType(type_uint, 2);
spv::Id type_uint4 = builder.makeVectorType(type_uint, 4);
spv::Id type_float = builder.makeFloatType(32);
spv::Id type_float2 = builder.makeVectorType(type_float, 2);
spv::Id type_float4 = builder.makeVectorType(type_float, 4);
const TransferModeInfo& mode = kTransferModes[size_t(key.mode)];
const TransferPipelineLayoutInfo& pipeline_layout_info =
kTransferPipelineLayoutInfos[size_t(mode.pipeline_layout)];
// If not dest_is_color, it's depth, or stencil bit - 40-sample columns are
// swapped as opposed to color source.
bool dest_is_color = (mode.output == TransferOutput::kColor);
xenos::ColorRenderTargetFormat dest_color_format =
xenos::ColorRenderTargetFormat(key.dest_resource_format);
xenos::DepthRenderTargetFormat dest_depth_format =
xenos::DepthRenderTargetFormat(key.dest_resource_format);
bool dest_is_64bpp =
dest_is_color && xenos::IsColorRenderTargetFormat64bpp(dest_color_format);
xenos::ColorRenderTargetFormat source_color_format =
xenos::ColorRenderTargetFormat(key.source_resource_format);
xenos::DepthRenderTargetFormat source_depth_format =
xenos::DepthRenderTargetFormat(key.source_resource_format);
// If not source_is_color, it's depth / stencil - 40-sample columns are
// swapped as opposed to color destination.
bool source_is_color = (pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetColorTextureBit) != 0;
bool source_is_64bpp;
uint32_t source_color_format_component_count;
uint32_t source_color_texture_component_mask;
bool source_color_is_uint;
spv::Id source_color_component_type;
if (source_is_color) {
assert_zero(pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetDepthStencilTexturesBit);
source_is_64bpp =
xenos::IsColorRenderTargetFormat64bpp(source_color_format);
source_color_format_component_count =
xenos::GetColorRenderTargetFormatComponentCount(source_color_format);
if (mode.output == TransferOutput::kStencilBit) {
if (source_is_64bpp && !dest_is_64bpp) {
// Need one component, but choosing from the two 32bpp halves of the
// 64bpp sample.
source_color_texture_component_mask =
0b1 | (0b1 << (source_color_format_component_count >> 1));
} else {
// Red is at least 8 bits per component in all formats.
source_color_texture_component_mask = 0b1;
}
} else {
source_color_texture_component_mask =
(uint32_t(1) << source_color_format_component_count) - 1;
}
GetColorOwnershipTransferVulkanFormat(source_color_format,
&source_color_is_uint);
source_color_component_type = source_color_is_uint ? type_uint : type_float;
} else {
source_is_64bpp = false;
source_color_format_component_count = 0;
source_color_texture_component_mask = 0;
source_color_is_uint = false;
source_color_component_type = spv::NoType;
}
std::vector<spv::Id> main_interface;
// Outputs.
bool shader_uses_stencil_reference_output =
mode.output == TransferOutput::kDepth &&
vulkan_device->extensions().ext_EXT_shader_stencil_export;
bool dest_color_is_uint = false;
uint32_t dest_color_component_count = 0;
spv::Id type_fragment_data_component = spv::NoResult;
spv::Id type_fragment_data = spv::NoResult;
spv::Id output_fragment_data = spv::NoResult;
spv::Id output_fragment_depth = spv::NoResult;
spv::Id output_fragment_stencil_ref = spv::NoResult;
switch (mode.output) {
case TransferOutput::kColor:
GetColorOwnershipTransferVulkanFormat(dest_color_format,
&dest_color_is_uint);
dest_color_component_count =
xenos::GetColorRenderTargetFormatComponentCount(dest_color_format);
type_fragment_data_component =
dest_color_is_uint ? type_uint : type_float;
type_fragment_data =
dest_color_component_count > 1
? builder.makeVectorType(type_fragment_data_component,
dest_color_component_count)
: type_fragment_data_component;
output_fragment_data = builder.createVariable(
spv::NoPrecision, spv::StorageClassOutput, type_fragment_data,
"xe_transfer_fragment_data");
builder.addDecoration(output_fragment_data, spv::DecorationLocation,
key.dest_color_rt_index);
main_interface.push_back(output_fragment_data);
break;
case TransferOutput::kDepth:
output_fragment_depth =
builder.createVariable(spv::NoPrecision, spv::StorageClassOutput,
type_float, "gl_FragDepth");
builder.addDecoration(output_fragment_depth, spv::DecorationBuiltIn,
spv::BuiltInFragDepth);
main_interface.push_back(output_fragment_depth);
if (shader_uses_stencil_reference_output) {
builder.addExtension("SPV_EXT_shader_stencil_export");
builder.addCapability(spv::CapabilityStencilExportEXT);
output_fragment_stencil_ref =
builder.createVariable(spv::NoPrecision, spv::StorageClassOutput,
type_int, "gl_FragStencilRefARB");
builder.addDecoration(output_fragment_stencil_ref,
spv::DecorationBuiltIn,
spv::BuiltInFragStencilRefEXT);
main_interface.push_back(output_fragment_stencil_ref);
}
break;
default:
break;
}
// Bindings.
// Generating SPIR-V 1.0, no need to add bindings to the entry point's
// interface until SPIR-V 1.4.
// Color source.
bool source_is_multisampled =
key.source_msaa_samples != xenos::MsaaSamples::k1X;
spv::Id source_color_texture = spv::NoResult;
if (pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetColorTextureBit) {
source_color_texture = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniformConstant,
builder.makeImageType(source_color_component_type, spv::Dim2D, false,
false, source_is_multisampled, 1,
spv::ImageFormatUnknown),
"xe_transfer_color");
builder.addDecoration(
source_color_texture, spv::DecorationDescriptorSet,
xe::bit_count(pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetColorTextureBit - 1)));
builder.addDecoration(source_color_texture, spv::DecorationBinding, 0);
}
// Depth / stencil source.
spv::Id source_depth_texture = spv::NoResult;
spv::Id source_stencil_texture = spv::NoResult;
if (pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetDepthStencilTexturesBit) {
uint32_t source_depth_stencil_descriptor_set =
xe::bit_count(pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetDepthStencilTexturesBit - 1));
// Using `depth == false` in makeImageType because comparisons are not
// required, and other values of `depth` are causing issues in drivers.
// https://github.com/microsoft/DirectXShaderCompiler/issues/1107
if (mode.output != TransferOutput::kStencilBit) {
source_depth_texture = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniformConstant,
builder.makeImageType(type_float, spv::Dim2D, false, false,
source_is_multisampled, 1,
spv::ImageFormatUnknown),
"xe_transfer_depth");
builder.addDecoration(source_depth_texture, spv::DecorationDescriptorSet,
source_depth_stencil_descriptor_set);
builder.addDecoration(source_depth_texture, spv::DecorationBinding, 0);
}
if (mode.output != TransferOutput::kDepth ||
shader_uses_stencil_reference_output) {
source_stencil_texture = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniformConstant,
builder.makeImageType(type_uint, spv::Dim2D, false, false,
source_is_multisampled, 1,
spv::ImageFormatUnknown),
"xe_transfer_stencil");
builder.addDecoration(source_stencil_texture,
spv::DecorationDescriptorSet,
source_depth_stencil_descriptor_set);
builder.addDecoration(source_stencil_texture, spv::DecorationBinding, 1);
}
}
// Host depth source buffer.
spv::Id host_depth_source_buffer = spv::NoResult;
if (pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetHostDepthBufferBit) {
id_vector_temp.clear();
id_vector_temp.push_back(builder.makeRuntimeArray(type_uint));
// Storage buffers have std430 packing, no padding to 4-component vectors.
builder.addDecoration(id_vector_temp.back(), spv::DecorationArrayStride,
sizeof(uint32_t));
spv::Id type_host_depth_source_buffer =
builder.makeStructType(id_vector_temp, "XeTransferHostDepthBuffer");
builder.addMemberName(type_host_depth_source_buffer, 0, "host_depth");
builder.addMemberDecoration(type_host_depth_source_buffer, 0,
spv::DecorationNonWritable);
builder.addMemberDecoration(type_host_depth_source_buffer, 0,
spv::DecorationOffset, 0);
// Block since SPIR-V 1.3, but since SPIR-V 1.0 is generated, it's
// BufferBlock.
builder.addDecoration(type_host_depth_source_buffer,
spv::DecorationBufferBlock);
// StorageBuffer since SPIR-V 1.3, but since SPIR-V 1.0 is generated, it's
// Uniform.
host_depth_source_buffer = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniform,
type_host_depth_source_buffer, "xe_transfer_host_depth_buffer");
builder.addDecoration(
host_depth_source_buffer, spv::DecorationDescriptorSet,
xe::bit_count(pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetHostDepthBufferBit - 1)));
builder.addDecoration(host_depth_source_buffer, spv::DecorationBinding, 0);
}
// Host depth source texture (the depth / stencil descriptor set is reused,
// but stencil is not needed).
spv::Id host_depth_source_texture = spv::NoResult;
if (pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetHostDepthStencilTexturesBit) {
host_depth_source_texture = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniformConstant,
builder.makeImageType(
type_float, spv::Dim2D, false, false,
key.host_depth_source_msaa_samples != xenos::MsaaSamples::k1X, 1,
spv::ImageFormatUnknown),
"xe_transfer_host_depth");
builder.addDecoration(
host_depth_source_texture, spv::DecorationDescriptorSet,
xe::bit_count(
pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetHostDepthStencilTexturesBit - 1)));
builder.addDecoration(host_depth_source_texture, spv::DecorationBinding, 0);
}
// Push constants.
id_vector_temp.clear();
uint32_t push_constants_member_host_depth_address = UINT32_MAX;
if (pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordHostDepthAddressBit) {
push_constants_member_host_depth_address = uint32_t(id_vector_temp.size());
id_vector_temp.push_back(type_uint);
}
uint32_t push_constants_member_address = UINT32_MAX;
if (pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordAddressBit) {
push_constants_member_address = uint32_t(id_vector_temp.size());
id_vector_temp.push_back(type_uint);
}
uint32_t push_constants_member_stencil_mask = UINT32_MAX;
if (pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordStencilMaskBit) {
push_constants_member_stencil_mask = uint32_t(id_vector_temp.size());
id_vector_temp.push_back(type_uint);
}
spv::Id push_constants = spv::NoResult;
if (!id_vector_temp.empty()) {
spv::Id type_push_constants =
builder.makeStructType(id_vector_temp, "XeTransferPushConstants");
if (pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordHostDepthAddressBit) {
assert_true(push_constants_member_host_depth_address != UINT32_MAX);
builder.addMemberName(type_push_constants,
push_constants_member_host_depth_address,
"host_depth_address");
builder.addMemberDecoration(
type_push_constants, push_constants_member_host_depth_address,
spv::DecorationOffset,
sizeof(uint32_t) *
xe::bit_count(
pipeline_layout_info.used_push_constant_dwords &
(kTransferUsedPushConstantDwordHostDepthAddressBit - 1)));
}
if (pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordAddressBit) {
assert_true(push_constants_member_address != UINT32_MAX);
builder.addMemberName(type_push_constants, push_constants_member_address,
"address");
builder.addMemberDecoration(
type_push_constants, push_constants_member_address,
spv::DecorationOffset,
sizeof(uint32_t) *
xe::bit_count(pipeline_layout_info.used_push_constant_dwords &
(kTransferUsedPushConstantDwordAddressBit - 1)));
}
if (pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordStencilMaskBit) {
assert_true(push_constants_member_stencil_mask != UINT32_MAX);
builder.addMemberName(type_push_constants,
push_constants_member_stencil_mask, "stencil_mask");
builder.addMemberDecoration(
type_push_constants, push_constants_member_stencil_mask,
spv::DecorationOffset,
sizeof(uint32_t) *
xe::bit_count(
pipeline_layout_info.used_push_constant_dwords &
(kTransferUsedPushConstantDwordStencilMaskBit - 1)));
}
builder.addDecoration(type_push_constants, spv::DecorationBlock);
push_constants = builder.createVariable(
spv::NoPrecision, spv::StorageClassPushConstant, type_push_constants,
"xe_transfer_push_constants");
}
// Coordinate inputs.
spv::Id input_fragment_coord = builder.createVariable(
spv::NoPrecision, spv::StorageClassInput, type_float4, "gl_FragCoord");
builder.addDecoration(input_fragment_coord, spv::DecorationBuiltIn,
spv::BuiltInFragCoord);
main_interface.push_back(input_fragment_coord);
spv::Id input_sample_id = spv::NoResult;
spv::Id spec_const_sample_id = spv::NoResult;
if (key.dest_msaa_samples != xenos::MsaaSamples::k1X) {
if (device_properties.sampleRateShading) {
// One draw for all samples.
builder.addCapability(spv::CapabilitySampleRateShading);
input_sample_id = builder.createVariable(
spv::NoPrecision, spv::StorageClassInput, type_int, "gl_SampleID");
builder.addDecoration(input_sample_id, spv::DecorationFlat);
builder.addDecoration(input_sample_id, spv::DecorationBuiltIn,
spv::BuiltInSampleId);
main_interface.push_back(input_sample_id);
} else {
// One sample per draw, with different sample masks.
spec_const_sample_id = builder.makeUintConstant(0, true);
builder.addName(spec_const_sample_id, "xe_transfer_sample_id");
builder.addDecoration(spec_const_sample_id, spv::DecorationSpecId, 0);
}
}
// Begin the main function.
std::vector<spv::Id> main_param_types;
std::vector<std::vector<spv::Decoration>> main_precisions;
spv::Block* main_entry;
spv::Function* main_function =
builder.makeFunctionEntry(spv::NoPrecision, type_void, "main",
main_param_types, main_precisions, &main_entry);
// Working with unsigned numbers for simplicity now, bitcasting to signed will
// be done at texture fetch.
uint32_t tile_width_samples =
xenos::kEdramTileWidthSamples * draw_resolution_scale_x();
uint32_t tile_height_samples =
xenos::kEdramTileHeightSamples * draw_resolution_scale_y();
// Split the destination pixel index into 32bpp tile and 32bpp-tile-relative
// pixel index.
// Note that division by non-power-of-two constants will include a 4-cycle
// 32*32 multiplication on AMD, even though so many bits are not needed for
// the pixel position - however, if an OpUnreachable path is inserted for the
// case when the position has upper bits set, for some reason, the code for it
// is not eliminated when compiling the shader for AMD via RenderDoc on
// Windows, as of June 2022.
uint_vector_temp.clear();
uint_vector_temp.push_back(0);
uint_vector_temp.push_back(1);
spv::Id dest_pixel_coord = builder.createUnaryOp(
spv::OpConvertFToU, type_uint2,
builder.createRvalueSwizzle(
spv::NoPrecision, type_float2,
builder.createLoad(input_fragment_coord, spv::NoPrecision),
uint_vector_temp));
spv::Id dest_pixel_x =
builder.createCompositeExtract(dest_pixel_coord, type_uint, 0);
spv::Id const_dest_tile_width_pixels = builder.makeUintConstant(
tile_width_samples >>
(uint32_t(dest_is_64bpp) +
uint32_t(key.dest_msaa_samples >= xenos::MsaaSamples::k4X)));
spv::Id dest_tile_index_x = builder.createBinOp(
spv::OpUDiv, type_uint, dest_pixel_x, const_dest_tile_width_pixels);
spv::Id dest_tile_pixel_x = builder.createBinOp(
spv::OpUMod, type_uint, dest_pixel_x, const_dest_tile_width_pixels);
spv::Id dest_pixel_y =
builder.createCompositeExtract(dest_pixel_coord, type_uint, 1);
spv::Id const_dest_tile_height_pixels = builder.makeUintConstant(
tile_height_samples >>
uint32_t(key.dest_msaa_samples >= xenos::MsaaSamples::k2X));
spv::Id dest_tile_index_y = builder.createBinOp(
spv::OpUDiv, type_uint, dest_pixel_y, const_dest_tile_height_pixels);
spv::Id dest_tile_pixel_y = builder.createBinOp(
spv::OpUMod, type_uint, dest_pixel_y, const_dest_tile_height_pixels);
assert_true(push_constants_member_address != UINT32_MAX);
id_vector_temp.clear();
id_vector_temp.push_back(
builder.makeIntConstant(int32_t(push_constants_member_address)));
spv::Id address_constant = builder.createLoad(
builder.createAccessChain(spv::StorageClassPushConstant, push_constants,
id_vector_temp),
spv::NoPrecision);
// Calculate the 32bpp tile index from its X and Y parts.
spv::Id dest_tile_index = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, address_constant,
builder.makeUintConstant(0),
builder.makeUintConstant(xenos::kEdramPitchTilesBits)),
dest_tile_index_y),
dest_tile_index_x);
// Load the destination sample index.
spv::Id dest_sample_id = spv::NoResult;
if (key.dest_msaa_samples != xenos::MsaaSamples::k1X) {
if (device_properties.sampleRateShading) {
assert_true(input_sample_id != spv::NoResult);
dest_sample_id = builder.createUnaryOp(
spv::OpBitcast, type_uint,
builder.createLoad(input_sample_id, spv::NoPrecision));
} else {
assert_true(spec_const_sample_id != spv::NoResult);
// Already uint.
dest_sample_id = spec_const_sample_id;
}
}
// Transform the destination framebuffer pixel and sample coordinates into the
// source texture pixel and sample coordinates.
// First sample bit at 4x with Vulkan standard locations - horizontal sample.
// Second sample bit at 4x with Vulkan standard locations - vertical sample.
// At 2x:
// - Native 2x: top is 1 in Vulkan, bottom is 0.
// - 2x as 4x: top is 0, bottom is 3.
spv::Id source_sample_id = dest_sample_id;
spv::Id source_tile_pixel_x = dest_tile_pixel_x;
spv::Id source_tile_pixel_y = dest_tile_pixel_y;
spv::Id source_color_half = spv::NoResult;
if (!source_is_64bpp && dest_is_64bpp) {
// 32bpp -> 64bpp, need two samples of the source.
if (key.source_msaa_samples >= xenos::MsaaSamples::k4X) {
// 32bpp -> 64bpp, 4x ->.
// Source has 32bpp halves in two adjacent samples.
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// 32bpp -> 64bpp, 4x -> 4x.
// 1 destination horizontal sample = 2 source horizontal samples.
// D p0,0 s0,0 = S p0,0 s0,0 | S p0,0 s1,0
// D p0,0 s1,0 = S p1,0 s0,0 | S p1,0 s1,0
// D p0,0 s0,1 = S p0,0 s0,1 | S p0,0 s1,1
// D p0,0 s1,1 = S p1,0 s0,1 | S p1,0 s1,1
// Thus destination horizontal sample -> source horizontal pixel,
// vertical samples are 1:1.
source_sample_id =
builder.createBinOp(spv::OpBitwiseAnd, type_uint, dest_sample_id,
builder.makeUintConstant(1 << 1));
source_tile_pixel_x = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, dest_sample_id, dest_tile_pixel_x,
builder.makeUintConstant(1), builder.makeUintConstant(31));
} else if (key.dest_msaa_samples == xenos::MsaaSamples::k2X) {
// 32bpp -> 64bpp, 4x -> 2x.
// 1 destination horizontal pixel = 2 source horizontal samples.
// D p0,0 s0 = S p0,0 s0,0 | S p0,0 s1,0
// D p0,0 s1 = S p0,0 s0,1 | S p0,0 s1,1
// D p1,0 s0 = S p1,0 s0,0 | S p1,0 s1,0
// D p1,0 s1 = S p1,0 s0,1 | S p1,0 s1,1
// Pixel index can be reused. Sample 1 (for native 2x) or 0 (for 2x as
// 4x) should become samples 01, sample 0 or 3 should become samples 23.
if (msaa_2x_attachments_supported_) {
source_sample_id = builder.createBinOp(
spv::OpShiftLeftLogical, type_uint,
builder.createBinOp(spv::OpBitwiseXor, type_uint, dest_sample_id,
builder.makeUintConstant(1)),
builder.makeUintConstant(1));
} else {
source_sample_id =
builder.createBinOp(spv::OpBitwiseAnd, type_uint, dest_sample_id,
builder.makeUintConstant(1 << 1));
}
} else {
// 32bpp -> 64bpp, 4x -> 1x.
// 1 destination horizontal pixel = 2 source horizontal samples.
// D p0,0 = S p0,0 s0,0 | S p0,0 s1,0
// D p0,1 = S p0,0 s0,1 | S p0,0 s1,1
// Horizontal pixel index can be reused. Vertical pixel 1 should
// become sample 2.
source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, builder.makeUintConstant(0),
dest_tile_pixel_y, builder.makeUintConstant(1),
builder.makeUintConstant(1));
source_tile_pixel_y =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_tile_pixel_y, builder.makeUintConstant(1));
}
} else {
// 32bpp -> 64bpp, 1x/2x ->.
// Source has 32bpp halves in two adjacent pixels.
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// 32bpp -> 64bpp, 1x/2x -> 4x.
// The X part.
// 1 destination horizontal sample = 2 source horizontal pixels.
source_tile_pixel_x = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint,
builder.createBinOp(spv::OpShiftLeftLogical, type_uint,
dest_tile_pixel_x, builder.makeUintConstant(2)),
dest_sample_id, builder.makeUintConstant(1),
builder.makeUintConstant(1));
// Y is handled by common code.
} else {
// 32bpp -> 64bpp, 1x/2x -> 1x/2x.
// The X part.
// 1 destination horizontal pixel = 2 source horizontal pixels.
source_tile_pixel_x =
builder.createBinOp(spv::OpShiftLeftLogical, type_uint,
dest_tile_pixel_x, builder.makeUintConstant(1));
// Y is handled by common code.
}
}
} else if (source_is_64bpp && !dest_is_64bpp) {
// 64bpp -> 32bpp, also the half to load.
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// 64bpp -> 32bpp, -> 4x.
// The needed half is in the destination horizontal sample index.
if (key.source_msaa_samples >= xenos::MsaaSamples::k4X) {
// 64bpp -> 32bpp, 4x -> 4x.
// D p0,0 s0,0 = S s0,0 low
// D p0,0 s1,0 = S s0,0 high
// D p1,0 s0,0 = S s1,0 low
// D p1,0 s1,0 = S s1,0 high
// Vertical pixel and sample (second bit) addressing is the same.
// However, 1 horizontal destination pixel = 1 horizontal source sample.
source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, dest_sample_id, dest_tile_pixel_x,
builder.makeUintConstant(0), builder.makeUintConstant(1));
// 2 destination horizontal samples = 1 source horizontal sample, thus
// 2 destination horizontal pixels = 1 source horizontal pixel.
source_tile_pixel_x =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_tile_pixel_x, builder.makeUintConstant(1));
} else {
// 64bpp -> 32bpp, 1x/2x -> 4x.
// 2 destination horizontal samples = 1 source horizontal pixel, thus
// 1 destination horizontal pixel = 1 source horizontal pixel. Can reuse
// horizontal pixel index.
// Y is handled by common code.
}
// Half from the destination horizontal sample index.
source_color_half =
builder.createBinOp(spv::OpBitwiseAnd, type_uint, dest_sample_id,
builder.makeUintConstant(1));
} else {
// 64bpp -> 32bpp, -> 1x/2x.
// The needed half is in the destination horizontal pixel index.
if (key.source_msaa_samples >= xenos::MsaaSamples::k4X) {
// 64bpp -> 32bpp, 4x -> 1x/2x.
// (Destination horizontal pixel >> 1) & 1 = source horizontal sample
// (first bit).
source_sample_id = builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1), builder.makeUintConstant(1));
if (key.dest_msaa_samples == xenos::MsaaSamples::k2X) {
// 64bpp -> 32bpp, 4x -> 2x.
// Destination vertical samples (1/0 in the first bit for native 2x or
// 0/1 in the second bit for 2x as 4x) = source vertical samples
// (second bit).
if (msaa_2x_attachments_supported_) {
source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, source_sample_id,
builder.createBinOp(spv::OpBitwiseXor, type_uint,
dest_sample_id,
builder.makeUintConstant(1)),
builder.makeUintConstant(1), builder.makeUintConstant(1));
} else {
source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, dest_sample_id,
source_sample_id, builder.makeUintConstant(0),
builder.makeUintConstant(1));
}
} else {
// 64bpp -> 32bpp, 4x -> 1x.
// 1 destination vertical pixel = 1 source vertical sample.
source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, source_sample_id,
source_tile_pixel_y, builder.makeUintConstant(1),
builder.makeUintConstant(1));
source_tile_pixel_y = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_y,
builder.makeUintConstant(1));
}
// 2 destination horizontal pixels = 1 source horizontal sample.
// 4 destination horizontal pixels = 1 source horizontal pixel.
source_tile_pixel_x =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_tile_pixel_x, builder.makeUintConstant(2));
} else {
// 64bpp -> 32bpp, 1x/2x -> 1x/2x.
// The X part.
// 2 destination horizontal pixels = 1 destination source pixel.
source_tile_pixel_x =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_tile_pixel_x, builder.makeUintConstant(1));
// Y is handled by common code.
}
// Half from the destination horizontal pixel index.
source_color_half =
builder.createBinOp(spv::OpBitwiseAnd, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1));
}
assert_true(source_color_half != spv::NoResult);
} else {
// Same bit count.
if (key.source_msaa_samples != key.dest_msaa_samples) {
if (key.source_msaa_samples >= xenos::MsaaSamples::k4X) {
// Same BPP, 4x -> 1x/2x.
if (key.dest_msaa_samples == xenos::MsaaSamples::k2X) {
// Same BPP, 4x -> 2x.
// Horizontal pixels to samples. Vertical sample (1/0 in the first bit
// for native 2x or 0/1 in the second bit for 2x as 4x) to second
// sample bit.
if (msaa_2x_attachments_supported_) {
source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, dest_tile_pixel_x,
builder.createBinOp(spv::OpBitwiseXor, type_uint,
dest_sample_id,
builder.makeUintConstant(1)),
builder.makeUintConstant(1), builder.makeUintConstant(31));
} else {
source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, dest_sample_id,
dest_tile_pixel_x, builder.makeUintConstant(0),
builder.makeUintConstant(1));
}
source_tile_pixel_x = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1));
} else {
// Same BPP, 4x -> 1x.
// Pixels to samples.
source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint,
builder.createBinOp(spv::OpBitwiseAnd, type_uint,
dest_tile_pixel_x,
builder.makeUintConstant(1)),
dest_tile_pixel_y, builder.makeUintConstant(1),
builder.makeUintConstant(1));
source_tile_pixel_x = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1));
source_tile_pixel_y = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_y,
builder.makeUintConstant(1));
}
} else {
// Same BPP, 1x/2x -> 1x/2x/4x (as long as they're different).
// Only the X part - Y is handled by common code.
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// Horizontal samples to pixels.
source_tile_pixel_x = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, dest_sample_id,
dest_tile_pixel_x, builder.makeUintConstant(1),
builder.makeUintConstant(31));
}
}
}
}
// Common source Y and sample index for 1x/2x AA sources, independent of bits
// per sample.
if (key.source_msaa_samples < xenos::MsaaSamples::k4X &&
key.source_msaa_samples != key.dest_msaa_samples) {
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// 1x/2x -> 4x.
if (key.source_msaa_samples == xenos::MsaaSamples::k2X) {
// 2x -> 4x.
// Vertical samples (second bit) of 4x destination to vertical sample
// (1, 0 for native 2x, or 0, 3 for 2x as 4x) of 2x source.
source_sample_id =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_sample_id, builder.makeUintConstant(1));
if (msaa_2x_attachments_supported_) {
source_sample_id = builder.createBinOp(spv::OpBitwiseXor, type_uint,
source_sample_id,
builder.makeUintConstant(1));
} else {
source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, source_sample_id,
source_sample_id, builder.makeUintConstant(1),
builder.makeUintConstant(1));
}
} else {
// 1x -> 4x.
// Vertical samples (second bit) to Y pixels.
source_tile_pixel_y = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint,
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_sample_id, builder.makeUintConstant(1)),
dest_tile_pixel_y, builder.makeUintConstant(1),
builder.makeUintConstant(31));
}
} else {
// 1x/2x -> different 1x/2x.
if (key.source_msaa_samples == xenos::MsaaSamples::k2X) {
// 2x -> 1x.
// Vertical pixels of 2x destination to vertical samples (1, 0 for
// native 2x, or 0, 3 for 2x as 4x) of 1x source.
source_sample_id =
builder.createBinOp(spv::OpBitwiseAnd, type_uint, dest_tile_pixel_y,
builder.makeUintConstant(1));
if (msaa_2x_attachments_supported_) {
source_sample_id = builder.createBinOp(spv::OpBitwiseXor, type_uint,
source_sample_id,
builder.makeUintConstant(1));
} else {
source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, source_sample_id,
source_sample_id, builder.makeUintConstant(1),
builder.makeUintConstant(1));
}
source_tile_pixel_y =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_tile_pixel_y, builder.makeUintConstant(1));
} else {
// 1x -> 2x.
// Vertical samples (1/0 in the first bit for native 2x or 0/1 in the
// second bit for 2x as 4x) of 2x destination to vertical pixels of 1x
// source.
if (msaa_2x_attachments_supported_) {
source_tile_pixel_y = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint,
builder.createBinOp(spv::OpBitwiseXor, type_uint, dest_sample_id,
builder.makeUintConstant(1)),
dest_tile_pixel_y, builder.makeUintConstant(1),
builder.makeUintConstant(31));
} else {
source_tile_pixel_y = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint,
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_sample_id, builder.makeUintConstant(1)),
dest_tile_pixel_y, builder.makeUintConstant(1),
builder.makeUintConstant(31));
}
}
}
}
uint32_t source_pixel_width_dwords_log2 =
uint32_t(key.source_msaa_samples >= xenos::MsaaSamples::k4X) +
uint32_t(source_is_64bpp);
if (source_is_color != dest_is_color) {
// Copying between color and depth / stencil - swap 40-32bpp-sample columns
// in the pixel index within the source 32bpp tile.
uint32_t source_32bpp_tile_half_pixels =
tile_width_samples >> (1 + source_pixel_width_dwords_log2);
source_tile_pixel_x = builder.createUnaryOp(
spv::OpBitcast, type_uint,
builder.createBinOp(
spv::OpIAdd, type_int,
builder.createUnaryOp(spv::OpBitcast, type_int,
source_tile_pixel_x),
builder.createTriOp(
spv::OpSelect, type_int,
builder.createBinOp(
spv::OpULessThan, builder.makeBoolType(),
source_tile_pixel_x,
builder.makeUintConstant(source_32bpp_tile_half_pixels)),
builder.makeIntConstant(int32_t(source_32bpp_tile_half_pixels)),
builder.makeIntConstant(
-int32_t(source_32bpp_tile_half_pixels)))));
}
// Transform the destination 32bpp tile index into the source. After the
// addition, it may be negative - in which case, the transfer is done across
// EDRAM addressing wrapping, and xenos::kEdramTileCount must be added to it,
// but `& (xenos::kEdramTileCount - 1)` handles that regardless of the sign.
spv::Id source_tile_index = builder.createBinOp(
spv::OpBitwiseAnd, type_uint,
builder.createUnaryOp(
spv::OpBitcast, type_uint,
builder.createBinOp(
spv::OpIAdd, type_int,
builder.createUnaryOp(spv::OpBitcast, type_int, dest_tile_index),
builder.createTriOp(
spv::OpBitFieldSExtract, type_int,
builder.createUnaryOp(spv::OpBitcast, type_int,
address_constant),
builder.makeUintConstant(xenos::kEdramPitchTilesBits * 2),
builder.makeUintConstant(xenos::kEdramBaseTilesBits + 1)))),
builder.makeUintConstant(xenos::kEdramTileCount - 1));
// Split the source 32bpp tile index into X and Y tile index within the source
// image.
spv::Id source_pitch_tiles = builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, address_constant,
builder.makeUintConstant(xenos::kEdramPitchTilesBits),
builder.makeUintConstant(xenos::kEdramPitchTilesBits));
spv::Id source_tile_index_y = builder.createBinOp(
spv::OpUDiv, type_uint, source_tile_index, source_pitch_tiles);
spv::Id source_tile_index_x = builder.createBinOp(
spv::OpUMod, type_uint, source_tile_index, source_pitch_tiles);
// Finally calculate the source texture coordinates.
spv::Id source_pixel_x_int = builder.createUnaryOp(
spv::OpBitcast, type_int,
builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.makeUintConstant(tile_width_samples >>
source_pixel_width_dwords_log2),
source_tile_index_x),
source_tile_pixel_x));
spv::Id source_pixel_y_int = builder.createUnaryOp(
spv::OpBitcast, type_int,
builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.makeUintConstant(
tile_height_samples >>
uint32_t(key.source_msaa_samples >= xenos::MsaaSamples::k2X)),
source_tile_index_y),
source_tile_pixel_y));
// Load the source.
spv::Builder::TextureParameters source_texture_parameters = {};
id_vector_temp.clear();
id_vector_temp.push_back(source_pixel_x_int);
id_vector_temp.push_back(source_pixel_y_int);
spv::Id source_coordinates[2] = {
builder.createCompositeConstruct(type_int2, id_vector_temp),
};
spv::Id source_sample_ids_int[2] = {};
if (key.source_msaa_samples != xenos::MsaaSamples::k1X) {
source_sample_ids_int[0] =
builder.createUnaryOp(spv::OpBitcast, type_int, source_sample_id);
} else {
source_texture_parameters.lod = builder.makeIntConstant(0);
}
// Go to the next sample or pixel along X if need to load two dwords.
bool source_load_is_two_32bpp_samples = !source_is_64bpp && dest_is_64bpp;
if (source_load_is_two_32bpp_samples) {
if (key.source_msaa_samples >= xenos::MsaaSamples::k4X) {
source_coordinates[1] = source_coordinates[0];
source_sample_ids_int[1] = builder.createBinOp(
spv::OpBitwiseOr, type_int, source_sample_ids_int[0],
builder.makeIntConstant(1));
} else {
id_vector_temp.clear();
id_vector_temp.push_back(builder.createBinOp(spv::OpBitwiseOr, type_int,
source_pixel_x_int,
builder.makeIntConstant(1)));
id_vector_temp.push_back(source_pixel_y_int);
source_coordinates[1] =
builder.createCompositeConstruct(type_int2, id_vector_temp);
source_sample_ids_int[1] = source_sample_ids_int[0];
}
}
spv::Id source_color[2][4] = {};
if (source_color_texture != spv::NoResult) {
source_texture_parameters.sampler =
builder.createLoad(source_color_texture, spv::NoPrecision);
assert_true(source_color_component_type != spv::NoType);
spv::Id source_color_vec4_type =
builder.makeVectorType(source_color_component_type, 4);
for (uint32_t i = 0; i <= uint32_t(source_load_is_two_32bpp_samples); ++i) {
source_texture_parameters.coords = source_coordinates[i];
source_texture_parameters.sample = source_sample_ids_int[i];
spv::Id source_color_vec4 = builder.createTextureCall(
spv::NoPrecision, source_color_vec4_type, false, true, false, false,
false, source_texture_parameters, spv::ImageOperandsMaskNone);
uint32_t source_color_components_remaining =
source_color_texture_component_mask;
uint32_t source_color_component_index;
while (xe::bit_scan_forward(source_color_components_remaining,
&source_color_component_index)) {
source_color_components_remaining &=
~(uint32_t(1) << source_color_component_index);
source_color[i][source_color_component_index] =
builder.createCompositeExtract(source_color_vec4,
source_color_component_type,
source_color_component_index);
}
}
}
spv::Id source_depth_float[2] = {};
if (source_depth_texture != spv::NoResult) {
source_texture_parameters.sampler =
builder.createLoad(source_depth_texture, spv::NoPrecision);
for (uint32_t i = 0; i <= uint32_t(source_load_is_two_32bpp_samples); ++i) {
source_texture_parameters.coords = source_coordinates[i];
source_texture_parameters.sample = source_sample_ids_int[i];
source_depth_float[i] = builder.createCompositeExtract(
builder.createTextureCall(
spv::NoPrecision, type_float4, false, true, false, false, false,
source_texture_parameters, spv::ImageOperandsMaskNone),
type_float, 0);
}
}
spv::Id source_stencil[2] = {};
if (source_stencil_texture != spv::NoResult) {
source_texture_parameters.sampler =
builder.createLoad(source_stencil_texture, spv::NoPrecision);
for (uint32_t i = 0; i <= uint32_t(source_load_is_two_32bpp_samples); ++i) {
source_texture_parameters.coords = source_coordinates[i];
source_texture_parameters.sample = source_sample_ids_int[i];
source_stencil[i] = builder.createCompositeExtract(
builder.createTextureCall(
spv::NoPrecision, type_uint4, false, true, false, false, false,
source_texture_parameters, spv::ImageOperandsMaskNone),
type_uint, 0);
}
}
// Pick the needed 32bpp half of the 64bpp color.
if (source_is_64bpp && !dest_is_64bpp) {
uint32_t source_color_half_component_count =
source_color_format_component_count >> 1;
assert_true(source_color_half != spv::NoResult);
spv::Id source_color_is_second_half =
builder.createBinOp(spv::OpINotEqual, type_bool, source_color_half,
builder.makeUintConstant(0));
if (mode.output == TransferOutput::kStencilBit) {
source_color[0][0] = builder.createTriOp(
spv::OpSelect, source_color_component_type,
source_color_is_second_half,
source_color[0][source_color_half_component_count],
source_color[0][0]);
} else {
for (uint32_t i = 0; i < source_color_half_component_count; ++i) {
source_color[0][i] = builder.createTriOp(
spv::OpSelect, source_color_component_type,
source_color_is_second_half,
source_color[0][source_color_half_component_count + i],
source_color[0][i]);
}
}
}
if (output_fragment_stencil_ref != spv::NoResult &&
source_stencil[0] != spv::NoResult) {
// For the depth -> depth case, write the stencil directly to the output.
assert_true(mode.output == TransferOutput::kDepth);
builder.createStore(
builder.createUnaryOp(spv::OpBitcast, type_int, source_stencil[0]),
output_fragment_stencil_ref);
}
if (dest_is_64bpp) {
// Construct the 64bpp color from two 32-bit samples or one 64-bit sample.
// If `packed` (two uints) are created, use the generic path involving
// unpacking.
// Otherwise, the fragment data output must be written to directly by the
// reached control flow path.
spv::Id packed[2] = {};
if (source_is_color) {
switch (source_color_format) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA: {
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale = builder.makeFloatConstant(255.0f);
spv::Id component_width = builder.makeUintConstant(8);
for (uint32_t i = 0; i < 2; ++i) {
packed[i] = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
source_color[i][0], unorm_scale),
unorm_round_offset));
for (uint32_t j = 1; j < 4; ++j) {
packed[i] = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed[i],
builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
source_color[i][j], unorm_scale),
unorm_round_offset)),
builder.makeUintConstant(8 * j), component_width);
}
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10: {
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale_rgb = builder.makeFloatConstant(1023.0f);
spv::Id width_rgb = builder.makeUintConstant(10);
spv::Id unorm_scale_a = builder.makeFloatConstant(3.0f);
spv::Id width_a = builder.makeUintConstant(2);
for (uint32_t i = 0; i < 2; ++i) {
packed[i] = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
source_color[i][0], unorm_scale_rgb),
unorm_round_offset));
for (uint32_t j = 1; j < 4; ++j) {
packed[i] = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed[i],
builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float, source_color[i][j],
j == 3 ? unorm_scale_a : unorm_scale_rgb),
unorm_round_offset)),
builder.makeUintConstant(10 * j),
j == 3 ? width_a : width_rgb);
}
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::
k_2_10_10_10_FLOAT_AS_16_16_16_16: {
spv::Id width_rgb = builder.makeUintConstant(10);
spv::Id float_0 = builder.makeFloatConstant(0.0f);
spv::Id float_1 = builder.makeFloatConstant(1.0f);
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale_a = builder.makeFloatConstant(3.0f);
spv::Id offset_a = builder.makeUintConstant(30);
spv::Id width_a = builder.makeUintConstant(2);
for (uint32_t i = 0; i < 2; ++i) {
// Float16 has a wider range for both color and alpha, also NaNs -
// clamp and convert.
packed[i] = SpirvShaderTranslator::UnclampedFloat32To7e3(
builder, source_color[i][0], ext_inst_glsl_std_450);
for (uint32_t j = 1; j < 3; ++j) {
packed[i] = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed[i],
SpirvShaderTranslator::UnclampedFloat32To7e3(
builder, source_color[i][j], ext_inst_glsl_std_450),
builder.makeUintConstant(10 * j), width_rgb);
}
// Saturate and convert the alpha.
spv::Id alpha_saturated = builder.createTriBuiltinCall(
type_float, ext_inst_glsl_std_450, GLSLstd450NClamp,
source_color[i][3], float_0, float_1);
packed[i] = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed[i],
builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
alpha_saturated, unorm_scale_a),
unorm_round_offset)),
offset_a, width_a);
}
} break;
// All 64bpp formats, and all 16 bits per component formats, are
// represented as integers in ownership transfer for safe handling of
// NaN encodings and -32768 / -32767.
// TODO(Triang3l): Handle the case when that's not true (no multisampled
// sampled images, no 16-bit UNORM, no cross-packing 32bpp aliasing on a
// portability subset device or a 64bpp format where that wouldn't help
// anyway).
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT: {
if (dest_color_format ==
xenos::ColorRenderTargetFormat::k_32_32_FLOAT) {
spv::Id component_offset_width = builder.makeUintConstant(16);
spv::Id color_16_in_32[2];
for (uint32_t i = 0; i < 2; ++i) {
color_16_in_32[i] = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, source_color[i][0],
source_color[i][1], component_offset_width,
component_offset_width);
}
id_vector_temp.clear();
id_vector_temp.push_back(color_16_in_32[0]);
id_vector_temp.push_back(color_16_in_32[1]);
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} else {
id_vector_temp.clear();
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(source_color[i >> 1][i & 1]);
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
}
} break;
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT: {
if (dest_color_format ==
xenos::ColorRenderTargetFormat::k_32_32_FLOAT) {
spv::Id component_offset_width = builder.makeUintConstant(16);
spv::Id color_16_in_32[2];
for (uint32_t i = 0; i < 2; ++i) {
color_16_in_32[i] = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, source_color[0][i << 1],
source_color[0][(i << 1) + 1], component_offset_width,
component_offset_width);
}
id_vector_temp.clear();
id_vector_temp.push_back(color_16_in_32[0]);
id_vector_temp.push_back(color_16_in_32[1]);
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} else {
id_vector_temp.clear();
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(source_color[0][i]);
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
}
} break;
// Float32 is transferred as uint32 to preserve NaN encodings. However,
// multisampled sampled image support is optional in Vulkan.
case xenos::ColorRenderTargetFormat::k_32_FLOAT: {
for (uint32_t i = 0; i < 2; ++i) {
packed[i] = source_color[i][0];
if (!source_color_is_uint) {
packed[i] =
builder.createUnaryOp(spv::OpBitcast, type_uint, packed[i]);
}
}
} break;
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT: {
for (uint32_t i = 0; i < 2; ++i) {
packed[i] = source_color[0][i];
if (!source_color_is_uint) {
packed[i] =
builder.createUnaryOp(spv::OpBitcast, type_uint, packed[i]);
}
}
} break;
}
} else {
assert_true(source_depth_texture != spv::NoResult);
assert_true(source_stencil_texture != spv::NoResult);
spv::Id depth_offset = builder.makeUintConstant(8);
spv::Id depth_width = builder.makeUintConstant(24);
for (uint32_t i = 0; i < 2; ++i) {
spv::Id depth24 = spv::NoResult;
switch (source_depth_format) {
case xenos::DepthRenderTargetFormat::kD24S8: {
// Round to the nearest even integer. This seems to be the
// correct conversion, adding +0.5 and rounding towards zero results
// in red instead of black in the 4D5307E6 clear shader.
depth24 = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createUnaryBuiltinCall(
type_float, ext_inst_glsl_std_450, GLSLstd450RoundEven,
builder.createBinOp(
spv::OpFMul, type_float, source_depth_float[i],
builder.makeFloatConstant(float(0xFFFFFF)))));
} break;
case xenos::DepthRenderTargetFormat::kD24FS8: {
depth24 = SpirvShaderTranslator::PreClampedDepthTo20e4(
builder, source_depth_float[i], depth_float24_round(), true,
ext_inst_glsl_std_450);
} break;
}
// Merge depth and stencil.
packed[i] = builder.createQuadOp(spv::OpBitFieldInsert, type_uint,
source_stencil[i], depth24,
depth_offset, depth_width);
}
}
// Common path unless there was a specialized one - unpack two packed 32-bit
// parts.
if (packed[0] != spv::NoResult) {
assert_true(packed[1] != spv::NoResult);
if (dest_color_format == xenos::ColorRenderTargetFormat::k_32_32_FLOAT) {
id_vector_temp.clear();
id_vector_temp.push_back(packed[0]);
id_vector_temp.push_back(packed[1]);
// Multisampled sampled images are optional in Vulkan, and image views
// of different formats can't be created separately for sampled image
// and color attachment usages, so no multisampled integer sampled image
// support implies no multisampled integer framebuffer attachment
// support in Xenia.
if (!dest_color_is_uint) {
for (spv::Id& float32 : id_vector_temp) {
float32 =
builder.createUnaryOp(spv::OpBitcast, type_float, float32);
}
}
builder.createStore(builder.createCompositeConstruct(type_fragment_data,
id_vector_temp),
output_fragment_data);
} else {
spv::Id const_uint_0 = builder.makeUintConstant(0);
spv::Id const_uint_16 = builder.makeUintConstant(16);
id_vector_temp.clear();
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, packed[i >> 1],
(i & 1) ? const_uint_16 : const_uint_0, const_uint_16));
}
// TODO(Triang3l): Handle the case when that's not true (no multisampled
// sampled images, no 16-bit UNORM, no cross-packing 32bpp aliasing on a
// portability subset device or a 64bpp format where that wouldn't help
// anyway).
builder.createStore(builder.createCompositeConstruct(type_fragment_data,
id_vector_temp),
output_fragment_data);
}
}
} else {
// If `packed` is created, use the generic path involving unpacking.
// - For a color destination, the packed 32bpp color.
// - For a depth / stencil destination, stencil in 0:7, depth in 8:31
// normally, or depth in 0:23 and zeros in 24:31 with packed_only_depth.
// - For a stencil bit, stencil in 0:7.
// Otherwise, the fragment data or fragment depth / stencil output must be
// written to directly by the reached control flow path.
spv::Id packed = spv::NoResult;
bool packed_only_depth = false;
if (source_is_color) {
switch (source_color_format) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA: {
if (dest_is_color &&
(dest_color_format == xenos::ColorRenderTargetFormat::k_8_8_8_8 ||
dest_color_format ==
xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA)) {
// Same format - passthrough.
id_vector_temp.clear();
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(source_color[0][i]);
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} else {
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale = builder.makeFloatConstant(255.0f);
uint32_t packed_component_offset = 0;
if (mode.output == TransferOutput::kDepth) {
// When need only depth, not stencil, skip the red component, and
// put the depth from GBA directly in the lower bits.
packed_component_offset = 1;
packed_only_depth = true;
if (output_fragment_stencil_ref != spv::NoResult) {
builder.createStore(
builder.createUnaryOp(
spv::OpBitcast, type_int,
builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
source_color[0][0],
unorm_scale),
unorm_round_offset))),
output_fragment_stencil_ref);
}
}
packed = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float,
source_color[0][packed_component_offset], unorm_scale),
unorm_round_offset));
if (mode.output != TransferOutput::kStencilBit) {
spv::Id component_width = builder.makeUintConstant(8);
for (uint32_t i = 1; i < 4 - packed_component_offset; ++i) {
packed = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed,
builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float,
source_color[0][packed_component_offset + i],
unorm_scale),
unorm_round_offset)),
builder.makeUintConstant(8 * i), component_width);
}
}
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10: {
if (dest_is_color &&
(dest_color_format ==
xenos::ColorRenderTargetFormat::k_2_10_10_10 ||
dest_color_format == xenos::ColorRenderTargetFormat::
k_2_10_10_10_AS_10_10_10_10)) {
id_vector_temp.clear();
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(source_color[0][i]);
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} else {
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale_rgb = builder.makeFloatConstant(1023.0f);
packed = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
source_color[0][0], unorm_scale_rgb),
unorm_round_offset));
if (mode.output != TransferOutput::kStencilBit) {
spv::Id width_rgb = builder.makeUintConstant(10);
spv::Id unorm_scale_a = builder.makeFloatConstant(3.0f);
spv::Id width_a = builder.makeUintConstant(2);
for (uint32_t i = 1; i < 4; ++i) {
packed = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed,
builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float, source_color[0][i],
i == 3 ? unorm_scale_a : unorm_scale_rgb),
unorm_round_offset)),
builder.makeUintConstant(10 * i),
i == 3 ? width_a : width_rgb);
}
}
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::
k_2_10_10_10_FLOAT_AS_16_16_16_16: {
if (dest_is_color &&
(dest_color_format ==
xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT ||
dest_color_format == xenos::ColorRenderTargetFormat::
k_2_10_10_10_FLOAT_AS_16_16_16_16)) {
id_vector_temp.clear();
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(source_color[0][i]);
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} else {
// Float16 has a wider range for both color and alpha, also NaNs -
// clamp and convert.
packed = SpirvShaderTranslator::UnclampedFloat32To7e3(
builder, source_color[0][0], ext_inst_glsl_std_450);
if (mode.output != TransferOutput::kStencilBit) {
spv::Id width_rgb = builder.makeUintConstant(10);
for (uint32_t i = 1; i < 3; ++i) {
packed = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed,
SpirvShaderTranslator::UnclampedFloat32To7e3(
builder, source_color[0][i], ext_inst_glsl_std_450),
builder.makeUintConstant(10 * i), width_rgb);
}
// Saturate and convert the alpha.
spv::Id alpha_saturated = builder.createTriBuiltinCall(
type_float, ext_inst_glsl_std_450, GLSLstd450NClamp,
source_color[0][3], builder.makeFloatConstant(0.0f),
builder.makeFloatConstant(1.0f));
packed = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed,
builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
alpha_saturated,
builder.makeFloatConstant(3.0f)),
builder.makeFloatConstant(0.5f))),
builder.makeUintConstant(30), builder.makeUintConstant(2));
}
}
} break;
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT:
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT: {
// All 64bpp formats, and all 16 bits per component formats, are
// represented as integers in ownership transfer for safe handling of
// NaN encodings and -32768 / -32767.
// TODO(Triang3l): Handle the case when that's not true (no
// multisampled sampled images, no 16-bit UNORM, no cross-packing
// 32bpp aliasing on a portability subset device or a 64bpp format
// where that wouldn't help anyway).
if (dest_is_color &&
(dest_color_format == xenos::ColorRenderTargetFormat::k_16_16 ||
dest_color_format ==
xenos::ColorRenderTargetFormat::k_16_16_FLOAT)) {
id_vector_temp.clear();
for (uint32_t i = 0; i < 2; ++i) {
id_vector_temp.push_back(source_color[0][i]);
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} else {
packed = source_color[0][0];
if (mode.output != TransferOutput::kStencilBit) {
spv::Id component_offset_width = builder.makeUintConstant(16);
packed = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed, source_color[0][1],
component_offset_width, component_offset_width);
}
}
} break;
// Float32 is transferred as uint32 to preserve NaN encodings. However,
// multisampled sampled image support is optional in Vulkan.
case xenos::ColorRenderTargetFormat::k_32_FLOAT:
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT: {
packed = source_color[0][0];
if (!source_color_is_uint) {
packed = builder.createUnaryOp(spv::OpBitcast, type_uint, packed);
}
} break;
}
} else if (source_depth_float[0] != spv::NoResult) {
if (mode.output == TransferOutput::kDepth &&
dest_depth_format == source_depth_format) {
builder.createStore(source_depth_float[0], output_fragment_depth);
} else {
switch (source_depth_format) {
case xenos::DepthRenderTargetFormat::kD24S8: {
// Round to the nearest even integer. This seems to be the correct
// conversion, adding +0.5 and rounding towards zero results in red
// instead of black in the 4D5307E6 clear shader.
packed = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createUnaryBuiltinCall(
type_float, ext_inst_glsl_std_450, GLSLstd450RoundEven,
builder.createBinOp(
spv::OpFMul, type_float, source_depth_float[0],
builder.makeFloatConstant(float(0xFFFFFF)))));
} break;
case xenos::DepthRenderTargetFormat::kD24FS8: {
packed = SpirvShaderTranslator::PreClampedDepthTo20e4(
builder, source_depth_float[0], depth_float24_round(), true,
ext_inst_glsl_std_450);
} break;
}
if (mode.output == TransferOutput::kDepth) {
packed_only_depth = true;
} else {
// Merge depth and stencil.
packed = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, source_stencil[0], packed,
builder.makeUintConstant(8), builder.makeUintConstant(24));
}
}
}
switch (mode.output) {
case TransferOutput::kColor: {
// Unless a special path was taken, unpack the raw 32bpp value into the
// 32bpp color output.
if (packed != spv::NoResult) {
switch (dest_color_format) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA: {
spv::Id component_width = builder.makeUintConstant(8);
spv::Id unorm_scale = builder.makeFloatConstant(1.0f / 255.0f);
id_vector_temp.clear();
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(builder.createBinOp(
spv::OpFMul, type_float,
builder.createUnaryOp(
spv::OpConvertUToF, type_float,
builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, packed,
builder.makeUintConstant(8 * i), component_width)),
unorm_scale));
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10: {
spv::Id width_rgb = builder.makeUintConstant(10);
spv::Id unorm_scale_rgb =
builder.makeFloatConstant(1.0f / 1023.0f);
spv::Id width_a = builder.makeUintConstant(2);
spv::Id unorm_scale_a = builder.makeFloatConstant(1.0f / 3.0f);
id_vector_temp.clear();
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(builder.createBinOp(
spv::OpFMul, type_float,
builder.createUnaryOp(
spv::OpConvertUToF, type_float,
builder.createTriOp(spv::OpBitFieldUExtract, type_uint,
packed,
builder.makeUintConstant(10 * i),
i == 3 ? width_a : width_rgb)),
i == 3 ? unorm_scale_a : unorm_scale_rgb));
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::
k_2_10_10_10_FLOAT_AS_16_16_16_16: {
id_vector_temp.clear();
// Color.
spv::Id width_rgb = builder.makeUintConstant(10);
for (uint32_t i = 0; i < 3; ++i) {
id_vector_temp.push_back(SpirvShaderTranslator::Float7e3To32(
builder, packed, 10 * i, false, ext_inst_glsl_std_450));
}
// Alpha.
id_vector_temp.push_back(builder.createBinOp(
spv::OpFMul, type_float,
builder.createUnaryOp(
spv::OpConvertUToF, type_float,
builder.createTriOp(spv::OpBitFieldUExtract, type_uint,
packed, builder.makeUintConstant(30),
builder.makeUintConstant(2))),
builder.makeFloatConstant(1.0f / 3.0f)));
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} break;
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT: {
// All 16 bits per component formats are represented as integers
// in ownership transfer for safe handling of NaN encodings and
// -32768 / -32767.
// TODO(Triang3l): Handle the case when that's not true (no
// multisampled sampled images, no 16-bit UNORM, no cross-packing
// 32bpp aliasing on a portability subset device or a 64bpp format
// where that wouldn't help anyway).
spv::Id component_offset_width = builder.makeUintConstant(16);
id_vector_temp.clear();
for (uint32_t i = 0; i < 2; ++i) {
id_vector_temp.push_back(builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, packed,
i ? component_offset_width : builder.makeUintConstant(0),
component_offset_width));
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} break;
case xenos::ColorRenderTargetFormat::k_32_FLOAT: {
// Float32 is transferred as uint32 to preserve NaN encodings.
// However, multisampled sampled images are optional in Vulkan,
// and image views of different formats can't be created
// separately for sampled image and color attachment usages, so no
// multisampled integer sampled image support implies no
// multisampled integer framebuffer attachment support in Xenia.
spv::Id float32 = packed;
if (!dest_color_is_uint) {
float32 =
builder.createUnaryOp(spv::OpBitcast, type_float, float32);
}
builder.createStore(float32, output_fragment_data);
} break;
default:
// A 64bpp format (handled separately) or an invalid one.
assert_unhandled_case(dest_color_format);
}
}
} break;
case TransferOutput::kDepth: {
if (packed) {
spv::Id guest_depth24 = packed;
if (!packed_only_depth) {
// Extract the depth bits.
guest_depth24 =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
guest_depth24, builder.makeUintConstant(8));
}
// Load the host float32 depth, check if, when converted to the guest
// format, it's the same as the guest source, thus up to date, and if
// it is, write host float32 depth, otherwise do the guest -> host
// conversion.
spv::Id host_depth32 = spv::NoResult;
if (host_depth_source_texture != spv::NoResult) {
// Convert position and sample index from within the destination
// tile to within the host depth source tile, like for the guest
// render target, but for 32bpp -> 32bpp only.
spv::Id host_depth_source_sample_id = dest_sample_id;
spv::Id host_depth_source_tile_pixel_x = dest_tile_pixel_x;
spv::Id host_depth_source_tile_pixel_y = dest_tile_pixel_y;
if (key.host_depth_source_msaa_samples != key.dest_msaa_samples) {
if (key.host_depth_source_msaa_samples >=
xenos::MsaaSamples::k4X) {
// 4x -> 1x/2x.
if (key.dest_msaa_samples == xenos::MsaaSamples::k2X) {
// 4x -> 2x.
// Horizontal pixels to samples. Vertical sample (1/0 in the
// first bit for native 2x or 0/1 in the second bit for 2x as
// 4x) to second sample bit.
if (msaa_2x_attachments_supported_) {
host_depth_source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, dest_tile_pixel_x,
builder.createBinOp(spv::OpBitwiseXor, type_uint,
dest_sample_id,
builder.makeUintConstant(1)),
builder.makeUintConstant(1),
builder.makeUintConstant(31));
} else {
host_depth_source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, dest_sample_id,
dest_tile_pixel_x, builder.makeUintConstant(0),
builder.makeUintConstant(1));
}
host_depth_source_tile_pixel_x = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1));
} else {
// 4x -> 1x.
// Pixels to samples.
host_depth_source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint,
builder.createBinOp(spv::OpBitwiseAnd, type_uint,
dest_tile_pixel_x,
builder.makeUintConstant(1)),
dest_tile_pixel_y, builder.makeUintConstant(1),
builder.makeUintConstant(1));
host_depth_source_tile_pixel_x = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1));
host_depth_source_tile_pixel_y = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_y,
builder.makeUintConstant(1));
}
} else {
// 1x/2x -> 1x/2x/4x (as long as they're different).
// Only the X part - Y is handled by common code.
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// Horizontal samples to pixels.
host_depth_source_tile_pixel_x = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, dest_sample_id,
dest_tile_pixel_x, builder.makeUintConstant(1),
builder.makeUintConstant(31));
}
}
// Host depth source Y and sample index for 1x/2x AA sources.
if (key.host_depth_source_msaa_samples <
xenos::MsaaSamples::k4X) {
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// 1x/2x -> 4x.
if (key.host_depth_source_msaa_samples ==
xenos::MsaaSamples::k2X) {
// 2x -> 4x.
// Vertical samples (second bit) of 4x destination to
// vertical sample (1, 0 for native 2x, or 0, 3 for 2x as
// 4x) of 2x source.
host_depth_source_sample_id = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_sample_id,
builder.makeUintConstant(1));
if (msaa_2x_attachments_supported_) {
host_depth_source_sample_id =
builder.createBinOp(spv::OpBitwiseXor, type_uint,
host_depth_source_sample_id,
builder.makeUintConstant(1));
} else {
host_depth_source_sample_id =
builder.createQuadOp(spv::OpBitFieldInsert, type_uint,
host_depth_source_sample_id,
host_depth_source_sample_id,
builder.makeUintConstant(1),
builder.makeUintConstant(1));
}
} else {
// 1x -> 4x.
// Vertical samples (second bit) to Y pixels.
host_depth_source_tile_pixel_y = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint,
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_sample_id,
builder.makeUintConstant(1)),
dest_tile_pixel_y, builder.makeUintConstant(1),
builder.makeUintConstant(31));
}
} else {
// 1x/2x -> different 1x/2x.
if (key.host_depth_source_msaa_samples ==
xenos::MsaaSamples::k2X) {
// 2x -> 1x.
// Vertical pixels of 2x destination to vertical samples (1,
// 0 for native 2x, or 0, 3 for 2x as 4x) of 1x source.
host_depth_source_sample_id = builder.createBinOp(
spv::OpBitwiseAnd, type_uint, dest_tile_pixel_y,
builder.makeUintConstant(1));
if (msaa_2x_attachments_supported_) {
host_depth_source_sample_id =
builder.createBinOp(spv::OpBitwiseXor, type_uint,
host_depth_source_sample_id,
builder.makeUintConstant(1));
} else {
host_depth_source_sample_id =
builder.createQuadOp(spv::OpBitFieldInsert, type_uint,
host_depth_source_sample_id,
host_depth_source_sample_id,
builder.makeUintConstant(1),
builder.makeUintConstant(1));
}
host_depth_source_tile_pixel_y = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_y,
builder.makeUintConstant(1));
} else {
// 1x -> 2x.
// Vertical samples (1/0 in the first bit for native 2x or
// 0/1 in the second bit for 2x as 4x) of 2x destination to
// vertical pixels of 1x source.
if (msaa_2x_attachments_supported_) {
host_depth_source_tile_pixel_y = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint,
builder.createBinOp(spv::OpBitwiseXor, type_uint,
dest_sample_id,
builder.makeUintConstant(1)),
dest_tile_pixel_y, builder.makeUintConstant(1),
builder.makeUintConstant(31));
} else {
host_depth_source_tile_pixel_y = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint,
builder.createBinOp(spv::OpShiftRightLogical,
type_uint, dest_sample_id,
builder.makeUintConstant(1)),
dest_tile_pixel_y, builder.makeUintConstant(1),
builder.makeUintConstant(31));
}
}
}
}
}
assert_true(push_constants_member_host_depth_address != UINT32_MAX);
id_vector_temp.clear();
id_vector_temp.push_back(builder.makeIntConstant(
int32_t(push_constants_member_host_depth_address)));
spv::Id host_depth_address_constant = builder.createLoad(
builder.createAccessChain(spv::StorageClassPushConstant,
push_constants, id_vector_temp),
spv::NoPrecision);
// Transform the destination tile index into the host depth source.
// After the addition, it may be negative - in which case, the
// transfer is done across EDRAM addressing wrapping, and
// xenos::kEdramTileCount must be added to it, but
// `& (xenos::kEdramTileCount - 1)` handles that regardless of the
// sign.
spv::Id host_depth_source_tile_index = builder.createBinOp(
spv::OpBitwiseAnd, type_uint,
builder.createUnaryOp(
spv::OpBitcast, type_uint,
builder.createBinOp(
spv::OpIAdd, type_int,
builder.createUnaryOp(spv::OpBitcast, type_int,
dest_tile_index),
builder.createTriOp(
spv::OpBitFieldSExtract, type_int,
builder.createUnaryOp(spv::OpBitcast, type_int,
host_depth_address_constant),
builder.makeUintConstant(
xenos::kEdramPitchTilesBits * 2),
builder.makeUintConstant(
xenos::kEdramBaseTilesBits + 1)))),
builder.makeUintConstant(xenos::kEdramTileCount - 1));
// Split the host depth source tile index into X and Y tile index
// within the source image.
spv::Id host_depth_source_pitch_tiles = builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, host_depth_address_constant,
builder.makeUintConstant(xenos::kEdramPitchTilesBits),
builder.makeUintConstant(xenos::kEdramPitchTilesBits));
spv::Id host_depth_source_tile_index_y = builder.createBinOp(
spv::OpUDiv, type_uint, host_depth_source_tile_index,
host_depth_source_pitch_tiles);
spv::Id host_depth_source_tile_index_x = builder.createBinOp(
spv::OpUMod, type_uint, host_depth_source_tile_index,
host_depth_source_pitch_tiles);
// Finally calculate the host depth source texture coordinates.
spv::Id host_depth_source_pixel_x_int = builder.createUnaryOp(
spv::OpBitcast, type_int,
builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(spv::OpIMul, type_uint,
builder.makeUintConstant(
tile_width_samples >>
uint32_t(key.source_msaa_samples >=
xenos::MsaaSamples::k4X)),
host_depth_source_tile_index_x),
host_depth_source_tile_pixel_x));
spv::Id host_depth_source_pixel_y_int = builder.createUnaryOp(
spv::OpBitcast, type_int,
builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(spv::OpIMul, type_uint,
builder.makeUintConstant(
tile_height_samples >>
uint32_t(key.source_msaa_samples >=
xenos::MsaaSamples::k2X)),
host_depth_source_tile_index_y),
host_depth_source_tile_pixel_y));
// Load the host depth source.
spv::Builder::TextureParameters
host_depth_source_texture_parameters = {};
host_depth_source_texture_parameters.sampler =
builder.createLoad(host_depth_source_texture, spv::NoPrecision);
id_vector_temp.clear();
id_vector_temp.push_back(host_depth_source_pixel_x_int);
id_vector_temp.push_back(host_depth_source_pixel_y_int);
host_depth_source_texture_parameters.coords =
builder.createCompositeConstruct(type_int2, id_vector_temp);
if (key.host_depth_source_msaa_samples != xenos::MsaaSamples::k1X) {
host_depth_source_texture_parameters.sample =
builder.createUnaryOp(spv::OpBitcast, type_int,
host_depth_source_sample_id);
} else {
host_depth_source_texture_parameters.lod =
builder.makeIntConstant(0);
}
host_depth32 = builder.createCompositeExtract(
builder.createTextureCall(spv::NoPrecision, type_float4, false,
true, false, false, false,
host_depth_source_texture_parameters,
spv::ImageOperandsMaskNone),
type_float, 0);
} else if (host_depth_source_buffer != spv::NoResult) {
// Get the address in the EDRAM scratch buffer and load from there.
// The beginning of the buffer is (0, 0) of the destination.
// 40-sample columns are not swapped for addressing simplicity
// (because this is used for depth -> depth transfers, where
// swapping isn't needed).
// Convert samples to pixels.
assert_true(key.host_depth_source_msaa_samples ==
xenos::MsaaSamples::k1X);
spv::Id dest_tile_sample_x = dest_tile_pixel_x;
spv::Id dest_tile_sample_y = dest_tile_pixel_y;
if (key.dest_msaa_samples >= xenos::MsaaSamples::k2X) {
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// Horizontal sample index in bit 0.
dest_tile_sample_x = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, dest_sample_id,
dest_tile_pixel_x, builder.makeUintConstant(1),
builder.makeUintConstant(31));
}
// Vertical sample index as 1 or 0 in bit 0 for true 2x or as 0
// or 1 in bit 1 for 4x or for 2x emulated as 4x.
dest_tile_sample_y = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint,
builder.createBinOp(
(key.dest_msaa_samples == xenos::MsaaSamples::k2X &&
msaa_2x_attachments_supported_)
? spv::OpBitwiseXor
: spv::OpShiftRightLogical,
type_uint, dest_sample_id, builder.makeUintConstant(1)),
dest_tile_pixel_y, builder.makeUintConstant(1),
builder.makeUintConstant(31));
}
// Combine the tile sample index and the tile index.
// The tile index doesn't need to be wrapped, as the host depth is
// written to the beginning of the buffer, without the base offset.
spv::Id host_depth_offset = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.makeUintConstant(tile_width_samples *
tile_height_samples),
dest_tile_index),
builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.makeUintConstant(tile_width_samples),
dest_tile_sample_y),
dest_tile_sample_x));
id_vector_temp.clear();
// The only SSBO structure member.
id_vector_temp.push_back(builder.makeIntConstant(0));
id_vector_temp.push_back(builder.createUnaryOp(
spv::OpBitcast, type_int, host_depth_offset));
// StorageBuffer since SPIR-V 1.3, but since SPIR-V 1.0 is
// generated, it's Uniform.
host_depth32 = builder.createUnaryOp(
spv::OpBitcast, type_float,
builder.createLoad(
builder.createAccessChain(spv::StorageClassUniform,
host_depth_source_buffer,
id_vector_temp),
spv::NoPrecision));
}
spv::Block* depth24_to_depth32_header = builder.getBuildPoint();
spv::Id depth24_to_depth32_convert_id = spv::NoResult;
spv::Block* depth24_to_depth32_merge = nullptr;
spv::Id host_depth24 = spv::NoResult;
if (host_depth32 != spv::NoResult) {
// Convert the host depth value to the guest format and check if it
// matches the value in the currently owning guest render target.
switch (dest_depth_format) {
case xenos::DepthRenderTargetFormat::kD24S8: {
// Round to the nearest even integer. This seems to be the
// correct conversion, adding +0.5 and rounding towards zero
// results in red instead of black in the 4D5307E6 clear shader.
host_depth24 = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createUnaryBuiltinCall(
type_float, ext_inst_glsl_std_450, GLSLstd450RoundEven,
builder.createBinOp(
spv::OpFMul, type_float, host_depth32,
builder.makeFloatConstant(float(0xFFFFFF)))));
} break;
case xenos::DepthRenderTargetFormat::kD24FS8: {
host_depth24 = SpirvShaderTranslator::PreClampedDepthTo20e4(
builder, host_depth32, depth_float24_round(), true,
ext_inst_glsl_std_450);
} break;
}
assert_true(host_depth24 != spv::NoResult);
// Update the header block pointer after the conversion (to avoid
// assuming that the conversion doesn't branch).
depth24_to_depth32_header = builder.getBuildPoint();
spv::Id host_depth_outdated = builder.createBinOp(
spv::OpINotEqual, type_bool, guest_depth24, host_depth24);
spv::Block& depth24_to_depth32_convert_entry =
builder.makeNewBlock();
{
spv::Block& depth24_to_depth32_merge_block =
builder.makeNewBlock();
depth24_to_depth32_merge = &depth24_to_depth32_merge_block;
}
builder.createSelectionMerge(depth24_to_depth32_merge,
spv::SelectionControlMaskNone);
builder.createConditionalBranch(host_depth_outdated,
&depth24_to_depth32_convert_entry,
depth24_to_depth32_merge);
builder.setBuildPoint(&depth24_to_depth32_convert_entry);
}
// Convert the guest 24-bit depth to float32 (in an open conditional
// if the host depth is also loaded).
spv::Id guest_depth32 = spv::NoResult;
switch (dest_depth_format) {
case xenos::DepthRenderTargetFormat::kD24S8: {
// Multiplying by 1.0 / 0xFFFFFF produces an incorrect result (for
// 0xC00000, for instance - which is 2_10_10_10 clear to 0001) -
// rescale from 0...0xFFFFFF to 0...0x1000000 doing what true
// float division followed by multiplication does (on x86-64 MSVC
// with default SSE rounding) - values starting from 0x800000
// become bigger by 1; then accurately bias the result's exponent.
guest_depth32 = builder.createBinOp(
spv::OpFMul, type_float,
builder.createUnaryOp(
spv::OpConvertUToF, type_float,
builder.createBinOp(
spv::OpIAdd, type_uint, guest_depth24,
builder.createBinOp(spv::OpShiftRightLogical,
type_uint, guest_depth24,
builder.makeUintConstant(23)))),
builder.makeFloatConstant(1.0f / float(1 << 24)));
} break;
case xenos::DepthRenderTargetFormat::kD24FS8: {
guest_depth32 = SpirvShaderTranslator::Depth20e4To32(
builder, guest_depth24, 0, true, false,
ext_inst_glsl_std_450);
} break;
}
assert_true(guest_depth32 != spv::NoResult);
spv::Id fragment_depth32 = guest_depth32;
if (host_depth32 != spv::NoResult) {
assert_not_null(depth24_to_depth32_merge);
spv::Id depth24_to_depth32_result_block_id =
builder.getBuildPoint()->getId();
builder.createBranch(depth24_to_depth32_merge);
builder.setBuildPoint(depth24_to_depth32_merge);
id_vector_temp.clear();
id_vector_temp.push_back(guest_depth32);
id_vector_temp.push_back(depth24_to_depth32_result_block_id);
id_vector_temp.push_back(host_depth32);
id_vector_temp.push_back(depth24_to_depth32_header->getId());
fragment_depth32 =
builder.createOp(spv::OpPhi, type_float, id_vector_temp);
}
builder.createStore(fragment_depth32, output_fragment_depth);
// Unpack the stencil into the stencil reference output if needed and
// not already written.
if (!packed_only_depth &&
output_fragment_stencil_ref != spv::NoResult) {
builder.createStore(
builder.createUnaryOp(
spv::OpBitcast, type_int,
builder.createBinOp(spv::OpBitwiseAnd, type_uint, packed,
builder.makeUintConstant(UINT8_MAX))),
output_fragment_stencil_ref);
}
}
} break;
case TransferOutput::kStencilBit: {
if (packed) {
// Kill the sample if the needed stencil bit is not set.
assert_true(push_constants_member_stencil_mask != UINT32_MAX);
id_vector_temp.clear();
id_vector_temp.push_back(builder.makeIntConstant(
int32_t(push_constants_member_stencil_mask)));
spv::Id stencil_mask_constant = builder.createLoad(
builder.createAccessChain(spv::StorageClassPushConstant,
push_constants, id_vector_temp),
spv::NoPrecision);
SpirvBuilder::IfBuilder stencil_kill_if(
builder.createBinOp(
spv::OpIEqual, type_bool,
builder.createBinOp(spv::OpBitwiseAnd, type_uint, packed,
stencil_mask_constant),
builder.makeUintConstant(0)),
spv::SelectionControlMaskNone, builder);
builder.createNoResultOp(spv::OpKill);
// OpKill terminates the block.
stencil_kill_if.makeEndIf(false);
}
} break;
}
}
// End the main function and make it the entry point.
builder.leaveFunction();
builder.addExecutionMode(main_function, spv::ExecutionModeOriginUpperLeft);
if (output_fragment_depth != spv::NoResult) {
builder.addExecutionMode(main_function, spv::ExecutionModeDepthReplacing);
}
if (output_fragment_stencil_ref != spv::NoResult) {
builder.addExecutionMode(main_function,
spv::ExecutionModeStencilRefReplacingEXT);
}
spv::Instruction* entry_point =
builder.addEntryPoint(spv::ExecutionModelFragment, main_function, "main");
for (spv::Id interface_id : main_interface) {
entry_point->addIdOperand(interface_id);
}
// Serialize the shader code.
std::vector<unsigned int> shader_code;
builder.dump(shader_code);
// Create the shader module, and store the handle even if creation fails not
// to try to create it again later.
VkShaderModule shader_module = ui::vulkan::util::CreateShaderModule(
vulkan_device, reinterpret_cast<const uint32_t*>(shader_code.data()),
sizeof(uint32_t) * shader_code.size());
if (shader_module == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the render target ownership "
"transfer shader 0x{:08X}",
key.key);
}
transfer_shaders_.emplace(key, shader_module);
return shader_module;
}
VkPipeline const* VulkanRenderTargetCache::GetTransferPipelines(
TransferPipelineKey key) {
auto pipeline_it = transfer_pipelines_.find(key);
if (pipeline_it != transfer_pipelines_.end()) {
return pipeline_it->second[0] != VK_NULL_HANDLE ? pipeline_it->second.data()
: nullptr;
}
VkRenderPass render_pass =
GetHostRenderTargetsRenderPass(key.render_pass_key);
VkShaderModule fragment_shader_module = GetTransferShader(key.shader_key);
if (render_pass == VK_NULL_HANDLE ||
fragment_shader_module == VK_NULL_HANDLE) {
transfer_pipelines_.emplace(key, std::array<VkPipeline, 4>{});
return nullptr;
}
const TransferModeInfo& mode = kTransferModes[size_t(key.shader_key.mode)];
const ui::vulkan::VulkanDevice* const vulkan_device =
command_processor_.GetVulkanDevice();
const ui::vulkan::VulkanDevice::Functions& dfn = vulkan_device->functions();
const VkDevice device = vulkan_device->device();
const ui::vulkan::VulkanDevice::Properties& device_properties =
vulkan_device->properties();
uint32_t dest_sample_count = uint32_t(1)
<< uint32_t(key.shader_key.dest_msaa_samples);
bool dest_is_masked_sample =
dest_sample_count > 1 && !device_properties.sampleRateShading;
VkPipelineShaderStageCreateInfo shader_stages[2];
shader_stages[0].sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO;
shader_stages[0].pNext = nullptr;
shader_stages[0].flags = 0;
shader_stages[0].stage = VK_SHADER_STAGE_VERTEX_BIT;
shader_stages[0].module = transfer_passthrough_vertex_shader_;
shader_stages[0].pName = "main";
shader_stages[0].pSpecializationInfo = nullptr;
shader_stages[1].sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO;
shader_stages[1].pNext = nullptr;
shader_stages[1].flags = 0;
shader_stages[1].stage = VK_SHADER_STAGE_FRAGMENT_BIT;
shader_stages[1].module = fragment_shader_module;
shader_stages[1].pName = "main";
shader_stages[1].pSpecializationInfo = nullptr;
VkSpecializationMapEntry sample_id_specialization_map_entry;
uint32_t sample_id_specialization_constant;
VkSpecializationInfo sample_id_specialization_info;
if (dest_is_masked_sample) {
sample_id_specialization_map_entry.constantID = 0;
sample_id_specialization_map_entry.offset = 0;
sample_id_specialization_map_entry.size = sizeof(uint32_t);
sample_id_specialization_constant = 0;
sample_id_specialization_info.mapEntryCount = 1;
sample_id_specialization_info.pMapEntries =
&sample_id_specialization_map_entry;
sample_id_specialization_info.dataSize =
sizeof(sample_id_specialization_constant);
sample_id_specialization_info.pData = &sample_id_specialization_constant;
shader_stages[1].pSpecializationInfo = &sample_id_specialization_info;
}
VkVertexInputBindingDescription vertex_input_binding;
vertex_input_binding.binding = 0;
vertex_input_binding.stride = sizeof(float) * 2;
vertex_input_binding.inputRate = VK_VERTEX_INPUT_RATE_VERTEX;
VkVertexInputAttributeDescription vertex_input_attribute;
vertex_input_attribute.location = 0;
vertex_input_attribute.binding = 0;
vertex_input_attribute.format = VK_FORMAT_R32G32_SFLOAT;
vertex_input_attribute.offset = 0;
VkPipelineVertexInputStateCreateInfo vertex_input_state;
vertex_input_state.sType =
VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO;
vertex_input_state.pNext = nullptr;
vertex_input_state.flags = 0;
vertex_input_state.vertexBindingDescriptionCount = 1;
vertex_input_state.pVertexBindingDescriptions = &vertex_input_binding;
vertex_input_state.vertexAttributeDescriptionCount = 1;
vertex_input_state.pVertexAttributeDescriptions = &vertex_input_attribute;
VkPipelineInputAssemblyStateCreateInfo input_assembly_state;
input_assembly_state.sType =
VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO;
input_assembly_state.pNext = nullptr;
input_assembly_state.flags = 0;
input_assembly_state.topology = VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST;
input_assembly_state.primitiveRestartEnable = VK_FALSE;
// Dynamic, to stay within maxViewportDimensions while preferring a
// power-of-two factor for converting from pixel coordinates to NDC for exact
// precision.
VkPipelineViewportStateCreateInfo viewport_state;
viewport_state.sType = VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO;
viewport_state.pNext = nullptr;
viewport_state.flags = 0;
viewport_state.viewportCount = 1;
viewport_state.pViewports = nullptr;
viewport_state.scissorCount = 1;
viewport_state.pScissors = nullptr;
VkPipelineRasterizationStateCreateInfo rasterization_state = {};
rasterization_state.sType =
VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO;
rasterization_state.polygonMode = VK_POLYGON_MODE_FILL;
rasterization_state.cullMode = VK_CULL_MODE_NONE;
rasterization_state.frontFace = VK_FRONT_FACE_COUNTER_CLOCKWISE;
rasterization_state.lineWidth = 1.0f;
// For samples other than the first, will be changed for the pipelines for
// other samples.
VkSampleMask sample_mask = UINT32_MAX;
VkPipelineMultisampleStateCreateInfo multisample_state = {};
multisample_state.sType =
VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO;
multisample_state.rasterizationSamples =
(dest_sample_count == 2 && !msaa_2x_attachments_supported_)
? VK_SAMPLE_COUNT_4_BIT
: VkSampleCountFlagBits(dest_sample_count);
if (dest_sample_count > 1) {
if (device_properties.sampleRateShading) {
multisample_state.sampleShadingEnable = VK_TRUE;
multisample_state.minSampleShading = 1.0f;
if (dest_sample_count == 2 && !msaa_2x_attachments_supported_) {
// Emulating 2x MSAA as samples 0 and 3 of 4x MSAA when 2x is not
// supported.
sample_mask = 0b1001;
}
} else {
sample_mask = 0b1;
}
if (sample_mask != UINT32_MAX) {
multisample_state.pSampleMask = &sample_mask;
}
}
// Whether the depth / stencil state is used depends on the presence of a
// depth attachment in the render pass - but not making assumptions about
// whether the render pass contains any specific attachments, so setting up
// valid depth / stencil state unconditionally.
VkPipelineDepthStencilStateCreateInfo depth_stencil_state = {};
depth_stencil_state.sType =
VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO;
if (mode.output == TransferOutput::kDepth) {
depth_stencil_state.depthTestEnable = VK_TRUE;
depth_stencil_state.depthWriteEnable = VK_TRUE;
depth_stencil_state.depthCompareOp = cvars::depth_transfer_not_equal_test
? VK_COMPARE_OP_NOT_EQUAL
: VK_COMPARE_OP_ALWAYS;
}
if ((mode.output == TransferOutput::kDepth &&
vulkan_device->extensions().ext_EXT_shader_stencil_export) ||
mode.output == TransferOutput::kStencilBit) {
depth_stencil_state.stencilTestEnable = VK_TRUE;
depth_stencil_state.front.failOp = VK_STENCIL_OP_KEEP;
depth_stencil_state.front.passOp = VK_STENCIL_OP_REPLACE;
depth_stencil_state.front.depthFailOp = VK_STENCIL_OP_REPLACE;
// Using ALWAYS, not NOT_EQUAL, so depth writing is unaffected by stencil
// being different.
depth_stencil_state.front.compareOp = VK_COMPARE_OP_ALWAYS;
// Will be dynamic for stencil bit output.
depth_stencil_state.front.writeMask = UINT8_MAX;
depth_stencil_state.front.reference = UINT8_MAX;
depth_stencil_state.back = depth_stencil_state.front;
}
// Whether the color blend state is used depends on the presence of color
// attachments in the render pass - but not making assumptions about whether
// the render pass contains any specific attachments, so setting up valid
// color blend state unconditionally.
VkPipelineColorBlendAttachmentState
color_blend_attachments[xenos::kMaxColorRenderTargets] = {};
VkPipelineColorBlendStateCreateInfo color_blend_state = {};
color_blend_state.sType =
VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO;
color_blend_state.attachmentCount =
32 - xe::lzcnt(key.render_pass_key.depth_and_color_used >> 1);
color_blend_state.pAttachments = color_blend_attachments;
if (mode.output == TransferOutput::kColor) {
assert_true(device_properties.independentBlend);
color_blend_attachments[key.shader_key.dest_color_rt_index].colorWriteMask =
VK_COLOR_COMPONENT_R_BIT | VK_COLOR_COMPONENT_G_BIT |
VK_COLOR_COMPONENT_B_BIT | VK_COLOR_COMPONENT_A_BIT;
}
std::array<VkDynamicState, 3> dynamic_states;
VkPipelineDynamicStateCreateInfo dynamic_state;
dynamic_state.sType = VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO;
dynamic_state.pNext = nullptr;
dynamic_state.flags = 0;
dynamic_state.dynamicStateCount = 0;
dynamic_state.pDynamicStates = dynamic_states.data();
dynamic_states[dynamic_state.dynamicStateCount++] = VK_DYNAMIC_STATE_VIEWPORT;
dynamic_states[dynamic_state.dynamicStateCount++] = VK_DYNAMIC_STATE_SCISSOR;
if (mode.output == TransferOutput::kStencilBit) {
dynamic_states[dynamic_state.dynamicStateCount++] =
VK_DYNAMIC_STATE_STENCIL_WRITE_MASK;
}
std::array<VkPipeline, 4> pipelines{};
VkGraphicsPipelineCreateInfo pipeline_create_info;
pipeline_create_info.sType = VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO;
pipeline_create_info.pNext = nullptr;
pipeline_create_info.flags = 0;
if (dest_is_masked_sample) {
pipeline_create_info.flags |= VK_PIPELINE_CREATE_ALLOW_DERIVATIVES_BIT;
}
pipeline_create_info.stageCount = uint32_t(xe::countof(shader_stages));
pipeline_create_info.pStages = shader_stages;
pipeline_create_info.pVertexInputState = &vertex_input_state;
pipeline_create_info.pInputAssemblyState = &input_assembly_state;
pipeline_create_info.pTessellationState = nullptr;
pipeline_create_info.pViewportState = &viewport_state;
pipeline_create_info.pRasterizationState = &rasterization_state;
pipeline_create_info.pMultisampleState = &multisample_state;
pipeline_create_info.pDepthStencilState = &depth_stencil_state;
pipeline_create_info.pColorBlendState = &color_blend_state;
pipeline_create_info.pDynamicState = &dynamic_state;
pipeline_create_info.layout =
transfer_pipeline_layouts_[size_t(mode.pipeline_layout)];
pipeline_create_info.renderPass = render_pass;
pipeline_create_info.subpass = 0;
pipeline_create_info.basePipelineHandle = VK_NULL_HANDLE;
pipeline_create_info.basePipelineIndex = -1;
if (dfn.vkCreateGraphicsPipelines(device, VK_NULL_HANDLE, 1,
&pipeline_create_info, nullptr,
&pipelines[0]) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the render target ownership "
"transfer pipeline for render pass 0x{:08X}, shader 0x{:08X}",
key.render_pass_key.key, key.shader_key.key);
transfer_pipelines_.emplace(key, std::array<VkPipeline, 4>{});
return nullptr;
}
if (dest_is_masked_sample) {
assert_true(multisample_state.pSampleMask == &sample_mask);
pipeline_create_info.flags = (pipeline_create_info.flags &
~VK_PIPELINE_CREATE_ALLOW_DERIVATIVES_BIT) |
VK_PIPELINE_CREATE_DERIVATIVE_BIT;
pipeline_create_info.basePipelineHandle = pipelines[0];
for (uint32_t i = 1; i < dest_sample_count; ++i) {
// Emulating 2x MSAA as samples 0 and 3 of 4x MSAA when 2x is not
// supported.
uint32_t host_sample_index =
(dest_sample_count == 2 && !msaa_2x_attachments_supported_ && i == 1)
? 3
: i;
sample_id_specialization_constant = host_sample_index;
sample_mask = uint32_t(1) << host_sample_index;
if (dfn.vkCreateGraphicsPipelines(device, VK_NULL_HANDLE, 1,
&pipeline_create_info, nullptr,
&pipelines[i]) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the render target "
"ownership transfer pipeline for render pass 0x{:08X}, shader "
"0x{:08X}, sample {}",
key.render_pass_key.key, key.shader_key.key, i);
for (uint32_t j = 0; j < i; ++j) {
dfn.vkDestroyPipeline(device, pipelines[j], nullptr);
}
transfer_pipelines_.emplace(key, std::array<VkPipeline, 4>{});
return nullptr;
}
}
}
return transfer_pipelines_.emplace(key, pipelines).first->second.data();
}
void VulkanRenderTargetCache::PerformTransfersAndResolveClears(
uint32_t render_target_count, RenderTarget* const* render_targets,
const std::vector<Transfer>* render_target_transfers,
const uint64_t* render_target_resolve_clear_values,
const Transfer::Rectangle* resolve_clear_rectangle) {
assert_true(GetPath() == Path::kHostRenderTargets);
const ui::vulkan::VulkanDevice* const vulkan_device =
command_processor_.GetVulkanDevice();
uint64_t current_submission = command_processor_.GetCurrentSubmission();
DeferredCommandBuffer& command_buffer =
command_processor_.deferred_command_buffer();
bool resolve_clear_needed =
render_target_resolve_clear_values && resolve_clear_rectangle;
VkClearRect resolve_clear_rect;
if (resolve_clear_needed) {
// Assuming the rectangle is already clamped by the setup function from the
// common render target cache.
resolve_clear_rect.rect.offset.x =
int32_t(resolve_clear_rectangle->x_pixels * draw_resolution_scale_x());
resolve_clear_rect.rect.offset.y =
int32_t(resolve_clear_rectangle->y_pixels * draw_resolution_scale_y());
resolve_clear_rect.rect.extent.width =
resolve_clear_rectangle->width_pixels * draw_resolution_scale_x();
resolve_clear_rect.rect.extent.height =
resolve_clear_rectangle->height_pixels * draw_resolution_scale_y();
resolve_clear_rect.baseArrayLayer = 0;
resolve_clear_rect.layerCount = 1;
}
// Do host depth storing for the depth destination (assuming there can be only
// one depth destination) where depth destination == host depth source.
bool host_depth_store_set_up = false;
for (uint32_t i = 0; i < render_target_count; ++i) {
RenderTarget* dest_rt = render_targets[i];
if (!dest_rt) {
continue;
}
auto& dest_vulkan_rt = *static_cast<VulkanRenderTarget*>(dest_rt);
RenderTargetKey dest_rt_key = dest_vulkan_rt.key();
if (!dest_rt_key.is_depth) {
continue;
}
const std::vector<Transfer>& depth_transfers = render_target_transfers[i];
for (const Transfer& transfer : depth_transfers) {
if (transfer.host_depth_source != dest_rt) {
continue;
}
if (!host_depth_store_set_up) {
// Pipeline.
command_processor_.BindExternalComputePipeline(
host_depth_store_pipelines_[size_t(dest_rt_key.msaa_samples)]);
// Descriptor set bindings.
VkDescriptorSet host_depth_store_descriptor_sets[] = {
edram_storage_buffer_descriptor_set_,
dest_vulkan_rt.GetDescriptorSetTransferSource(),
};
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_COMPUTE, host_depth_store_pipeline_layout_,
0, uint32_t(xe::countof(host_depth_store_descriptor_sets)),
host_depth_store_descriptor_sets, 0, nullptr);
// Render target constant.
HostDepthStoreRenderTargetConstant
host_depth_store_render_target_constant =
GetHostDepthStoreRenderTargetConstant(
dest_rt_key.pitch_tiles_at_32bpp,
msaa_2x_attachments_supported_);
command_buffer.CmdVkPushConstants(
host_depth_store_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT,
uint32_t(offsetof(HostDepthStoreConstants, render_target)),
sizeof(host_depth_store_render_target_constant),
&host_depth_store_render_target_constant);
// Barriers - don't need to try to combine them with the rest of
// render target transfer barriers now - if this happens, after host
// depth storing, SHADER_READ -> DEPTH_STENCIL_ATTACHMENT_WRITE will be
// done anyway even in the best case, so it's not possible to have all
// the barriers in one place here.
UseEdramBuffer(EdramBufferUsage::kComputeWrite);
// Always transitioning both depth and stencil, not storing separate
// usage flags for depth and stencil.
command_processor_.PushImageMemoryBarrier(
dest_vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT),
dest_vulkan_rt.current_stage_mask(),
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
dest_vulkan_rt.current_access_mask(), VK_ACCESS_SHADER_READ_BIT,
dest_vulkan_rt.current_layout(),
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL);
dest_vulkan_rt.SetUsage(VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
VK_ACCESS_SHADER_READ_BIT,
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL);
host_depth_store_set_up = true;
}
Transfer::Rectangle
transfer_rectangles[Transfer::kMaxRectanglesWithCutout];
uint32_t transfer_rectangle_count = transfer.GetRectangles(
dest_rt_key.base_tiles, dest_rt_key.pitch_tiles_at_32bpp,
dest_rt_key.msaa_samples, false, transfer_rectangles,
resolve_clear_rectangle);
assert_not_zero(transfer_rectangle_count);
HostDepthStoreRectangleConstant host_depth_store_rectangle_constant;
for (uint32_t j = 0; j < transfer_rectangle_count; ++j) {
uint32_t group_count_x, group_count_y;
GetHostDepthStoreRectangleInfo(
transfer_rectangles[j], dest_rt_key.msaa_samples,
host_depth_store_rectangle_constant, group_count_x, group_count_y);
command_buffer.CmdVkPushConstants(
host_depth_store_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT,
uint32_t(offsetof(HostDepthStoreConstants, rectangle)),
sizeof(host_depth_store_rectangle_constant),
&host_depth_store_rectangle_constant);
command_processor_.SubmitBarriers(true);
command_buffer.CmdVkDispatch(group_count_x, group_count_y, 1);
MarkEdramBufferModified();
}
}
break;
}
constexpr VkPipelineStageFlags kSourceStageMask =
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT;
constexpr VkAccessFlags kSourceAccessMask = VK_ACCESS_SHADER_READ_BIT;
constexpr VkImageLayout kSourceLayout =
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
// Try to insert as many barriers as possible in one place, hoping that in the
// best case (no cross-copying between current render targets), barriers will
// need to be only inserted here, not between transfers. In case of
// cross-copying, if the destination use is going to happen before the source
// use, choose the destination state, otherwise the source state - to match
// the order in which transfers will actually happen (otherwise there will be
// just a useless switch back and forth).
for (uint32_t i = 0; i < render_target_count; ++i) {
RenderTarget* dest_rt = render_targets[i];
if (!dest_rt) {
continue;
}
const std::vector<Transfer>& dest_transfers = render_target_transfers[i];
if (!resolve_clear_needed && dest_transfers.empty()) {
continue;
}
// Transition the destination, only if not going to be used as a source
// earlier.
bool dest_used_previously_as_source = false;
for (uint32_t j = 0; j < i; ++j) {
for (const Transfer& previous_transfer : render_target_transfers[j]) {
if (previous_transfer.source == dest_rt ||
previous_transfer.host_depth_source == dest_rt) {
dest_used_previously_as_source = true;
break;
}
}
}
if (!dest_used_previously_as_source) {
auto& dest_vulkan_rt = *static_cast<VulkanRenderTarget*>(dest_rt);
VkPipelineStageFlags dest_dst_stage_mask;
VkAccessFlags dest_dst_access_mask;
VkImageLayout dest_new_layout;
dest_vulkan_rt.GetDrawUsage(&dest_dst_stage_mask, &dest_dst_access_mask,
&dest_new_layout);
command_processor_.PushImageMemoryBarrier(
dest_vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
dest_vulkan_rt.key().is_depth
? (VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)
: VK_IMAGE_ASPECT_COLOR_BIT),
dest_vulkan_rt.current_stage_mask(), dest_dst_stage_mask,
dest_vulkan_rt.current_access_mask(), dest_dst_access_mask,
dest_vulkan_rt.current_layout(), dest_new_layout);
dest_vulkan_rt.SetUsage(dest_dst_stage_mask, dest_dst_access_mask,
dest_new_layout);
}
// Transition the sources, only if not going to be used as destinations
// earlier.
for (const Transfer& transfer : dest_transfers) {
bool source_previously_used_as_dest = false;
bool host_depth_source_previously_used_as_dest = false;
for (uint32_t j = 0; j < i; ++j) {
if (render_target_transfers[j].empty()) {
continue;
}
const RenderTarget* previous_rt = render_targets[j];
if (transfer.source == previous_rt) {
source_previously_used_as_dest = true;
}
if (transfer.host_depth_source == previous_rt) {
host_depth_source_previously_used_as_dest = true;
}
}
if (!source_previously_used_as_dest) {
auto& source_vulkan_rt =
*static_cast<VulkanRenderTarget*>(transfer.source);
command_processor_.PushImageMemoryBarrier(
source_vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
source_vulkan_rt.key().is_depth
? (VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)
: VK_IMAGE_ASPECT_COLOR_BIT),
source_vulkan_rt.current_stage_mask(), kSourceStageMask,
source_vulkan_rt.current_access_mask(), kSourceAccessMask,
source_vulkan_rt.current_layout(), kSourceLayout);
source_vulkan_rt.SetUsage(kSourceStageMask, kSourceAccessMask,
kSourceLayout);
}
// transfer.host_depth_source == dest_rt means the EDRAM buffer will be
// used instead, no need to transition.
if (transfer.host_depth_source && transfer.host_depth_source != dest_rt &&
!host_depth_source_previously_used_as_dest) {
auto& host_depth_source_vulkan_rt =
*static_cast<VulkanRenderTarget*>(transfer.host_depth_source);
command_processor_.PushImageMemoryBarrier(
host_depth_source_vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT),
host_depth_source_vulkan_rt.current_stage_mask(), kSourceStageMask,
host_depth_source_vulkan_rt.current_access_mask(),
kSourceAccessMask, host_depth_source_vulkan_rt.current_layout(),
kSourceLayout);
host_depth_source_vulkan_rt.SetUsage(kSourceStageMask,
kSourceAccessMask, kSourceLayout);
}
}
}
if (host_depth_store_set_up) {
// Will be reading copied host depth from the EDRAM buffer.
UseEdramBuffer(EdramBufferUsage::kFragmentRead);
}
// Perform the transfers and clears.
TransferPipelineLayoutIndex last_transfer_pipeline_layout_index =
TransferPipelineLayoutIndex::kCount;
uint32_t transfer_descriptor_sets_bound = 0;
uint32_t transfer_push_constants_set = 0;
VkDescriptorSet last_descriptor_set_host_depth_stencil_textures =
VK_NULL_HANDLE;
VkDescriptorSet last_descriptor_set_depth_stencil_textures = VK_NULL_HANDLE;
VkDescriptorSet last_descriptor_set_color_texture = VK_NULL_HANDLE;
TransferAddressConstant last_host_depth_address_constant;
TransferAddressConstant last_address_constant;
for (uint32_t i = 0; i < render_target_count; ++i) {
RenderTarget* dest_rt = render_targets[i];
if (!dest_rt) {
continue;
}
const std::vector<Transfer>& current_transfers = render_target_transfers[i];
if (current_transfers.empty() && !resolve_clear_needed) {
continue;
}
auto& dest_vulkan_rt = *static_cast<VulkanRenderTarget*>(dest_rt);
RenderTargetKey dest_rt_key = dest_vulkan_rt.key();
// Late barriers in case there was cross-copying that prevented merging of
// barriers.
{
VkPipelineStageFlags dest_dst_stage_mask;
VkAccessFlags dest_dst_access_mask;
VkImageLayout dest_new_layout;
dest_vulkan_rt.GetDrawUsage(&dest_dst_stage_mask, &dest_dst_access_mask,
&dest_new_layout);
command_processor_.PushImageMemoryBarrier(
dest_vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
dest_rt_key.is_depth
? (VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)
: VK_IMAGE_ASPECT_COLOR_BIT),
dest_vulkan_rt.current_stage_mask(), dest_dst_stage_mask,
dest_vulkan_rt.current_access_mask(), dest_dst_access_mask,
dest_vulkan_rt.current_layout(), dest_new_layout);
dest_vulkan_rt.SetUsage(dest_dst_stage_mask, dest_dst_access_mask,
dest_new_layout);
}
// Get the objects needed for transfers to the destination.
// TODO(Triang3l): Reuse the guest render pass for transfers where possible
// (if the Vulkan format used for drawing is also usable for transfers - for
// instance, R8G8B8A8_UNORM can be used for both, so the guest pass can be
// reused, but R16G16B16A16_SFLOAT render targets use R16G16B16A16_UINT for
// transfers, so the transfer pass has to be separate) to avoid stores and
// loads on tile-based devices to make this actually applicable. Also
// overall perform all non-cross-copying transfers for the current
// framebuffer configuration in a single pass, to load / store only once.
RenderPassKey transfer_render_pass_key;
transfer_render_pass_key.msaa_samples = dest_rt_key.msaa_samples;
if (dest_rt_key.is_depth) {
transfer_render_pass_key.depth_and_color_used = 0b1;
transfer_render_pass_key.depth_format = dest_rt_key.GetDepthFormat();
} else {
transfer_render_pass_key.depth_and_color_used = 0b1 << 1;
transfer_render_pass_key.color_0_view_format =
dest_rt_key.GetColorFormat();
transfer_render_pass_key.color_rts_use_transfer_formats = 1;
}
VkRenderPass transfer_render_pass =
GetHostRenderTargetsRenderPass(transfer_render_pass_key);
if (transfer_render_pass == VK_NULL_HANDLE) {
continue;
}
const RenderTarget*
transfer_framebuffer_render_targets[1 + xenos::kMaxColorRenderTargets] =
{};
transfer_framebuffer_render_targets[dest_rt_key.is_depth ? 0 : 1] = dest_rt;
const Framebuffer* transfer_framebuffer = GetHostRenderTargetsFramebuffer(
transfer_render_pass_key, dest_rt_key.pitch_tiles_at_32bpp,
transfer_framebuffer_render_targets);
if (!transfer_framebuffer) {
continue;
}
// Don't enter the render pass immediately - may still insert source
// barriers later.
if (!current_transfers.empty()) {
uint32_t dest_pitch_tiles = dest_rt_key.GetPitchTiles();
bool dest_is_64bpp = dest_rt_key.Is64bpp();
// Gather shader keys and sort to reduce pipeline state and binding
// switches. Also gather stencil rectangles to clear if needed.
bool need_stencil_bit_draws =
dest_rt_key.is_depth &&
!vulkan_device->extensions().ext_EXT_shader_stencil_export;
current_transfer_invocations_.clear();
current_transfer_invocations_.reserve(
current_transfers.size() << uint32_t(need_stencil_bit_draws));
uint32_t rt_sort_index = 0;
TransferShaderKey new_transfer_shader_key;
new_transfer_shader_key.dest_msaa_samples = dest_rt_key.msaa_samples;
new_transfer_shader_key.dest_resource_format =
dest_rt_key.resource_format;
uint32_t stencil_clear_rectangle_count = 0;
for (uint32_t j = 0; j <= uint32_t(need_stencil_bit_draws); ++j) {
// j == 0 - color or depth.
// j == 1 - stencil bits.
// Stencil bit writing always requires a different root signature,
// handle these separately. Stencil never has a host depth source.
// Clear previously set sort indices.
for (const Transfer& transfer : current_transfers) {
auto host_depth_source_vulkan_rt =
static_cast<VulkanRenderTarget*>(transfer.host_depth_source);
if (host_depth_source_vulkan_rt) {
host_depth_source_vulkan_rt->SetTemporarySortIndex(UINT32_MAX);
}
assert_not_null(transfer.source);
auto& source_vulkan_rt =
*static_cast<VulkanRenderTarget*>(transfer.source);
source_vulkan_rt.SetTemporarySortIndex(UINT32_MAX);
}
for (const Transfer& transfer : current_transfers) {
assert_not_null(transfer.source);
auto& source_vulkan_rt =
*static_cast<VulkanRenderTarget*>(transfer.source);
VulkanRenderTarget* host_depth_source_vulkan_rt =
j ? nullptr
: static_cast<VulkanRenderTarget*>(transfer.host_depth_source);
if (host_depth_source_vulkan_rt &&
host_depth_source_vulkan_rt->temporary_sort_index() ==
UINT32_MAX) {
host_depth_source_vulkan_rt->SetTemporarySortIndex(rt_sort_index++);
}
if (source_vulkan_rt.temporary_sort_index() == UINT32_MAX) {
source_vulkan_rt.SetTemporarySortIndex(rt_sort_index++);
}
RenderTargetKey source_rt_key = source_vulkan_rt.key();
new_transfer_shader_key.source_msaa_samples =
source_rt_key.msaa_samples;
new_transfer_shader_key.source_resource_format =
source_rt_key.resource_format;
bool host_depth_source_is_copy =
host_depth_source_vulkan_rt == &dest_vulkan_rt;
// The host depth copy buffer has only raw samples.
new_transfer_shader_key.host_depth_source_msaa_samples =
(host_depth_source_vulkan_rt && !host_depth_source_is_copy)
? host_depth_source_vulkan_rt->key().msaa_samples
: xenos::MsaaSamples::k1X;
if (j) {
new_transfer_shader_key.mode =
source_rt_key.is_depth ? TransferMode::kDepthToStencilBit
: TransferMode::kColorToStencilBit;
stencil_clear_rectangle_count +=
transfer.GetRectangles(dest_rt_key.base_tiles, dest_pitch_tiles,
dest_rt_key.msaa_samples, dest_is_64bpp,
nullptr, resolve_clear_rectangle);
} else {
if (dest_rt_key.is_depth) {
if (host_depth_source_vulkan_rt) {
if (host_depth_source_is_copy) {
new_transfer_shader_key.mode =
source_rt_key.is_depth
? TransferMode::kDepthAndHostDepthCopyToDepth
: TransferMode::kColorAndHostDepthCopyToDepth;
} else {
new_transfer_shader_key.mode =
source_rt_key.is_depth
? TransferMode::kDepthAndHostDepthToDepth
: TransferMode::kColorAndHostDepthToDepth;
}
} else {
new_transfer_shader_key.mode =
source_rt_key.is_depth ? TransferMode::kDepthToDepth
: TransferMode::kColorToDepth;
}
} else {
new_transfer_shader_key.mode = source_rt_key.is_depth
? TransferMode::kDepthToColor
: TransferMode::kColorToColor;
}
}
current_transfer_invocations_.emplace_back(transfer,
new_transfer_shader_key);
if (j) {
current_transfer_invocations_.back().transfer.host_depth_source =
nullptr;
}
}
}
std::sort(current_transfer_invocations_.begin(),
current_transfer_invocations_.end());
for (auto it = current_transfer_invocations_.cbegin();
it != current_transfer_invocations_.cend(); ++it) {
assert_not_null(it->transfer.source);
auto& source_vulkan_rt =
*static_cast<VulkanRenderTarget*>(it->transfer.source);
command_processor_.PushImageMemoryBarrier(
source_vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
source_vulkan_rt.key().is_depth
? (VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)
: VK_IMAGE_ASPECT_COLOR_BIT),
source_vulkan_rt.current_stage_mask(), kSourceStageMask,
source_vulkan_rt.current_access_mask(), kSourceAccessMask,
source_vulkan_rt.current_layout(), kSourceLayout);
source_vulkan_rt.SetUsage(kSourceStageMask, kSourceAccessMask,
kSourceLayout);
auto host_depth_source_vulkan_rt =
static_cast<VulkanRenderTarget*>(it->transfer.host_depth_source);
if (host_depth_source_vulkan_rt) {
TransferShaderKey transfer_shader_key = it->shader_key;
if (transfer_shader_key.mode ==
TransferMode::kDepthAndHostDepthCopyToDepth ||
transfer_shader_key.mode ==
TransferMode::kColorAndHostDepthCopyToDepth) {
// Reading copied host depth from the EDRAM buffer.
UseEdramBuffer(EdramBufferUsage::kFragmentRead);
} else {
// Reading host depth from the texture.
command_processor_.PushImageMemoryBarrier(
host_depth_source_vulkan_rt->image(),
ui::vulkan::util::InitializeSubresourceRange(
VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT),
host_depth_source_vulkan_rt->current_stage_mask(),
kSourceStageMask,
host_depth_source_vulkan_rt->current_access_mask(),
kSourceAccessMask,
host_depth_source_vulkan_rt->current_layout(), kSourceLayout);
host_depth_source_vulkan_rt->SetUsage(
kSourceStageMask, kSourceAccessMask, kSourceLayout);
}
}
}
// Perform the transfers for the render target.
command_processor_.SubmitBarriersAndEnterRenderTargetCacheRenderPass(
transfer_render_pass, transfer_framebuffer);
if (stencil_clear_rectangle_count) {
VkClearAttachment* stencil_clear_attachment;
VkClearRect* stencil_clear_rect_write_ptr;
command_buffer.CmdClearAttachmentsEmplace(1, stencil_clear_attachment,
stencil_clear_rectangle_count,
stencil_clear_rect_write_ptr);
stencil_clear_attachment->aspectMask = VK_IMAGE_ASPECT_STENCIL_BIT;
stencil_clear_attachment->colorAttachment = 0;
stencil_clear_attachment->clearValue.depthStencil.depth = 0.0f;
stencil_clear_attachment->clearValue.depthStencil.stencil = 0;
for (const Transfer& transfer : current_transfers) {
Transfer::Rectangle transfer_stencil_clear_rectangles
[Transfer::kMaxRectanglesWithCutout];
uint32_t transfer_stencil_clear_rectangle_count =
transfer.GetRectangles(dest_rt_key.base_tiles, dest_pitch_tiles,
dest_rt_key.msaa_samples, dest_is_64bpp,
transfer_stencil_clear_rectangles,
resolve_clear_rectangle);
for (uint32_t j = 0; j < transfer_stencil_clear_rectangle_count;
++j) {
const Transfer::Rectangle& stencil_clear_rectangle =
transfer_stencil_clear_rectangles[j];
stencil_clear_rect_write_ptr->rect.offset.x = int32_t(
stencil_clear_rectangle.x_pixels * draw_resolution_scale_x());
stencil_clear_rect_write_ptr->rect.offset.y = int32_t(
stencil_clear_rectangle.y_pixels * draw_resolution_scale_y());
stencil_clear_rect_write_ptr->rect.extent.width =
stencil_clear_rectangle.width_pixels *
draw_resolution_scale_x();
stencil_clear_rect_write_ptr->rect.extent.height =
stencil_clear_rectangle.height_pixels *
draw_resolution_scale_y();
stencil_clear_rect_write_ptr->baseArrayLayer = 0;
stencil_clear_rect_write_ptr->layerCount = 1;
++stencil_clear_rect_write_ptr;
}
}
}
// Prefer power of two viewports for exact division by simply biasing the
// exponent.
VkViewport transfer_viewport;
transfer_viewport.x = 0.0f;
transfer_viewport.y = 0.0f;
transfer_viewport.width =
float(std::min(xe::next_pow2(transfer_framebuffer->host_extent.width),
vulkan_device->properties().maxViewportDimensions[0]));
transfer_viewport.height = float(
std::min(xe::next_pow2(transfer_framebuffer->host_extent.height),
vulkan_device->properties().maxViewportDimensions[1]));
transfer_viewport.minDepth = 0.0f;
transfer_viewport.maxDepth = 1.0f;
command_processor_.SetViewport(transfer_viewport);
// GetRectangles returns coordinates in guest pixels, so scale
// pixels_to_ndc to convert guest pixels to NDC correctly.
float pixels_to_ndc_x =
2.0f / transfer_viewport.width * draw_resolution_scale_x();
float pixels_to_ndc_y =
2.0f / transfer_viewport.height * draw_resolution_scale_y();
VkRect2D transfer_scissor;
transfer_scissor.offset.x = 0;
transfer_scissor.offset.y = 0;
transfer_scissor.extent = transfer_framebuffer->host_extent;
command_processor_.SetScissor(transfer_scissor);
for (auto it = current_transfer_invocations_.cbegin();
it != current_transfer_invocations_.cend(); ++it) {
const TransferInvocation& transfer_invocation_first = *it;
// Will be merging transfers from the same source into one mesh.
auto it_merged_first = it, it_merged_last = it;
uint32_t transfer_rectangle_count =
transfer_invocation_first.transfer.GetRectangles(
dest_rt_key.base_tiles, dest_pitch_tiles,
dest_rt_key.msaa_samples, dest_is_64bpp, nullptr,
resolve_clear_rectangle);
for (auto it_merge = std::next(it_merged_first);
it_merge != current_transfer_invocations_.cend(); ++it_merge) {
if (!transfer_invocation_first.CanBeMergedIntoOneDraw(*it_merge)) {
break;
}
transfer_rectangle_count += it_merge->transfer.GetRectangles(
dest_rt_key.base_tiles, dest_pitch_tiles,
dest_rt_key.msaa_samples, dest_is_64bpp, nullptr,
resolve_clear_rectangle);
it_merged_last = it_merge;
}
assert_not_zero(transfer_rectangle_count);
// Skip the merged transfers in the subsequent iterations.
it = it_merged_last;
assert_not_null(it->transfer.source);
auto& source_vulkan_rt =
*static_cast<VulkanRenderTarget*>(it->transfer.source);
auto host_depth_source_vulkan_rt =
static_cast<VulkanRenderTarget*>(it->transfer.host_depth_source);
TransferShaderKey transfer_shader_key = it->shader_key;
const TransferModeInfo& transfer_mode_info =
kTransferModes[size_t(transfer_shader_key.mode)];
TransferPipelineLayoutIndex transfer_pipeline_layout_index =
transfer_mode_info.pipeline_layout;
const TransferPipelineLayoutInfo& transfer_pipeline_layout_info =
kTransferPipelineLayoutInfos[size_t(
transfer_pipeline_layout_index)];
uint32_t transfer_sample_pipeline_count =
vulkan_device->properties().sampleRateShading
? 1
: uint32_t(1) << uint32_t(dest_rt_key.msaa_samples);
bool transfer_is_stencil_bit =
(transfer_pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordStencilMaskBit) != 0;
uint32_t transfer_vertex_count = 6 * transfer_rectangle_count;
VkBuffer transfer_vertex_buffer;
VkDeviceSize transfer_vertex_buffer_offset;
float* transfer_rectangle_write_ptr =
reinterpret_cast<float*>(transfer_vertex_buffer_pool_->Request(
current_submission, sizeof(float) * 2 * transfer_vertex_count,
sizeof(float), transfer_vertex_buffer,
transfer_vertex_buffer_offset));
if (!transfer_rectangle_write_ptr) {
continue;
}
for (auto it_merged = it_merged_first; it_merged <= it_merged_last;
++it_merged) {
Transfer::Rectangle transfer_invocation_rectangles
[Transfer::kMaxRectanglesWithCutout];
uint32_t transfer_invocation_rectangle_count =
it_merged->transfer.GetRectangles(
dest_rt_key.base_tiles, dest_pitch_tiles,
dest_rt_key.msaa_samples, dest_is_64bpp,
transfer_invocation_rectangles, resolve_clear_rectangle);
assert_not_zero(transfer_invocation_rectangle_count);
for (uint32_t j = 0; j < transfer_invocation_rectangle_count; ++j) {
const Transfer::Rectangle& transfer_rectangle =
transfer_invocation_rectangles[j];
float transfer_rectangle_x0 =
-1.0f + transfer_rectangle.x_pixels * pixels_to_ndc_x;
float transfer_rectangle_y0 =
-1.0f + transfer_rectangle.y_pixels * pixels_to_ndc_y;
float transfer_rectangle_x1 =
transfer_rectangle_x0 +
transfer_rectangle.width_pixels * pixels_to_ndc_x;
float transfer_rectangle_y1 =
transfer_rectangle_y0 +
transfer_rectangle.height_pixels * pixels_to_ndc_y;
// O-*
// |/
// *
*(transfer_rectangle_write_ptr++) = transfer_rectangle_x0;
*(transfer_rectangle_write_ptr++) = transfer_rectangle_y0;
// *-*
// |/
// O
*(transfer_rectangle_write_ptr++) = transfer_rectangle_x0;
*(transfer_rectangle_write_ptr++) = transfer_rectangle_y1;
// *-O
// |/
// *
*(transfer_rectangle_write_ptr++) = transfer_rectangle_x1;
*(transfer_rectangle_write_ptr++) = transfer_rectangle_y0;
// O
// /|
// *-*
*(transfer_rectangle_write_ptr++) = transfer_rectangle_x1;
*(transfer_rectangle_write_ptr++) = transfer_rectangle_y0;
// *
// /|
// O-*
*(transfer_rectangle_write_ptr++) = transfer_rectangle_x0;
*(transfer_rectangle_write_ptr++) = transfer_rectangle_y1;
// *
// /|
// *-O
*(transfer_rectangle_write_ptr++) = transfer_rectangle_x1;
*(transfer_rectangle_write_ptr++) = transfer_rectangle_y1;
}
}
command_buffer.CmdVkBindVertexBuffers(0, 1, &transfer_vertex_buffer,
&transfer_vertex_buffer_offset);
const VkPipeline* transfer_pipelines = GetTransferPipelines(
TransferPipelineKey(transfer_render_pass_key, transfer_shader_key));
if (!transfer_pipelines) {
continue;
}
command_processor_.BindExternalGraphicsPipeline(transfer_pipelines[0]);
if (last_transfer_pipeline_layout_index !=
transfer_pipeline_layout_index) {
last_transfer_pipeline_layout_index = transfer_pipeline_layout_index;
transfer_descriptor_sets_bound = 0;
transfer_push_constants_set = 0;
}
// Invalidate outdated bindings.
if (transfer_pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetHostDepthStencilTexturesBit) {
assert_not_null(host_depth_source_vulkan_rt);
VkDescriptorSet descriptor_set_host_depth_stencil_textures =
host_depth_source_vulkan_rt->GetDescriptorSetTransferSource();
if (last_descriptor_set_host_depth_stencil_textures !=
descriptor_set_host_depth_stencil_textures) {
last_descriptor_set_host_depth_stencil_textures =
descriptor_set_host_depth_stencil_textures;
transfer_descriptor_sets_bound &=
~kTransferUsedDescriptorSetHostDepthStencilTexturesBit;
}
}
if (transfer_pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetDepthStencilTexturesBit) {
VkDescriptorSet descriptor_set_depth_stencil_textures =
source_vulkan_rt.GetDescriptorSetTransferSource();
if (last_descriptor_set_depth_stencil_textures !=
descriptor_set_depth_stencil_textures) {
last_descriptor_set_depth_stencil_textures =
descriptor_set_depth_stencil_textures;
transfer_descriptor_sets_bound &=
~kTransferUsedDescriptorSetDepthStencilTexturesBit;
}
}
if (transfer_pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetColorTextureBit) {
VkDescriptorSet descriptor_set_color_texture =
source_vulkan_rt.GetDescriptorSetTransferSource();
if (last_descriptor_set_color_texture !=
descriptor_set_color_texture) {
last_descriptor_set_color_texture = descriptor_set_color_texture;
transfer_descriptor_sets_bound &=
~kTransferUsedDescriptorSetColorTextureBit;
}
}
if (transfer_pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordHostDepthAddressBit) {
assert_not_null(host_depth_source_vulkan_rt);
RenderTargetKey host_depth_source_rt_key =
host_depth_source_vulkan_rt->key();
TransferAddressConstant host_depth_address_constant;
host_depth_address_constant.dest_pitch = dest_pitch_tiles;
host_depth_address_constant.source_pitch =
host_depth_source_rt_key.GetPitchTiles();
host_depth_address_constant.source_to_dest =
int32_t(dest_rt_key.base_tiles) -
int32_t(host_depth_source_rt_key.base_tiles);
if (last_host_depth_address_constant != host_depth_address_constant) {
last_host_depth_address_constant = host_depth_address_constant;
transfer_push_constants_set &=
~kTransferUsedPushConstantDwordHostDepthAddressBit;
}
}
if (transfer_pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordAddressBit) {
RenderTargetKey source_rt_key = source_vulkan_rt.key();
TransferAddressConstant address_constant;
address_constant.dest_pitch = dest_pitch_tiles;
address_constant.source_pitch = source_rt_key.GetPitchTiles();
address_constant.source_to_dest = int32_t(dest_rt_key.base_tiles) -
int32_t(source_rt_key.base_tiles);
if (last_address_constant != address_constant) {
last_address_constant = address_constant;
transfer_push_constants_set &=
~kTransferUsedPushConstantDwordAddressBit;
}
}
// Apply the new bindings.
// TODO(Triang3l): Merge binding updates into spans.
VkPipelineLayout transfer_pipeline_layout =
transfer_pipeline_layouts_[size_t(transfer_pipeline_layout_index)];
uint32_t transfer_descriptor_sets_unbound =
transfer_pipeline_layout_info.used_descriptor_sets &
~transfer_descriptor_sets_bound;
if (transfer_descriptor_sets_unbound &
kTransferUsedDescriptorSetHostDepthBufferBit) {
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_GRAPHICS, transfer_pipeline_layout,
xe::bit_count(transfer_pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetHostDepthBufferBit - 1)),
1, &edram_storage_buffer_descriptor_set_, 0, nullptr);
transfer_descriptor_sets_bound |=
kTransferUsedDescriptorSetHostDepthBufferBit;
}
if (transfer_descriptor_sets_unbound &
kTransferUsedDescriptorSetHostDepthStencilTexturesBit) {
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_GRAPHICS, transfer_pipeline_layout,
xe::bit_count(
transfer_pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetHostDepthStencilTexturesBit - 1)),
1, &last_descriptor_set_host_depth_stencil_textures, 0, nullptr);
transfer_descriptor_sets_bound |=
kTransferUsedDescriptorSetHostDepthStencilTexturesBit;
}
if (transfer_descriptor_sets_unbound &
kTransferUsedDescriptorSetDepthStencilTexturesBit) {
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_GRAPHICS, transfer_pipeline_layout,
xe::bit_count(
transfer_pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetDepthStencilTexturesBit - 1)),
1, &last_descriptor_set_depth_stencil_textures, 0, nullptr);
transfer_descriptor_sets_bound |=
kTransferUsedDescriptorSetDepthStencilTexturesBit;
}
if (transfer_descriptor_sets_unbound &
kTransferUsedDescriptorSetColorTextureBit) {
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_GRAPHICS, transfer_pipeline_layout,
xe::bit_count(transfer_pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetColorTextureBit - 1)),
1, &last_descriptor_set_color_texture, 0, nullptr);
transfer_descriptor_sets_bound |=
kTransferUsedDescriptorSetColorTextureBit;
}
uint32_t transfer_push_constants_unset =
transfer_pipeline_layout_info.used_push_constant_dwords &
~transfer_push_constants_set;
if (transfer_push_constants_unset &
kTransferUsedPushConstantDwordHostDepthAddressBit) {
command_buffer.CmdVkPushConstants(
transfer_pipeline_layout, VK_SHADER_STAGE_FRAGMENT_BIT,
sizeof(uint32_t) *
xe::bit_count(
transfer_pipeline_layout_info.used_push_constant_dwords &
(kTransferUsedPushConstantDwordHostDepthAddressBit - 1)),
sizeof(uint32_t), &last_host_depth_address_constant);
transfer_push_constants_set |=
kTransferUsedPushConstantDwordHostDepthAddressBit;
}
if (transfer_push_constants_unset &
kTransferUsedPushConstantDwordAddressBit) {
command_buffer.CmdVkPushConstants(
transfer_pipeline_layout, VK_SHADER_STAGE_FRAGMENT_BIT,
sizeof(uint32_t) *
xe::bit_count(
transfer_pipeline_layout_info.used_push_constant_dwords &
(kTransferUsedPushConstantDwordAddressBit - 1)),
sizeof(uint32_t), &last_address_constant);
transfer_push_constants_set |=
kTransferUsedPushConstantDwordAddressBit;
}
for (uint32_t j = 0; j < transfer_sample_pipeline_count; ++j) {
if (j) {
command_processor_.BindExternalGraphicsPipeline(
transfer_pipelines[j]);
}
for (uint32_t k = 0; k < uint32_t(transfer_is_stencil_bit ? 8 : 1);
++k) {
if (transfer_is_stencil_bit) {
uint32_t transfer_stencil_bit = uint32_t(1) << k;
command_buffer.CmdVkPushConstants(
transfer_pipeline_layout, VK_SHADER_STAGE_FRAGMENT_BIT,
sizeof(uint32_t) *
xe::bit_count(
transfer_pipeline_layout_info
.used_push_constant_dwords &
(kTransferUsedPushConstantDwordStencilMaskBit - 1)),
sizeof(uint32_t), &transfer_stencil_bit);
command_buffer.CmdVkSetStencilWriteMask(
VK_STENCIL_FACE_FRONT_AND_BACK, transfer_stencil_bit);
}
command_buffer.CmdVkDraw(transfer_vertex_count, 1, 0, 0);
}
}
}
}
// Perform the clear.
if (resolve_clear_needed) {
command_processor_.SubmitBarriersAndEnterRenderTargetCacheRenderPass(
transfer_render_pass, transfer_framebuffer);
VkClearAttachment resolve_clear_attachment;
resolve_clear_attachment.colorAttachment = 0;
std::memset(&resolve_clear_attachment.clearValue, 0,
sizeof(resolve_clear_attachment.clearValue));
uint64_t clear_value = render_target_resolve_clear_values[i];
if (dest_rt_key.is_depth) {
resolve_clear_attachment.aspectMask =
VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT;
uint32_t depth_guest_clear_value =
(uint32_t(clear_value) >> 8) & 0xFFFFFF;
switch (dest_rt_key.GetDepthFormat()) {
case xenos::DepthRenderTargetFormat::kD24S8:
resolve_clear_attachment.clearValue.depthStencil.depth =
xenos::UNorm24To32(depth_guest_clear_value);
break;
case xenos::DepthRenderTargetFormat::kD24FS8:
// Taking [0, 2) -> [0, 1) remapping into account.
resolve_clear_attachment.clearValue.depthStencil.depth =
xenos::Float20e4To32(depth_guest_clear_value) * 0.5f;
break;
}
resolve_clear_attachment.clearValue.depthStencil.stencil =
uint32_t(clear_value) & 0xFF;
} else {
resolve_clear_attachment.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
switch (dest_rt_key.GetColorFormat()) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA: {
for (uint32_t j = 0; j < 4; ++j) {
resolve_clear_attachment.clearValue.color.float32[j] =
((clear_value >> (j * 8)) & 0xFF) * (1.0f / 0xFF);
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10: {
for (uint32_t j = 0; j < 3; ++j) {
resolve_clear_attachment.clearValue.color.float32[j] =
((clear_value >> (j * 10)) & 0x3FF) * (1.0f / 0x3FF);
}
resolve_clear_attachment.clearValue.color.float32[3] =
((clear_value >> 30) & 0x3) * (1.0f / 0x3);
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::
k_2_10_10_10_FLOAT_AS_16_16_16_16: {
for (uint32_t j = 0; j < 3; ++j) {
resolve_clear_attachment.clearValue.color.float32[j] =
xenos::Float7e3To32((clear_value >> (j * 10)) & 0x3FF);
}
resolve_clear_attachment.clearValue.color.float32[3] =
((clear_value >> 30) & 0x3) * (1.0f / 0x3);
} break;
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT: {
// Using uint for transfers and clears of both. Disregarding the
// current -32...32 vs. -1...1 settings for consistency with color
// clear via depth aliasing.
// TODO(Triang3l): Handle cases of unsupported multisampled 16_UINT
// and completely unsupported 16_UNORM.
for (uint32_t j = 0; j < 2; ++j) {
resolve_clear_attachment.clearValue.color.uint32[j] =
uint32_t(clear_value >> (j * 16)) & 0xFFFF;
}
} break;
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT: {
// Using uint for transfers and clears of both. Disregarding the
// current -32...32 vs. -1...1 settings for consistency with color
// clear via depth aliasing.
// TODO(Triang3l): Handle cases of unsupported multisampled 16_UINT
// and completely unsupported 16_UNORM.
for (uint32_t j = 0; j < 4; ++j) {
resolve_clear_attachment.clearValue.color.uint32[j] =
uint32_t(clear_value >> (j * 16)) & 0xFFFF;
}
} break;
case xenos::ColorRenderTargetFormat::k_32_FLOAT: {
// Using uint for proper denormal and NaN handling.
resolve_clear_attachment.clearValue.color.uint32[0] =
uint32_t(clear_value);
} break;
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT: {
// Using uint for proper denormal and NaN handling.
resolve_clear_attachment.clearValue.color.uint32[0] =
uint32_t(clear_value);
resolve_clear_attachment.clearValue.color.uint32[1] =
uint32_t(clear_value >> 32);
} break;
}
}
command_buffer.CmdVkClearAttachments(1, &resolve_clear_attachment, 1,
&resolve_clear_rect);
}
}
}
VkPipeline VulkanRenderTargetCache::GetDumpPipeline(DumpPipelineKey key) {
auto pipeline_it = dump_pipelines_.find(key);
if (pipeline_it != dump_pipelines_.end()) {
return pipeline_it->second;
}
std::vector<spv::Id> id_vector_temp;
SpirvBuilder builder(spv::Spv_1_0,
(SpirvShaderTranslator::kSpirvMagicToolId << 16) | 1,
nullptr);
spv::Id ext_inst_glsl_std_450 = builder.import("GLSL.std.450");
builder.addCapability(spv::CapabilityShader);
builder.setMemoryModel(spv::AddressingModelLogical, spv::MemoryModelGLSL450);
builder.setSource(spv::SourceLanguageUnknown, 0);
spv::Id type_void = builder.makeVoidType();
spv::Id type_int = builder.makeIntType(32);
spv::Id type_int2 = builder.makeVectorType(type_int, 2);
spv::Id type_uint = builder.makeUintType(32);
spv::Id type_uint2 = builder.makeVectorType(type_uint, 2);
spv::Id type_uint3 = builder.makeVectorType(type_uint, 3);
spv::Id type_float = builder.makeFloatType(32);
// Bindings.
// EDRAM buffer.
bool format_is_64bpp = !key.is_depth && xenos::IsColorRenderTargetFormat64bpp(
key.GetColorFormat());
id_vector_temp.clear();
id_vector_temp.push_back(
builder.makeRuntimeArray(format_is_64bpp ? type_uint2 : type_uint));
// Storage buffers have std430 packing, no padding to 4-component vectors.
builder.addDecoration(id_vector_temp.back(), spv::DecorationArrayStride,
sizeof(uint32_t) << uint32_t(format_is_64bpp));
spv::Id type_edram = builder.makeStructType(id_vector_temp, "XeEdram");
builder.addMemberName(type_edram, 0, "edram");
builder.addMemberDecoration(type_edram, 0, spv::DecorationNonReadable);
builder.addMemberDecoration(type_edram, 0, spv::DecorationOffset, 0);
// Block since SPIR-V 1.3, but since SPIR-V 1.0 is generated, it's
// BufferBlock.
builder.addDecoration(type_edram, spv::DecorationBufferBlock);
// StorageBuffer since SPIR-V 1.3, but since SPIR-V 1.0 is generated, it's
// Uniform.
spv::Id edram_buffer = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniform, type_edram, "xe_edram");
builder.addDecoration(edram_buffer, spv::DecorationDescriptorSet,
kDumpDescriptorSetEdram);
builder.addDecoration(edram_buffer, spv::DecorationBinding, 0);
// Color or depth source.
bool source_is_multisampled = key.msaa_samples != xenos::MsaaSamples::k1X;
bool source_is_uint;
if (key.is_depth) {
source_is_uint = false;
} else {
GetColorOwnershipTransferVulkanFormat(key.GetColorFormat(),
&source_is_uint);
}
spv::Id source_component_type = source_is_uint ? type_uint : type_float;
spv::Id source_texture = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniformConstant,
builder.makeImageType(source_component_type, spv::Dim2D, false, false,
source_is_multisampled, 1, spv::ImageFormatUnknown),
"xe_edram_dump_source");
builder.addDecoration(source_texture, spv::DecorationDescriptorSet,
kDumpDescriptorSetSource);
builder.addDecoration(source_texture, spv::DecorationBinding, 0);
// Stencil source.
spv::Id source_stencil_texture = spv::NoResult;
if (key.is_depth) {
source_stencil_texture = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniformConstant,
builder.makeImageType(type_uint, spv::Dim2D, false, false,
source_is_multisampled, 1,
spv::ImageFormatUnknown),
"xe_edram_dump_stencil");
builder.addDecoration(source_stencil_texture, spv::DecorationDescriptorSet,
kDumpDescriptorSetSource);
builder.addDecoration(source_stencil_texture, spv::DecorationBinding, 1);
}
// Push constants.
id_vector_temp.clear();
id_vector_temp.reserve(kDumpPushConstantCount);
for (uint32_t i = 0; i < kDumpPushConstantCount; ++i) {
id_vector_temp.push_back(type_uint);
}
spv::Id type_push_constants =
builder.makeStructType(id_vector_temp, "XeEdramDumpPushConstants");
builder.addMemberName(type_push_constants, kDumpPushConstantPitches,
"pitches");
builder.addMemberDecoration(type_push_constants, kDumpPushConstantPitches,
spv::DecorationOffset,
int(sizeof(uint32_t) * kDumpPushConstantPitches));
builder.addMemberName(type_push_constants, kDumpPushConstantOffsets,
"offsets");
builder.addMemberDecoration(type_push_constants, kDumpPushConstantOffsets,
spv::DecorationOffset,
int(sizeof(uint32_t) * kDumpPushConstantOffsets));
builder.addDecoration(type_push_constants, spv::DecorationBlock);
spv::Id push_constants = builder.createVariable(
spv::NoPrecision, spv::StorageClassPushConstant, type_push_constants,
"xe_edram_dump_push_constants");
// gl_GlobalInvocationID input.
spv::Id input_global_invocation_id =
builder.createVariable(spv::NoPrecision, spv::StorageClassInput,
type_uint3, "gl_GlobalInvocationID");
builder.addDecoration(input_global_invocation_id, spv::DecorationBuiltIn,
spv::BuiltInGlobalInvocationId);
// Begin the main function.
std::vector<spv::Id> main_param_types;
std::vector<std::vector<spv::Decoration>> main_precisions;
spv::Block* main_entry;
spv::Function* main_function =
builder.makeFunctionEntry(spv::NoPrecision, type_void, "main",
main_param_types, main_precisions, &main_entry);
// For now, as the exact addressing in 64bpp render targets relatively to
// 32bpp is unknown, treating 64bpp tiles as storing 40x16 samples rather than
// 80x16 for simplicity of addressing into the texture.
// Split the destination sample index into the 32bpp tile and the
// 32bpp-tile-relative sample index.
// Note that division by non-power-of-two constants will include a 4-cycle
// 32*32 multiplication on AMD, even though so many bits are not needed for
// the sample position - however, if an OpUnreachable path is inserted for the
// case when the position has upper bits set, for some reason, the code for it
// is not eliminated when compiling the shader for AMD via RenderDoc on
// Windows, as of June 2022.
spv::Id global_invocation_id =
builder.createLoad(input_global_invocation_id, spv::NoPrecision);
spv::Id rectangle_sample_x =
builder.createCompositeExtract(global_invocation_id, type_uint, 0);
uint32_t tile_width =
(xenos::kEdramTileWidthSamples >> uint32_t(format_is_64bpp)) *
draw_resolution_scale_x();
spv::Id const_tile_width = builder.makeUintConstant(tile_width);
spv::Id rectangle_tile_index_x = builder.createBinOp(
spv::OpUDiv, type_uint, rectangle_sample_x, const_tile_width);
spv::Id tile_sample_x = builder.createBinOp(
spv::OpUMod, type_uint, rectangle_sample_x, const_tile_width);
spv::Id rectangle_sample_y =
builder.createCompositeExtract(global_invocation_id, type_uint, 1);
uint32_t tile_height =
xenos::kEdramTileHeightSamples * draw_resolution_scale_y();
spv::Id const_tile_height = builder.makeUintConstant(tile_height);
spv::Id rectangle_tile_index_y = builder.createBinOp(
spv::OpUDiv, type_uint, rectangle_sample_y, const_tile_height);
spv::Id tile_sample_y = builder.createBinOp(
spv::OpUMod, type_uint, rectangle_sample_y, const_tile_height);
// Get the tile index in the EDRAM relative to the dump rectangle base tile.
id_vector_temp.clear();
id_vector_temp.push_back(builder.makeIntConstant(kDumpPushConstantPitches));
spv::Id pitches_constant = builder.createLoad(
builder.createAccessChain(spv::StorageClassPushConstant, push_constants,
id_vector_temp),
spv::NoPrecision);
spv::Id const_uint_0 = builder.makeUintConstant(0);
spv::Id const_edram_pitch_tiles_bits =
builder.makeUintConstant(xenos::kEdramPitchTilesBits);
spv::Id rectangle_tile_index = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.createTriOp(spv::OpBitFieldUExtract, type_uint,
pitches_constant, const_uint_0,
const_edram_pitch_tiles_bits),
rectangle_tile_index_y),
rectangle_tile_index_x);
// Add the base tile in the dispatch to the dispatch-local tile index, not
// wrapping yet so in case of a wraparound, the address relative to the base
// in the image after subtraction of the base won't be negative.
id_vector_temp.clear();
id_vector_temp.push_back(builder.makeIntConstant(kDumpPushConstantOffsets));
spv::Id offsets_constant = builder.createLoad(
builder.createAccessChain(spv::StorageClassPushConstant, push_constants,
id_vector_temp),
spv::NoPrecision);
spv::Id const_edram_base_tiles_bits_plus_1 =
builder.makeUintConstant(xenos::kEdramBaseTilesBits + 1);
spv::Id edram_tile_index_non_wrapped = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createTriOp(spv::OpBitFieldUExtract, type_uint, offsets_constant,
const_uint_0, const_edram_base_tiles_bits_plus_1),
rectangle_tile_index);
// Combine the tile sample index and the tile index, wrapping the tile
// addressing, into the EDRAM sample index.
spv::Id edram_sample_address = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.makeUintConstant(tile_width * tile_height),
builder.createBinOp(
spv::OpBitwiseAnd, type_uint, edram_tile_index_non_wrapped,
builder.makeUintConstant(xenos::kEdramTileCount - 1))),
builder.createBinOp(spv::OpIAdd, type_uint,
builder.createBinOp(spv::OpIMul, type_uint,
const_tile_width, tile_sample_y),
tile_sample_x));
if (key.is_depth) {
// Swap 40-sample columns in the depth buffer in the destination address to
// get the final address of the sample in the EDRAM.
uint32_t tile_width_half = tile_width >> 1;
edram_sample_address = builder.createUnaryOp(
spv::OpBitcast, type_uint,
builder.createBinOp(
spv::OpIAdd, type_int,
builder.createUnaryOp(spv::OpBitcast, type_int,
edram_sample_address),
builder.createTriOp(
spv::OpSelect, type_int,
builder.createBinOp(spv::OpULessThan, builder.makeBoolType(),
tile_sample_x,
builder.makeUintConstant(tile_width_half)),
builder.makeIntConstant(int32_t(tile_width_half)),
builder.makeIntConstant(-int32_t(tile_width_half)))));
}
// Get the linear tile index within the source texture.
spv::Id source_tile_index = builder.createBinOp(
spv::OpISub, type_uint, edram_tile_index_non_wrapped,
builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, offsets_constant,
const_edram_base_tiles_bits_plus_1,
builder.makeUintConstant(xenos::kEdramBaseTilesBits)));
// Split the linear tile index in the source texture into X and Y in tiles.
spv::Id source_pitch_tiles = builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, pitches_constant,
const_edram_pitch_tiles_bits, const_edram_pitch_tiles_bits);
spv::Id source_tile_index_y = builder.createBinOp(
spv::OpUDiv, type_uint, source_tile_index, source_pitch_tiles);
spv::Id source_tile_index_x = builder.createBinOp(
spv::OpUMod, type_uint, source_tile_index, source_pitch_tiles);
// Combine the source tile offset and the sample index within the tile.
spv::Id source_sample_x = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(spv::OpIMul, type_uint, const_tile_width,
source_tile_index_x),
tile_sample_x);
spv::Id source_sample_y = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(spv::OpIMul, type_uint, const_tile_height,
source_tile_index_y),
tile_sample_y);
// Get the source pixel coordinate and the sample index within the pixel.
spv::Id source_pixel_x = source_sample_x, source_pixel_y = source_sample_y;
spv::Id source_sample_id = spv::NoResult;
if (source_is_multisampled) {
spv::Id const_uint_1 = builder.makeUintConstant(1);
source_pixel_y = builder.createBinOp(spv::OpShiftRightLogical, type_uint,
source_sample_y, const_uint_1);
if (key.msaa_samples >= xenos::MsaaSamples::k4X) {
source_pixel_x = builder.createBinOp(spv::OpShiftRightLogical, type_uint,
source_sample_x, const_uint_1);
// 4x MSAA source texture sample index - bit 0 for horizontal, bit 1 for
// vertical.
source_sample_id = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint,
builder.createBinOp(spv::OpBitwiseAnd, type_uint, source_sample_x,
const_uint_1),
source_sample_y, const_uint_1, const_uint_1);
} else {
// 2x MSAA source texture sample index - convert from the guest to
// the Vulkan standard sample locations.
source_sample_id = builder.createTriOp(
spv::OpSelect, type_uint,
builder.createBinOp(
spv::OpINotEqual, builder.makeBoolType(),
builder.createBinOp(spv::OpBitwiseAnd, type_uint, source_sample_y,
const_uint_1),
const_uint_0),
builder.makeUintConstant(draw_util::GetD3D10SampleIndexForGuest2xMSAA(
1, msaa_2x_attachments_supported_)),
builder.makeUintConstant(draw_util::GetD3D10SampleIndexForGuest2xMSAA(
0, msaa_2x_attachments_supported_)));
}
}
// Load the source, and pack the value into one or two 32-bit integers.
spv::Id packed[2] = {};
spv::Builder::TextureParameters source_texture_parameters = {};
source_texture_parameters.sampler =
builder.createLoad(source_texture, spv::NoPrecision);
id_vector_temp.clear();
id_vector_temp.push_back(
builder.createUnaryOp(spv::OpBitcast, type_int, source_pixel_x));
id_vector_temp.push_back(
builder.createUnaryOp(spv::OpBitcast, type_int, source_pixel_y));
source_texture_parameters.coords =
builder.createCompositeConstruct(type_int2, id_vector_temp);
if (source_is_multisampled) {
source_texture_parameters.sample =
builder.createUnaryOp(spv::OpBitcast, type_int, source_sample_id);
} else {
source_texture_parameters.lod = builder.makeIntConstant(0);
}
spv::Id source_vec4 = builder.createTextureCall(
spv::NoPrecision, builder.makeVectorType(source_component_type, 4), false,
true, false, false, false, source_texture_parameters,
spv::ImageOperandsMaskNone);
if (key.is_depth) {
source_texture_parameters.sampler =
builder.createLoad(source_stencil_texture, spv::NoPrecision);
spv::Id source_stencil = builder.createCompositeExtract(
builder.createTextureCall(
spv::NoPrecision, builder.makeVectorType(type_uint, 4), false, true,
false, false, false, source_texture_parameters,
spv::ImageOperandsMaskNone),
type_uint, 0);
spv::Id source_depth32 =
builder.createCompositeExtract(source_vec4, type_float, 0);
switch (key.GetDepthFormat()) {
case xenos::DepthRenderTargetFormat::kD24S8: {
// Round to the nearest even integer. This seems to be the correct
// conversion, adding +0.5 and rounding towards zero results in red
// instead of black in the 4D5307E6 clear shader.
packed[0] = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createUnaryBuiltinCall(
type_float, ext_inst_glsl_std_450, GLSLstd450RoundEven,
builder.createBinOp(
spv::OpFMul, type_float, source_depth32,
builder.makeFloatConstant(float(0xFFFFFF)))));
} break;
case xenos::DepthRenderTargetFormat::kD24FS8: {
packed[0] = SpirvShaderTranslator::PreClampedDepthTo20e4(
builder, source_depth32, depth_float24_round(), true,
ext_inst_glsl_std_450);
} break;
}
packed[0] = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, source_stencil, packed[0],
builder.makeUintConstant(8), builder.makeUintConstant(24));
} else {
switch (key.GetColorFormat()) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA: {
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale = builder.makeFloatConstant(255.0f);
packed[0] = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float,
builder.createCompositeExtract(source_vec4, type_float, 0),
unorm_scale),
unorm_round_offset));
spv::Id component_width = builder.makeUintConstant(8);
for (uint32_t i = 1; i < 4; ++i) {
packed[0] = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed[0],
builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
builder.createCompositeExtract(
source_vec4, type_float, i),
unorm_scale),
unorm_round_offset)),
builder.makeUintConstant(8 * i), component_width);
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10: {
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale_rgb = builder.makeFloatConstant(1023.0f);
packed[0] = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float,
builder.createCompositeExtract(source_vec4, type_float, 0),
unorm_scale_rgb),
unorm_round_offset));
spv::Id width_rgb = builder.makeUintConstant(10);
spv::Id unorm_scale_a = builder.makeFloatConstant(3.0f);
spv::Id width_a = builder.makeUintConstant(2);
for (uint32_t i = 1; i < 4; ++i) {
packed[0] = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed[0],
builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float,
builder.createCompositeExtract(source_vec4,
type_float, i),
i == 3 ? unorm_scale_a : unorm_scale_rgb),
unorm_round_offset)),
builder.makeUintConstant(10 * i), i == 3 ? width_a : width_rgb);
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT_AS_16_16_16_16: {
// Float16 has a wider range for both color and alpha, also NaNs - clamp
// and convert.
packed[0] = SpirvShaderTranslator::UnclampedFloat32To7e3(
builder, builder.createCompositeExtract(source_vec4, type_float, 0),
ext_inst_glsl_std_450);
spv::Id width_rgb = builder.makeUintConstant(10);
for (uint32_t i = 1; i < 3; ++i) {
packed[0] = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed[0],
SpirvShaderTranslator::UnclampedFloat32To7e3(
builder,
builder.createCompositeExtract(source_vec4, type_float, i),
ext_inst_glsl_std_450),
builder.makeUintConstant(10 * i), width_rgb);
}
// Saturate and convert the alpha.
spv::Id alpha_saturated = builder.createTriBuiltinCall(
type_float, ext_inst_glsl_std_450, GLSLstd450NClamp,
builder.createCompositeExtract(source_vec4, type_float, 3),
builder.makeFloatConstant(0.0f), builder.makeFloatConstant(1.0f));
packed[0] = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint, packed[0],
builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
alpha_saturated,
builder.makeFloatConstant(3.0f)),
builder.makeFloatConstant(0.5f))),
builder.makeUintConstant(30), builder.makeUintConstant(2));
} break;
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT:
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT: {
// All 64bpp formats, and all 16 bits per component formats, are
// represented as integers in ownership transfer for safe handling of
// NaN encodings and -32768 / -32767.
// TODO(Triang3l): Handle the case when that's not true (no multisampled
// sampled images, no 16-bit UNORM, no cross-packing 32bpp aliasing on a
// portability subset device or a 64bpp format where that wouldn't help
// anyway).
spv::Id component_offset_width = builder.makeUintConstant(16);
for (uint32_t i = 0; i <= uint32_t(format_is_64bpp); ++i) {
packed[i] = builder.createQuadOp(
spv::OpBitFieldInsert, type_uint,
builder.createCompositeExtract(source_vec4, type_uint, 2 * i),
builder.createCompositeExtract(source_vec4, type_uint, 2 * i + 1),
component_offset_width, component_offset_width);
}
} break;
// Float32 is transferred as uint32 to preserve NaN encodings. However,
// multisampled sampled image support is optional in Vulkan.
case xenos::ColorRenderTargetFormat::k_32_FLOAT:
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT: {
for (uint32_t i = 0; i <= uint32_t(format_is_64bpp); ++i) {
spv::Id& packed_ref = packed[i];
packed_ref = builder.createCompositeExtract(source_vec4,
source_component_type, i);
if (!source_is_uint) {
packed_ref =
builder.createUnaryOp(spv::OpBitcast, type_uint, packed_ref);
}
}
} break;
}
}
// Write the packed value to the EDRAM buffer.
spv::Id store_value = packed[0];
if (format_is_64bpp) {
id_vector_temp.clear();
id_vector_temp.push_back(packed[0]);
id_vector_temp.push_back(packed[1]);
store_value = builder.createCompositeConstruct(type_uint2, id_vector_temp);
}
id_vector_temp.clear();
// The only SSBO structure member.
id_vector_temp.push_back(builder.makeIntConstant(0));
id_vector_temp.push_back(
builder.createUnaryOp(spv::OpBitcast, type_int, edram_sample_address));
// StorageBuffer since SPIR-V 1.3, but since SPIR-V 1.0 is generated, it's
// Uniform.
builder.createStore(store_value,
builder.createAccessChain(spv::StorageClassUniform,
edram_buffer, id_vector_temp));
// End the main function and make it the entry point.
builder.leaveFunction();
builder.addExecutionMode(main_function, spv::ExecutionModeLocalSize,
kDumpSamplesPerGroupX, kDumpSamplesPerGroupY, 1);
spv::Instruction* entry_point = builder.addEntryPoint(
spv::ExecutionModelGLCompute, main_function, "main");
// Bindings only need to be added to the entry point's interface starting with
// SPIR-V 1.4 - emitting 1.0 here, so only inputs / outputs.
entry_point->addIdOperand(input_global_invocation_id);
// Serialize the shader code.
std::vector<unsigned int> shader_code;
builder.dump(shader_code);
// Create the pipeline, and store the handle even if creation fails not to try
// to create it again later.
VkPipeline pipeline = ui::vulkan::util::CreateComputePipeline(
command_processor_.GetVulkanDevice(),
key.is_depth ? dump_pipeline_layout_depth_ : dump_pipeline_layout_color_,
reinterpret_cast<const uint32_t*>(shader_code.data()),
sizeof(uint32_t) * shader_code.size());
if (pipeline == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create a render target dumping "
"pipeline for {}-sample render targets with format {}",
UINT32_C(1) << uint32_t(key.msaa_samples),
key.is_depth
? xenos::GetDepthRenderTargetFormatName(key.GetDepthFormat())
: xenos::GetColorRenderTargetFormatName(key.GetColorFormat()));
}
dump_pipelines_.emplace(key, pipeline);
return pipeline;
}
void VulkanRenderTargetCache::DumpRenderTargets(uint32_t dump_base,
uint32_t dump_row_length_used,
uint32_t dump_rows,
uint32_t dump_pitch) {
assert_true(GetPath() == Path::kHostRenderTargets);
GetResolveCopyRectanglesToDump(dump_base, dump_row_length_used, dump_rows,
dump_pitch, dump_rectangles_);
if (dump_rectangles_.empty()) {
return;
}
// Clear previously set temporary indices.
for (const ResolveCopyDumpRectangle& rectangle : dump_rectangles_) {
static_cast<VulkanRenderTarget*>(rectangle.render_target)
->SetTemporarySortIndex(UINT32_MAX);
}
// Gather all needed barriers and info needed to sort the invocations.
UseEdramBuffer(EdramBufferUsage::kComputeWrite);
dump_invocations_.clear();
dump_invocations_.reserve(dump_rectangles_.size());
constexpr VkPipelineStageFlags kRenderTargetDstStageMask =
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT;
constexpr VkAccessFlags kRenderTargetDstAccessMask =
VK_ACCESS_SHADER_READ_BIT;
constexpr VkImageLayout kRenderTargetNewLayout =
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
uint32_t rt_sort_index = 0;
for (const ResolveCopyDumpRectangle& rectangle : dump_rectangles_) {
auto& vulkan_rt =
*static_cast<VulkanRenderTarget*>(rectangle.render_target);
RenderTargetKey rt_key = vulkan_rt.key();
command_processor_.PushImageMemoryBarrier(
vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
rt_key.is_depth
? (VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)
: VK_IMAGE_ASPECT_COLOR_BIT),
vulkan_rt.current_stage_mask(), VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
vulkan_rt.current_access_mask(), VK_ACCESS_SHADER_READ_BIT,
vulkan_rt.current_layout(), VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL);
vulkan_rt.SetUsage(VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
VK_ACCESS_SHADER_READ_BIT,
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL);
if (vulkan_rt.temporary_sort_index() == UINT32_MAX) {
vulkan_rt.SetTemporarySortIndex(rt_sort_index++);
}
DumpPipelineKey pipeline_key;
pipeline_key.msaa_samples = rt_key.msaa_samples;
pipeline_key.resource_format = rt_key.resource_format;
pipeline_key.is_depth = rt_key.is_depth;
dump_invocations_.emplace_back(rectangle, pipeline_key);
}
// Sort the invocations to reduce context and binding switches.
std::sort(dump_invocations_.begin(), dump_invocations_.end());
// Dump the render targets.
DeferredCommandBuffer& command_buffer =
command_processor_.deferred_command_buffer();
bool edram_buffer_bound = false;
VkDescriptorSet last_source_descriptor_set = VK_NULL_HANDLE;
DumpPitches last_pitches;
DumpOffsets last_offsets;
bool pitches_bound = false, offsets_bound = false;
for (const DumpInvocation& invocation : dump_invocations_) {
const ResolveCopyDumpRectangle& rectangle = invocation.rectangle;
auto& vulkan_rt =
*static_cast<VulkanRenderTarget*>(rectangle.render_target);
RenderTargetKey rt_key = vulkan_rt.key();
DumpPipelineKey pipeline_key = invocation.pipeline_key;
VkPipeline pipeline = GetDumpPipeline(pipeline_key);
if (!pipeline) {
continue;
}
command_processor_.BindExternalComputePipeline(pipeline);
VkPipelineLayout pipeline_layout = rt_key.is_depth
? dump_pipeline_layout_depth_
: dump_pipeline_layout_color_;
// Only need to bind the EDRAM buffer once (relying on pipeline layout
// compatibility).
if (!edram_buffer_bound) {
edram_buffer_bound = true;
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_COMPUTE, pipeline_layout,
kDumpDescriptorSetEdram, 1, &edram_storage_buffer_descriptor_set_, 0,
nullptr);
}
VkDescriptorSet source_descriptor_set =
vulkan_rt.GetDescriptorSetTransferSource();
if (last_source_descriptor_set != source_descriptor_set) {
last_source_descriptor_set = source_descriptor_set;
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_COMPUTE, pipeline_layout,
kDumpDescriptorSetSource, 1, &source_descriptor_set, 0, nullptr);
}
DumpPitches pitches;
pitches.dest_pitch = dump_pitch;
pitches.source_pitch = rt_key.GetPitchTiles();
if (last_pitches != pitches) {
last_pitches = pitches;
pitches_bound = false;
}
if (!pitches_bound) {
pitches_bound = true;
command_buffer.CmdVkPushConstants(
pipeline_layout, VK_SHADER_STAGE_COMPUTE_BIT,
sizeof(uint32_t) * kDumpPushConstantPitches, sizeof(last_pitches),
&last_pitches);
}
DumpOffsets offsets;
offsets.source_base_tiles = rt_key.base_tiles;
ResolveCopyDumpRectangle::Dispatch
dispatches[ResolveCopyDumpRectangle::kMaxDispatches];
uint32_t dispatch_count =
rectangle.GetDispatches(dump_pitch, dump_row_length_used, dispatches);
for (uint32_t i = 0; i < dispatch_count; ++i) {
const ResolveCopyDumpRectangle::Dispatch& dispatch = dispatches[i];
offsets.dispatch_first_tile = dump_base + dispatch.offset;
if (last_offsets != offsets) {
last_offsets = offsets;
offsets_bound = false;
}
if (!offsets_bound) {
offsets_bound = true;
command_buffer.CmdVkPushConstants(
pipeline_layout, VK_SHADER_STAGE_COMPUTE_BIT,
sizeof(uint32_t) * kDumpPushConstantOffsets, sizeof(last_offsets),
&last_offsets);
}
command_processor_.SubmitBarriers(true);
command_buffer.CmdVkDispatch(
(draw_resolution_scale_x() *
(xenos::kEdramTileWidthSamples >> uint32_t(rt_key.Is64bpp())) *
dispatch.width_tiles +
(kDumpSamplesPerGroupX - 1)) /
kDumpSamplesPerGroupX,
(draw_resolution_scale_y() * xenos::kEdramTileHeightSamples *
dispatch.height_tiles +
(kDumpSamplesPerGroupY - 1)) /
kDumpSamplesPerGroupY,
1);
}
MarkEdramBufferModified();
}
}
} // namespace vulkan
} // namespace gpu
} // namespace xe