Files
Xenia-Canary/src/xenia/gpu/vulkan/vulkan_render_target_cache.cc
Triang3l 45050b2380 [GPU] Vulkan fragment shader interlock RB and related fixes/cleanup
Also fixes addressing of MSAA samples 2 and 3 for 64bpp color render targets in the ROV RB implementation on Direct3D 12.
Additionally, with FSI/ROV, alpha test and alpha to coverage are done only if the render target 0 was dynamically written to (according to the Direct3D 9 rules for writing to color render targets, though not sure if they actually apply to the alpha tests on Direct3D 9, but for safety).
There is also some code cleanup for things spotted during the development of the feature.
2022-10-09 22:06:41 +03:00

6338 lines
302 KiB
C++

/**
******************************************************************************
* Xenia : Xbox 360 Emulator Research Project *
******************************************************************************
* Copyright 2022 Ben Vanik. All rights reserved. *
* Released under the BSD license - see LICENSE in the root for more details. *
******************************************************************************
*/
#include "xenia/gpu/vulkan/vulkan_render_target_cache.h"
#include <algorithm>
#include <array>
#include <cstddef>
#include <cstdint>
#include <cstring>
#include <memory>
#include <tuple>
#include <utility>
#include <vector>
#include "third_party/glslang/SPIRV/GLSL.std.450.h"
#include "third_party/glslang/SPIRV/SpvBuilder.h"
#include "xenia/base/assert.h"
#include "xenia/base/cvar.h"
#include "xenia/base/logging.h"
#include "xenia/base/math.h"
#include "xenia/gpu/draw_util.h"
#include "xenia/gpu/registers.h"
#include "xenia/gpu/spirv_shader_translator.h"
#include "xenia/gpu/texture_cache.h"
#include "xenia/gpu/vulkan/deferred_command_buffer.h"
#include "xenia/gpu/vulkan/vulkan_command_processor.h"
#include "xenia/gpu/xenos.h"
#include "xenia/ui/vulkan/vulkan_util.h"
DEFINE_string(
render_target_path_vulkan, "",
"Render target emulation path to use on Vulkan.\n"
"Use: [any, fbo, fsi]\n"
" fbo:\n"
" Host framebuffers and fixed-function blending and depth / stencil "
"testing, copying between render targets when needed.\n"
" Lower accuracy (limited pixel format support).\n"
" Performance limited primarily by render target layout changes requiring "
"copying, but generally higher.\n"
" fsi:\n"
" Manual pixel packing, blending and depth / stencil testing, with free "
"render target layout changes.\n"
" Requires a GPU supporting fragment shader interlock.\n"
" Highest accuracy (all pixel formats handled in software).\n"
" Performance limited primarily by overdraw.\n"
" Any other value:\n"
" Choose what is considered the most optimal for the system (currently "
"always FB because the FSI path is much slower now).",
"GPU");
namespace xe {
namespace gpu {
namespace vulkan {
// Generated with `xb buildshaders`.
namespace shaders {
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/host_depth_store_1xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/host_depth_store_2xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/host_depth_store_4xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/passthrough_position_xy_vs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_clear_32bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_clear_32bpp_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_clear_64bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_clear_64bpp_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_32bpp_1x2xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_32bpp_1x2xmsaa_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_32bpp_4xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_32bpp_4xmsaa_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_64bpp_1x2xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_64bpp_1x2xmsaa_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_64bpp_4xmsaa_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_fast_64bpp_4xmsaa_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_128bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_128bpp_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_16bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_16bpp_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_32bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_32bpp_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_64bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_64bpp_scaled_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_8bpp_cs.h"
#include "xenia/gpu/shaders/bytecode/vulkan_spirv/resolve_full_8bpp_scaled_cs.h"
} // namespace shaders
const VulkanRenderTargetCache::ResolveCopyShaderCode
VulkanRenderTargetCache::kResolveCopyShaders[size_t(
draw_util::ResolveCopyShaderIndex::kCount)] = {
{shaders::resolve_fast_32bpp_1x2xmsaa_cs,
sizeof(shaders::resolve_fast_32bpp_1x2xmsaa_cs),
shaders::resolve_fast_32bpp_1x2xmsaa_scaled_cs,
sizeof(shaders::resolve_fast_32bpp_1x2xmsaa_scaled_cs)},
{shaders::resolve_fast_32bpp_4xmsaa_cs,
sizeof(shaders::resolve_fast_32bpp_4xmsaa_cs),
shaders::resolve_fast_32bpp_4xmsaa_scaled_cs,
sizeof(shaders::resolve_fast_32bpp_4xmsaa_scaled_cs)},
{shaders::resolve_fast_64bpp_1x2xmsaa_cs,
sizeof(shaders::resolve_fast_64bpp_1x2xmsaa_cs),
shaders::resolve_fast_64bpp_1x2xmsaa_scaled_cs,
sizeof(shaders::resolve_fast_64bpp_1x2xmsaa_scaled_cs)},
{shaders::resolve_fast_64bpp_4xmsaa_cs,
sizeof(shaders::resolve_fast_64bpp_4xmsaa_cs),
shaders::resolve_fast_64bpp_4xmsaa_scaled_cs,
sizeof(shaders::resolve_fast_64bpp_4xmsaa_scaled_cs)},
{shaders::resolve_full_8bpp_cs, sizeof(shaders::resolve_full_8bpp_cs),
shaders::resolve_full_8bpp_scaled_cs,
sizeof(shaders::resolve_full_8bpp_scaled_cs)},
{shaders::resolve_full_16bpp_cs, sizeof(shaders::resolve_full_16bpp_cs),
shaders::resolve_full_16bpp_scaled_cs,
sizeof(shaders::resolve_full_16bpp_scaled_cs)},
{shaders::resolve_full_32bpp_cs, sizeof(shaders::resolve_full_32bpp_cs),
shaders::resolve_full_32bpp_scaled_cs,
sizeof(shaders::resolve_full_32bpp_scaled_cs)},
{shaders::resolve_full_64bpp_cs, sizeof(shaders::resolve_full_64bpp_cs),
shaders::resolve_full_64bpp_scaled_cs,
sizeof(shaders::resolve_full_64bpp_scaled_cs)},
{shaders::resolve_full_128bpp_cs,
sizeof(shaders::resolve_full_128bpp_cs),
shaders::resolve_full_128bpp_scaled_cs,
sizeof(shaders::resolve_full_128bpp_scaled_cs)},
};
const VulkanRenderTargetCache::TransferPipelineLayoutInfo
VulkanRenderTargetCache::kTransferPipelineLayoutInfos[size_t(
TransferPipelineLayoutIndex::kCount)] = {
// kColor
{kTransferUsedDescriptorSetColorTextureBit,
kTransferUsedPushConstantDwordAddressBit},
// kDepth
{kTransferUsedDescriptorSetDepthStencilTexturesBit,
kTransferUsedPushConstantDwordAddressBit},
// kColorToStencilBit
{kTransferUsedDescriptorSetColorTextureBit,
kTransferUsedPushConstantDwordAddressBit |
kTransferUsedPushConstantDwordStencilMaskBit},
// kDepthToStencilBit
{kTransferUsedDescriptorSetDepthStencilTexturesBit,
kTransferUsedPushConstantDwordAddressBit |
kTransferUsedPushConstantDwordStencilMaskBit},
// kColorAndHostDepthTexture
{kTransferUsedDescriptorSetHostDepthStencilTexturesBit |
kTransferUsedDescriptorSetColorTextureBit,
kTransferUsedPushConstantDwordHostDepthAddressBit |
kTransferUsedPushConstantDwordAddressBit},
// kColorAndHostDepthBuffer
{kTransferUsedDescriptorSetHostDepthBufferBit |
kTransferUsedDescriptorSetColorTextureBit,
kTransferUsedPushConstantDwordHostDepthAddressBit |
kTransferUsedPushConstantDwordAddressBit},
// kDepthAndHostDepthTexture
{kTransferUsedDescriptorSetHostDepthStencilTexturesBit |
kTransferUsedDescriptorSetDepthStencilTexturesBit,
kTransferUsedPushConstantDwordHostDepthAddressBit |
kTransferUsedPushConstantDwordAddressBit},
// kDepthAndHostDepthBuffer
{kTransferUsedDescriptorSetHostDepthBufferBit |
kTransferUsedDescriptorSetDepthStencilTexturesBit,
kTransferUsedPushConstantDwordHostDepthAddressBit |
kTransferUsedPushConstantDwordAddressBit},
};
const VulkanRenderTargetCache::TransferModeInfo
VulkanRenderTargetCache::kTransferModes[size_t(TransferMode::kCount)] = {
// kColorToDepth
{TransferOutput::kDepth, TransferPipelineLayoutIndex::kColor},
// kColorToColor
{TransferOutput::kColor, TransferPipelineLayoutIndex::kColor},
// kDepthToDepth
{TransferOutput::kDepth, TransferPipelineLayoutIndex::kDepth},
// kDepthToColor
{TransferOutput::kColor, TransferPipelineLayoutIndex::kDepth},
// kColorToStencilBit
{TransferOutput::kStencilBit,
TransferPipelineLayoutIndex::kColorToStencilBit},
// kDepthToStencilBit
{TransferOutput::kStencilBit,
TransferPipelineLayoutIndex::kDepthToStencilBit},
// kColorAndHostDepthToDepth
{TransferOutput::kDepth,
TransferPipelineLayoutIndex::kColorAndHostDepthTexture},
// kDepthAndHostDepthToDepth
{TransferOutput::kDepth,
TransferPipelineLayoutIndex::kDepthAndHostDepthTexture},
// kColorAndHostDepthCopyToDepth
{TransferOutput::kDepth,
TransferPipelineLayoutIndex::kColorAndHostDepthBuffer},
// kDepthAndHostDepthCopyToDepth
{TransferOutput::kDepth,
TransferPipelineLayoutIndex::kDepthAndHostDepthBuffer},
};
VulkanRenderTargetCache::VulkanRenderTargetCache(
const RegisterFile& register_file, const Memory& memory,
TraceWriter& trace_writer, uint32_t draw_resolution_scale_x,
uint32_t draw_resolution_scale_y, VulkanCommandProcessor& command_processor)
: RenderTargetCache(register_file, memory, &trace_writer,
draw_resolution_scale_x, draw_resolution_scale_y),
command_processor_(command_processor),
trace_writer_(trace_writer) {}
VulkanRenderTargetCache::~VulkanRenderTargetCache() { Shutdown(true); }
bool VulkanRenderTargetCache::Initialize(uint32_t shared_memory_binding_count) {
const ui::vulkan::VulkanProvider& provider =
command_processor_.GetVulkanProvider();
const ui::vulkan::VulkanProvider::InstanceFunctions& ifn = provider.ifn();
VkPhysicalDevice physical_device = provider.physical_device();
const ui::vulkan::VulkanProvider::DeviceFunctions& dfn = provider.dfn();
VkDevice device = provider.device();
const VkPhysicalDeviceLimits& device_limits =
provider.device_properties().limits;
if (cvars::render_target_path_vulkan == "fsi") {
path_ = Path::kPixelShaderInterlock;
} else {
path_ = Path::kHostRenderTargets;
}
// Fragment shader interlock is a feature implemented by pretty advanced GPUs,
// closer to Direct3D 11 / OpenGL ES 3.2 level mainly, not Direct3D 10 /
// OpenGL ES 3.1. Thus, it's fine to demand a wide range of other optional
// features for the fragment shader interlock backend to work.
if (path_ == Path::kPixelShaderInterlock) {
const VkPhysicalDeviceFragmentShaderInterlockFeaturesEXT&
device_fragment_shader_interlock_features =
provider.device_fragment_shader_interlock_features();
const VkPhysicalDeviceFeatures& device_features =
provider.device_features();
// Interlocking between fragments with common sample coverage is enough, but
// interlocking more is acceptable too (fragmentShaderShadingRateInterlock
// would be okay too, but it's unlikely that an implementation would
// advertise only it and not any other ones, as it's a very specific feature
// interacting with another optional feature that is variable shading rate,
// so there's no need to overcomplicate the checks and the shader execution
// mode setting).
// Sample-rate shading is required by certain SPIR-V revisions to access the
// sample mask fragment shader input.
// Stanard sample locations are needed for calculating the depth at the
// samples.
// It's unlikely that a device exposing fragment shader interlock won't have
// a large enough storage buffer range and a sufficient SSBO slot count for
// all the shared memory buffers and the EDRAM buffer - an in a conflict
// between, for instance, the ability to vfetch and memexport in fragment
// shaders, and the usage of fragment shader interlock, prefer the former
// for simplicity.
if (!provider.device_extensions().ext_fragment_shader_interlock ||
!(device_fragment_shader_interlock_features
.fragmentShaderSampleInterlock ||
device_fragment_shader_interlock_features
.fragmentShaderPixelInterlock) ||
!device_features.fragmentStoresAndAtomics ||
!device_features.sampleRateShading ||
!device_limits.standardSampleLocations ||
shared_memory_binding_count >=
device_limits.maxDescriptorSetStorageBuffers) {
path_ = Path::kHostRenderTargets;
}
}
// Format support.
constexpr VkFormatFeatureFlags kUsedDepthFormatFeatures =
VK_FORMAT_FEATURE_SAMPLED_IMAGE_BIT |
VK_FORMAT_FEATURE_DEPTH_STENCIL_ATTACHMENT_BIT;
VkFormatProperties depth_unorm24_properties;
ifn.vkGetPhysicalDeviceFormatProperties(
physical_device, VK_FORMAT_D24_UNORM_S8_UINT, &depth_unorm24_properties);
depth_unorm24_vulkan_format_supported_ =
(depth_unorm24_properties.optimalTilingFeatures &
kUsedDepthFormatFeatures) == kUsedDepthFormatFeatures;
// 2x MSAA support.
// TODO(Triang3l): Handle sampledImageIntegerSampleCounts 4 not supported in
// transfers.
if (cvars::native_2x_msaa) {
// Multisampled integer sampled images are optional in Vulkan and in Xenia.
msaa_2x_attachments_supported_ =
(device_limits.framebufferColorSampleCounts &
device_limits.framebufferDepthSampleCounts &
device_limits.framebufferStencilSampleCounts &
device_limits.sampledImageColorSampleCounts &
device_limits.sampledImageDepthSampleCounts &
device_limits.sampledImageStencilSampleCounts &
VK_SAMPLE_COUNT_2_BIT) &&
(device_limits.sampledImageIntegerSampleCounts &
(VK_SAMPLE_COUNT_2_BIT | VK_SAMPLE_COUNT_4_BIT)) !=
VK_SAMPLE_COUNT_4_BIT;
msaa_2x_no_attachments_supported_ =
(device_limits.framebufferNoAttachmentsSampleCounts &
VK_SAMPLE_COUNT_2_BIT) != 0;
} else {
msaa_2x_attachments_supported_ = false;
msaa_2x_no_attachments_supported_ = false;
}
// Descriptor set layouts.
VkDescriptorSetLayoutBinding descriptor_set_layout_bindings[2];
descriptor_set_layout_bindings[0].binding = 0;
descriptor_set_layout_bindings[0].descriptorType =
VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
descriptor_set_layout_bindings[0].descriptorCount = 1;
descriptor_set_layout_bindings[0].stageFlags =
VK_SHADER_STAGE_FRAGMENT_BIT | VK_SHADER_STAGE_COMPUTE_BIT;
descriptor_set_layout_bindings[0].pImmutableSamplers = nullptr;
VkDescriptorSetLayoutCreateInfo descriptor_set_layout_create_info;
descriptor_set_layout_create_info.sType =
VK_STRUCTURE_TYPE_DESCRIPTOR_SET_LAYOUT_CREATE_INFO;
descriptor_set_layout_create_info.pNext = nullptr;
descriptor_set_layout_create_info.flags = 0;
descriptor_set_layout_create_info.bindingCount = 1;
descriptor_set_layout_create_info.pBindings = descriptor_set_layout_bindings;
if (dfn.vkCreateDescriptorSetLayout(
device, &descriptor_set_layout_create_info, nullptr,
&descriptor_set_layout_storage_buffer_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the descriptor set layout "
"with one storage buffer");
Shutdown();
return false;
}
descriptor_set_layout_bindings[0].descriptorType =
VK_DESCRIPTOR_TYPE_SAMPLED_IMAGE;
if (dfn.vkCreateDescriptorSetLayout(
device, &descriptor_set_layout_create_info, nullptr,
&descriptor_set_layout_sampled_image_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the descriptor set layout "
"with one sampled image");
Shutdown();
return false;
}
descriptor_set_layout_bindings[1].binding = 1;
descriptor_set_layout_bindings[1].descriptorType =
VK_DESCRIPTOR_TYPE_SAMPLED_IMAGE;
descriptor_set_layout_bindings[1].descriptorCount = 1;
descriptor_set_layout_bindings[1].stageFlags =
descriptor_set_layout_bindings[0].stageFlags;
descriptor_set_layout_bindings[1].pImmutableSamplers = nullptr;
descriptor_set_layout_create_info.bindingCount = 2;
if (dfn.vkCreateDescriptorSetLayout(
device, &descriptor_set_layout_create_info, nullptr,
&descriptor_set_layout_sampled_image_x2_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the descriptor set layout "
"with two sampled images");
Shutdown();
return false;
}
// Descriptor set pools.
// The pool sizes were chosen without a specific reason.
VkDescriptorPoolSize descriptor_set_layout_size;
descriptor_set_layout_size.type = VK_DESCRIPTOR_TYPE_SAMPLED_IMAGE;
descriptor_set_layout_size.descriptorCount = 1;
descriptor_set_pool_sampled_image_ =
std::make_unique<ui::vulkan::SingleLayoutDescriptorSetPool>(
provider, 256, 1, &descriptor_set_layout_size,
descriptor_set_layout_sampled_image_);
descriptor_set_layout_size.descriptorCount = 2;
descriptor_set_pool_sampled_image_x2_ =
std::make_unique<ui::vulkan::SingleLayoutDescriptorSetPool>(
provider, 256, 1, &descriptor_set_layout_size,
descriptor_set_layout_sampled_image_x2_);
// EDRAM contents reinterpretation buffer.
// 90 MB with 9x resolution scaling - within the minimum
// maxStorageBufferRange.
if (!ui::vulkan::util::CreateDedicatedAllocationBuffer(
provider,
VkDeviceSize(xenos::kEdramSizeBytes *
(draw_resolution_scale_x() * draw_resolution_scale_y())),
VK_BUFFER_USAGE_TRANSFER_SRC_BIT | VK_BUFFER_USAGE_TRANSFER_DST_BIT |
VK_BUFFER_USAGE_STORAGE_BUFFER_BIT,
ui::vulkan::util::MemoryPurpose::kDeviceLocal, edram_buffer_,
edram_buffer_memory_)) {
XELOGE("VulkanRenderTargetCache: Failed to create the EDRAM buffer");
Shutdown();
return false;
}
if (GetPath() == Path::kPixelShaderInterlock) {
// The first operation will likely be drawing.
edram_buffer_usage_ = EdramBufferUsage::kFragmentReadWrite;
} else {
// The first operation will likely be depth self-comparison.
edram_buffer_usage_ = EdramBufferUsage::kFragmentRead;
}
edram_buffer_modification_status_ =
EdramBufferModificationStatus::kUnmodified;
VkDescriptorPoolSize edram_storage_buffer_descriptor_pool_size;
edram_storage_buffer_descriptor_pool_size.type =
VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
edram_storage_buffer_descriptor_pool_size.descriptorCount = 1;
VkDescriptorPoolCreateInfo edram_storage_buffer_descriptor_pool_create_info;
edram_storage_buffer_descriptor_pool_create_info.sType =
VK_STRUCTURE_TYPE_DESCRIPTOR_POOL_CREATE_INFO;
edram_storage_buffer_descriptor_pool_create_info.pNext = nullptr;
edram_storage_buffer_descriptor_pool_create_info.flags = 0;
edram_storage_buffer_descriptor_pool_create_info.maxSets = 1;
edram_storage_buffer_descriptor_pool_create_info.poolSizeCount = 1;
edram_storage_buffer_descriptor_pool_create_info.pPoolSizes =
&edram_storage_buffer_descriptor_pool_size;
if (dfn.vkCreateDescriptorPool(
device, &edram_storage_buffer_descriptor_pool_create_info, nullptr,
&edram_storage_buffer_descriptor_pool_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the EDRAM buffer storage "
"buffer descriptor pool");
Shutdown();
return false;
}
VkDescriptorSetAllocateInfo edram_storage_buffer_descriptor_set_allocate_info;
edram_storage_buffer_descriptor_set_allocate_info.sType =
VK_STRUCTURE_TYPE_DESCRIPTOR_SET_ALLOCATE_INFO;
edram_storage_buffer_descriptor_set_allocate_info.pNext = nullptr;
edram_storage_buffer_descriptor_set_allocate_info.descriptorPool =
edram_storage_buffer_descriptor_pool_;
edram_storage_buffer_descriptor_set_allocate_info.descriptorSetCount = 1;
edram_storage_buffer_descriptor_set_allocate_info.pSetLayouts =
&descriptor_set_layout_storage_buffer_;
if (dfn.vkAllocateDescriptorSets(
device, &edram_storage_buffer_descriptor_set_allocate_info,
&edram_storage_buffer_descriptor_set_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to allocate the EDRAM buffer storage "
"buffer descriptor set");
Shutdown();
return false;
}
VkDescriptorBufferInfo edram_storage_buffer_descriptor_buffer_info;
edram_storage_buffer_descriptor_buffer_info.buffer = edram_buffer_;
edram_storage_buffer_descriptor_buffer_info.offset = 0;
edram_storage_buffer_descriptor_buffer_info.range = VK_WHOLE_SIZE;
VkWriteDescriptorSet edram_storage_buffer_descriptor_write;
edram_storage_buffer_descriptor_write.sType =
VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
edram_storage_buffer_descriptor_write.pNext = nullptr;
edram_storage_buffer_descriptor_write.dstSet =
edram_storage_buffer_descriptor_set_;
edram_storage_buffer_descriptor_write.dstBinding = 0;
edram_storage_buffer_descriptor_write.dstArrayElement = 0;
edram_storage_buffer_descriptor_write.descriptorCount = 1;
edram_storage_buffer_descriptor_write.descriptorType =
VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
edram_storage_buffer_descriptor_write.pImageInfo = nullptr;
edram_storage_buffer_descriptor_write.pBufferInfo =
&edram_storage_buffer_descriptor_buffer_info;
edram_storage_buffer_descriptor_write.pTexelBufferView = nullptr;
dfn.vkUpdateDescriptorSets(device, 1, &edram_storage_buffer_descriptor_write,
0, nullptr);
bool draw_resolution_scaled = IsDrawResolutionScaled();
// Resolve copy pipeline layout.
VkDescriptorSetLayout
resolve_copy_descriptor_set_layouts[kResolveCopyDescriptorSetCount] = {};
resolve_copy_descriptor_set_layouts[kResolveCopyDescriptorSetEdram] =
descriptor_set_layout_storage_buffer_;
resolve_copy_descriptor_set_layouts[kResolveCopyDescriptorSetDest] =
command_processor_.GetSingleTransientDescriptorLayout(
VulkanCommandProcessor::SingleTransientDescriptorLayout ::
kStorageBufferCompute);
VkPushConstantRange resolve_copy_push_constant_range;
resolve_copy_push_constant_range.stageFlags = VK_SHADER_STAGE_COMPUTE_BIT;
resolve_copy_push_constant_range.offset = 0;
// Potentially binding all of the shared memory at 1x resolution, but only
// portions with scaled resolution.
resolve_copy_push_constant_range.size =
draw_resolution_scaled
? sizeof(draw_util::ResolveCopyShaderConstants::DestRelative)
: sizeof(draw_util::ResolveCopyShaderConstants);
VkPipelineLayoutCreateInfo resolve_copy_pipeline_layout_create_info;
resolve_copy_pipeline_layout_create_info.sType =
VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO;
resolve_copy_pipeline_layout_create_info.pNext = nullptr;
resolve_copy_pipeline_layout_create_info.flags = 0;
resolve_copy_pipeline_layout_create_info.setLayoutCount =
kResolveCopyDescriptorSetCount;
resolve_copy_pipeline_layout_create_info.pSetLayouts =
resolve_copy_descriptor_set_layouts;
resolve_copy_pipeline_layout_create_info.pushConstantRangeCount = 1;
resolve_copy_pipeline_layout_create_info.pPushConstantRanges =
&resolve_copy_push_constant_range;
if (dfn.vkCreatePipelineLayout(
device, &resolve_copy_pipeline_layout_create_info, nullptr,
&resolve_copy_pipeline_layout_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the resolve copy pipeline "
"layout");
Shutdown();
return false;
}
// Resolve copy pipelines.
for (size_t i = 0; i < size_t(draw_util::ResolveCopyShaderIndex::kCount);
++i) {
const draw_util::ResolveCopyShaderInfo& resolve_copy_shader_info =
draw_util::resolve_copy_shader_info[i];
const ResolveCopyShaderCode& resolve_copy_shader_code =
kResolveCopyShaders[i];
// Somewhat verification whether resolve_copy_shaders_ is up to date.
assert_true(resolve_copy_shader_code.unscaled &&
resolve_copy_shader_code.unscaled_size_bytes &&
resolve_copy_shader_code.scaled &&
resolve_copy_shader_code.scaled_size_bytes);
VkPipeline resolve_copy_pipeline = ui::vulkan::util::CreateComputePipeline(
provider, resolve_copy_pipeline_layout_,
draw_resolution_scaled ? resolve_copy_shader_code.scaled
: resolve_copy_shader_code.unscaled,
draw_resolution_scaled ? resolve_copy_shader_code.scaled_size_bytes
: resolve_copy_shader_code.unscaled_size_bytes);
if (resolve_copy_pipeline == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the resolve copy "
"pipeline {}",
resolve_copy_shader_info.debug_name);
Shutdown();
return false;
}
provider.SetDeviceObjectName(VK_OBJECT_TYPE_PIPELINE, resolve_copy_pipeline,
resolve_copy_shader_info.debug_name);
resolve_copy_pipelines_[i] = resolve_copy_pipeline;
}
// TODO(Triang3l): All paths (FSI).
if (path_ == Path::kHostRenderTargets) {
// Host render targets.
depth_float24_round_ = cvars::depth_float24_round;
// Host depth storing pipeline layout.
VkDescriptorSetLayout host_depth_store_descriptor_set_layouts[] = {
// Destination EDRAM storage buffer.
descriptor_set_layout_storage_buffer_,
// Source depth / stencil texture (only depth is used).
descriptor_set_layout_sampled_image_x2_,
};
VkPushConstantRange host_depth_store_push_constant_range;
host_depth_store_push_constant_range.stageFlags =
VK_SHADER_STAGE_COMPUTE_BIT;
host_depth_store_push_constant_range.offset = 0;
host_depth_store_push_constant_range.size = sizeof(HostDepthStoreConstants);
VkPipelineLayoutCreateInfo host_depth_store_pipeline_layout_create_info;
host_depth_store_pipeline_layout_create_info.sType =
VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO;
host_depth_store_pipeline_layout_create_info.pNext = nullptr;
host_depth_store_pipeline_layout_create_info.flags = 0;
host_depth_store_pipeline_layout_create_info.setLayoutCount =
uint32_t(xe::countof(host_depth_store_descriptor_set_layouts));
host_depth_store_pipeline_layout_create_info.pSetLayouts =
host_depth_store_descriptor_set_layouts;
host_depth_store_pipeline_layout_create_info.pushConstantRangeCount = 1;
host_depth_store_pipeline_layout_create_info.pPushConstantRanges =
&host_depth_store_push_constant_range;
if (dfn.vkCreatePipelineLayout(
device, &host_depth_store_pipeline_layout_create_info, nullptr,
&host_depth_store_pipeline_layout_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the host depth storing "
"pipeline layout");
Shutdown();
return false;
}
const std::pair<const uint32_t*, size_t> host_depth_store_shaders[] = {
{shaders::host_depth_store_1xmsaa_cs,
sizeof(shaders::host_depth_store_1xmsaa_cs)},
{shaders::host_depth_store_2xmsaa_cs,
sizeof(shaders::host_depth_store_2xmsaa_cs)},
{shaders::host_depth_store_4xmsaa_cs,
sizeof(shaders::host_depth_store_4xmsaa_cs)},
};
for (size_t i = 0; i < xe::countof(host_depth_store_shaders); ++i) {
const std::pair<const uint32_t*, size_t> host_depth_store_shader =
host_depth_store_shaders[i];
VkPipeline host_depth_store_pipeline =
ui::vulkan::util::CreateComputePipeline(
provider, host_depth_store_pipeline_layout_,
host_depth_store_shader.first, host_depth_store_shader.second);
if (host_depth_store_pipeline == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the {}-sample host "
"depth storing pipeline",
uint32_t(1) << i);
Shutdown();
return false;
}
host_depth_store_pipelines_[i] = host_depth_store_pipeline;
}
// Transfer and clear vertex buffer, for quads of up to tile granularity.
transfer_vertex_buffer_pool_ =
std::make_unique<ui::vulkan::VulkanUploadBufferPool>(
provider, VK_BUFFER_USAGE_VERTEX_BUFFER_BIT,
std::max(ui::vulkan::VulkanUploadBufferPool::kDefaultPageSize,
sizeof(float) * 2 * 6 *
Transfer::kMaxCutoutBorderRectangles *
xenos::kEdramTileCount));
// Transfer vertex shader.
transfer_passthrough_vertex_shader_ = ui::vulkan::util::CreateShaderModule(
provider, shaders::passthrough_position_xy_vs,
sizeof(shaders::passthrough_position_xy_vs));
if (transfer_passthrough_vertex_shader_ == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the render target "
"ownership transfer vertex shader");
Shutdown();
return false;
}
// Transfer pipeline layouts.
VkDescriptorSetLayout transfer_pipeline_layout_descriptor_set_layouts
[kTransferUsedDescriptorSetCount];
VkPushConstantRange transfer_pipeline_layout_push_constant_range;
transfer_pipeline_layout_push_constant_range.stageFlags =
VK_SHADER_STAGE_FRAGMENT_BIT;
transfer_pipeline_layout_push_constant_range.offset = 0;
VkPipelineLayoutCreateInfo transfer_pipeline_layout_create_info;
transfer_pipeline_layout_create_info.sType =
VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO;
transfer_pipeline_layout_create_info.pNext = nullptr;
transfer_pipeline_layout_create_info.flags = 0;
transfer_pipeline_layout_create_info.pSetLayouts =
transfer_pipeline_layout_descriptor_set_layouts;
transfer_pipeline_layout_create_info.pPushConstantRanges =
&transfer_pipeline_layout_push_constant_range;
for (size_t i = 0; i < size_t(TransferPipelineLayoutIndex::kCount); ++i) {
const TransferPipelineLayoutInfo& transfer_pipeline_layout_info =
kTransferPipelineLayoutInfos[i];
transfer_pipeline_layout_create_info.setLayoutCount = 0;
uint32_t transfer_pipeline_layout_descriptor_sets_remaining =
transfer_pipeline_layout_info.used_descriptor_sets;
uint32_t transfer_pipeline_layout_descriptor_set_index;
while (xe::bit_scan_forward(
transfer_pipeline_layout_descriptor_sets_remaining,
&transfer_pipeline_layout_descriptor_set_index)) {
transfer_pipeline_layout_descriptor_sets_remaining &=
~(uint32_t(1) << transfer_pipeline_layout_descriptor_set_index);
VkDescriptorSetLayout transfer_pipeline_layout_descriptor_set_layout =
VK_NULL_HANDLE;
switch (TransferUsedDescriptorSet(
transfer_pipeline_layout_descriptor_set_index)) {
case kTransferUsedDescriptorSetHostDepthBuffer:
transfer_pipeline_layout_descriptor_set_layout =
descriptor_set_layout_storage_buffer_;
break;
case kTransferUsedDescriptorSetHostDepthStencilTextures:
case kTransferUsedDescriptorSetDepthStencilTextures:
transfer_pipeline_layout_descriptor_set_layout =
descriptor_set_layout_sampled_image_x2_;
break;
case kTransferUsedDescriptorSetColorTexture:
transfer_pipeline_layout_descriptor_set_layout =
descriptor_set_layout_sampled_image_;
break;
default:
assert_unhandled_case(TransferUsedDescriptorSet(
transfer_pipeline_layout_descriptor_set_index));
}
transfer_pipeline_layout_descriptor_set_layouts
[transfer_pipeline_layout_create_info.setLayoutCount++] =
transfer_pipeline_layout_descriptor_set_layout;
}
transfer_pipeline_layout_push_constant_range.size = uint32_t(
sizeof(uint32_t) *
xe::bit_count(
transfer_pipeline_layout_info.used_push_constant_dwords));
transfer_pipeline_layout_create_info.pushConstantRangeCount =
transfer_pipeline_layout_info.used_push_constant_dwords ? 1 : 0;
if (dfn.vkCreatePipelineLayout(
device, &transfer_pipeline_layout_create_info, nullptr,
&transfer_pipeline_layouts_[i]) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the render target "
"ownership transfer pipeline layout {}",
i);
Shutdown();
return false;
}
}
// Dump pipeline layouts.
VkDescriptorSetLayout
dump_pipeline_layout_descriptor_set_layouts[kDumpDescriptorSetCount];
dump_pipeline_layout_descriptor_set_layouts[kDumpDescriptorSetEdram] =
descriptor_set_layout_storage_buffer_;
dump_pipeline_layout_descriptor_set_layouts[kDumpDescriptorSetSource] =
descriptor_set_layout_sampled_image_;
VkPushConstantRange dump_pipeline_layout_push_constant_range;
dump_pipeline_layout_push_constant_range.stageFlags =
VK_SHADER_STAGE_COMPUTE_BIT;
dump_pipeline_layout_push_constant_range.offset = 0;
dump_pipeline_layout_push_constant_range.size =
sizeof(uint32_t) * kDumpPushConstantCount;
VkPipelineLayoutCreateInfo dump_pipeline_layout_create_info;
dump_pipeline_layout_create_info.sType =
VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO;
dump_pipeline_layout_create_info.pNext = nullptr;
dump_pipeline_layout_create_info.flags = 0;
dump_pipeline_layout_create_info.setLayoutCount =
uint32_t(xe::countof(dump_pipeline_layout_descriptor_set_layouts));
dump_pipeline_layout_create_info.pSetLayouts =
dump_pipeline_layout_descriptor_set_layouts;
dump_pipeline_layout_create_info.pushConstantRangeCount = 1;
dump_pipeline_layout_create_info.pPushConstantRanges =
&dump_pipeline_layout_push_constant_range;
if (dfn.vkCreatePipelineLayout(device, &dump_pipeline_layout_create_info,
nullptr, &dump_pipeline_layout_color_) !=
VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the color render target "
"dumping pipeline layout");
Shutdown();
return false;
}
dump_pipeline_layout_descriptor_set_layouts[kDumpDescriptorSetSource] =
descriptor_set_layout_sampled_image_x2_;
if (dfn.vkCreatePipelineLayout(device, &dump_pipeline_layout_create_info,
nullptr, &dump_pipeline_layout_depth_) !=
VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the depth render target "
"dumping pipeline layout");
Shutdown();
return false;
}
} else if (path_ == Path::kPixelShaderInterlock) {
// Pixel (fragment) shader interlock.
// Blending is done in linear space directly in shaders.
gamma_render_target_as_srgb_ = false;
// Always true float24 depth rounded to the nearest even.
depth_float24_round_ = true;
// The pipeline layout and the pipelines for clearing the EDRAM buffer in
// resolves.
VkPushConstantRange resolve_fsi_clear_push_constant_range;
resolve_fsi_clear_push_constant_range.stageFlags =
VK_SHADER_STAGE_COMPUTE_BIT;
resolve_fsi_clear_push_constant_range.offset = 0;
resolve_fsi_clear_push_constant_range.size =
sizeof(draw_util::ResolveClearShaderConstants);
VkPipelineLayoutCreateInfo resolve_fsi_clear_pipeline_layout_create_info;
resolve_fsi_clear_pipeline_layout_create_info.sType =
VK_STRUCTURE_TYPE_PIPELINE_LAYOUT_CREATE_INFO;
resolve_fsi_clear_pipeline_layout_create_info.pNext = nullptr;
resolve_fsi_clear_pipeline_layout_create_info.flags = 0;
resolve_fsi_clear_pipeline_layout_create_info.setLayoutCount = 1;
resolve_fsi_clear_pipeline_layout_create_info.pSetLayouts =
&descriptor_set_layout_storage_buffer_;
resolve_fsi_clear_pipeline_layout_create_info.pushConstantRangeCount = 1;
resolve_fsi_clear_pipeline_layout_create_info.pPushConstantRanges =
&resolve_fsi_clear_push_constant_range;
if (dfn.vkCreatePipelineLayout(
device, &resolve_fsi_clear_pipeline_layout_create_info, nullptr,
&resolve_fsi_clear_pipeline_layout_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the resolve EDRAM buffer "
"clear pipeline layout");
Shutdown();
return false;
}
resolve_fsi_clear_32bpp_pipeline_ = ui::vulkan::util::CreateComputePipeline(
provider, resolve_fsi_clear_pipeline_layout_,
draw_resolution_scaled ? shaders::resolve_clear_32bpp_scaled_cs
: shaders::resolve_clear_32bpp_cs,
draw_resolution_scaled ? sizeof(shaders::resolve_clear_32bpp_scaled_cs)
: sizeof(shaders::resolve_clear_32bpp_cs));
if (resolve_fsi_clear_32bpp_pipeline_ == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the 32bpp resolve EDRAM "
"buffer clear pipeline");
Shutdown();
return false;
}
resolve_fsi_clear_64bpp_pipeline_ = ui::vulkan::util::CreateComputePipeline(
provider, resolve_fsi_clear_pipeline_layout_,
draw_resolution_scaled ? shaders::resolve_clear_64bpp_scaled_cs
: shaders::resolve_clear_64bpp_cs,
draw_resolution_scaled ? sizeof(shaders::resolve_clear_64bpp_scaled_cs)
: sizeof(shaders::resolve_clear_64bpp_cs));
if (resolve_fsi_clear_32bpp_pipeline_ == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the 64bpp resolve EDRAM "
"buffer clear pipeline");
Shutdown();
return false;
}
// Common render pass.
VkSubpassDescription fsi_subpass = {};
fsi_subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS;
// Fragment shader interlock provides synchronization and ordering within a
// subpass, create an external by-region dependency to maintain interlocking
// between passes. Framebuffer-global dependencies will be made with
// explicit barriers when the addressing of the EDRAM buffer relatively to
// the fragment coordinates is changed.
VkSubpassDependency fsi_subpass_dependencies[2];
fsi_subpass_dependencies[0].srcSubpass = VK_SUBPASS_EXTERNAL;
fsi_subpass_dependencies[0].dstSubpass = 0;
fsi_subpass_dependencies[0].srcStageMask =
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT;
fsi_subpass_dependencies[0].dstStageMask =
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT;
fsi_subpass_dependencies[0].srcAccessMask =
VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT;
fsi_subpass_dependencies[0].dstAccessMask =
VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT;
fsi_subpass_dependencies[0].dependencyFlags = VK_DEPENDENCY_BY_REGION_BIT;
fsi_subpass_dependencies[1] = fsi_subpass_dependencies[0];
std::swap(fsi_subpass_dependencies[1].srcSubpass,
fsi_subpass_dependencies[1].dstSubpass);
VkRenderPassCreateInfo fsi_render_pass_create_info;
fsi_render_pass_create_info.sType =
VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO;
fsi_render_pass_create_info.pNext = nullptr;
fsi_render_pass_create_info.flags = 0;
fsi_render_pass_create_info.attachmentCount = 0;
fsi_render_pass_create_info.pAttachments = nullptr;
fsi_render_pass_create_info.subpassCount = 1;
fsi_render_pass_create_info.pSubpasses = &fsi_subpass;
fsi_render_pass_create_info.dependencyCount =
uint32_t(xe::countof(fsi_subpass_dependencies));
fsi_render_pass_create_info.pDependencies = fsi_subpass_dependencies;
if (dfn.vkCreateRenderPass(device, &fsi_render_pass_create_info, nullptr,
&fsi_render_pass_) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the fragment shader "
"interlock render backend render pass");
Shutdown();
return false;
}
// Common framebuffer.
VkFramebufferCreateInfo fsi_framebuffer_create_info;
fsi_framebuffer_create_info.sType =
VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO;
fsi_framebuffer_create_info.pNext = nullptr;
fsi_framebuffer_create_info.flags = 0;
fsi_framebuffer_create_info.renderPass = fsi_render_pass_;
fsi_framebuffer_create_info.attachmentCount = 0;
fsi_framebuffer_create_info.pAttachments = nullptr;
fsi_framebuffer_create_info.width = std::min(
xenos::kTexture2DCubeMaxWidthHeight * draw_resolution_scale_x(),
device_limits.maxFramebufferWidth);
fsi_framebuffer_create_info.height = std::min(
xenos::kTexture2DCubeMaxWidthHeight * draw_resolution_scale_y(),
device_limits.maxFramebufferHeight);
fsi_framebuffer_create_info.layers = 1;
if (dfn.vkCreateFramebuffer(device, &fsi_framebuffer_create_info, nullptr,
&fsi_framebuffer_.framebuffer) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the fragment shader "
"interlock render backend framebuffer");
Shutdown();
return false;
}
fsi_framebuffer_.host_extent.width = fsi_framebuffer_create_info.width;
fsi_framebuffer_.host_extent.height = fsi_framebuffer_create_info.height;
} else {
assert_unhandled_case(path_);
Shutdown();
return false;
}
// Reset the last update structures, to keep the defaults consistent between
// paths regardless of whether the update for the path actually modifies them.
last_update_render_pass_key_ = RenderPassKey();
last_update_render_pass_ = VK_NULL_HANDLE;
last_update_framebuffer_pitch_tiles_at_32bpp_ = 0;
std::memset(last_update_framebuffer_attachments_, 0,
sizeof(last_update_framebuffer_attachments_));
last_update_framebuffer_ = VK_NULL_HANDLE;
InitializeCommon();
return true;
}
void VulkanRenderTargetCache::Shutdown(bool from_destructor) {
const ui::vulkan::VulkanProvider& provider =
command_processor_.GetVulkanProvider();
const ui::vulkan::VulkanProvider::DeviceFunctions& dfn = provider.dfn();
VkDevice device = provider.device();
// Destroy all render targets before the descriptor set pool is destroyed -
// may happen if shutting down the VulkanRenderTargetCache by destroying it,
// so ShutdownCommon is called by the RenderTargetCache destructor, when it's
// already too late.
DestroyAllRenderTargets(true);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipeline, device,
resolve_fsi_clear_64bpp_pipeline_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipeline, device,
resolve_fsi_clear_32bpp_pipeline_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipelineLayout, device,
resolve_fsi_clear_pipeline_layout_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyFramebuffer, device,
fsi_framebuffer_.framebuffer);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyRenderPass, device,
fsi_render_pass_);
for (const auto& dump_pipeline_pair : dump_pipelines_) {
// May be null to prevent recreation attempts.
if (dump_pipeline_pair.second != VK_NULL_HANDLE) {
dfn.vkDestroyPipeline(device, dump_pipeline_pair.second, nullptr);
}
}
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipelineLayout, device,
dump_pipeline_layout_depth_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipelineLayout, device,
dump_pipeline_layout_color_);
for (const auto& transfer_pipeline_array_pair : transfer_pipelines_) {
for (VkPipeline transfer_pipeline : transfer_pipeline_array_pair.second) {
// May be null to prevent recreation attempts.
if (transfer_pipeline != VK_NULL_HANDLE) {
dfn.vkDestroyPipeline(device, transfer_pipeline, nullptr);
}
}
}
transfer_pipelines_.clear();
for (const auto& transfer_shader_pair : transfer_shaders_) {
if (transfer_shader_pair.second != VK_NULL_HANDLE) {
dfn.vkDestroyShaderModule(device, transfer_shader_pair.second, nullptr);
}
}
transfer_shaders_.clear();
for (size_t i = 0; i < size_t(TransferPipelineLayoutIndex::kCount); ++i) {
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipelineLayout, device,
transfer_pipeline_layouts_[i]);
}
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyShaderModule, device,
transfer_passthrough_vertex_shader_);
transfer_vertex_buffer_pool_.reset();
for (size_t i = 0; i < xe::countof(host_depth_store_pipelines_); ++i) {
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipeline, device,
host_depth_store_pipelines_[i]);
}
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipelineLayout, device,
host_depth_store_pipeline_layout_);
last_update_framebuffer_ = VK_NULL_HANDLE;
for (const auto& framebuffer_pair : framebuffers_) {
dfn.vkDestroyFramebuffer(device, framebuffer_pair.second.framebuffer,
nullptr);
}
framebuffers_.clear();
last_update_render_pass_ = VK_NULL_HANDLE;
for (const auto& render_pass_pair : render_passes_) {
if (render_pass_pair.second != VK_NULL_HANDLE) {
dfn.vkDestroyRenderPass(device, render_pass_pair.second, nullptr);
}
}
render_passes_.clear();
for (VkPipeline& resolve_copy_pipeline : resolve_copy_pipelines_) {
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipeline, device,
resolve_copy_pipeline);
}
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyPipelineLayout, device,
resolve_copy_pipeline_layout_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyDescriptorPool, device,
edram_storage_buffer_descriptor_pool_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyBuffer, device,
edram_buffer_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkFreeMemory, device,
edram_buffer_memory_);
descriptor_set_pool_sampled_image_x2_.reset();
descriptor_set_pool_sampled_image_.reset();
ui::vulkan::util::DestroyAndNullHandle(
dfn.vkDestroyDescriptorSetLayout, device,
descriptor_set_layout_sampled_image_x2_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyDescriptorSetLayout,
device,
descriptor_set_layout_sampled_image_);
ui::vulkan::util::DestroyAndNullHandle(dfn.vkDestroyDescriptorSetLayout,
device,
descriptor_set_layout_storage_buffer_);
if (!from_destructor) {
ShutdownCommon();
}
}
void VulkanRenderTargetCache::ClearCache() {
const ui::vulkan::VulkanProvider& provider =
command_processor_.GetVulkanProvider();
const ui::vulkan::VulkanProvider::DeviceFunctions& dfn = provider.dfn();
VkDevice device = provider.device();
// Framebuffer objects must be destroyed because they reference views of
// attachment images, which may be removed by the common ClearCache.
last_update_framebuffer_ = VK_NULL_HANDLE;
for (const auto& framebuffer_pair : framebuffers_) {
dfn.vkDestroyFramebuffer(device, framebuffer_pair.second.framebuffer,
nullptr);
}
framebuffers_.clear();
last_update_render_pass_ = VK_NULL_HANDLE;
for (const auto& render_pass_pair : render_passes_) {
dfn.vkDestroyRenderPass(device, render_pass_pair.second, nullptr);
}
render_passes_.clear();
RenderTargetCache::ClearCache();
}
void VulkanRenderTargetCache::CompletedSubmissionUpdated() {
if (transfer_vertex_buffer_pool_) {
transfer_vertex_buffer_pool_->Reclaim(
command_processor_.GetCompletedSubmission());
}
}
void VulkanRenderTargetCache::EndSubmission() {
if (transfer_vertex_buffer_pool_) {
transfer_vertex_buffer_pool_->FlushWrites();
}
}
bool VulkanRenderTargetCache::Resolve(const Memory& memory,
VulkanSharedMemory& shared_memory,
VulkanTextureCache& texture_cache,
uint32_t& written_address_out,
uint32_t& written_length_out) {
written_address_out = 0;
written_length_out = 0;
bool draw_resolution_scaled = IsDrawResolutionScaled();
draw_util::ResolveInfo resolve_info;
if (!draw_util::GetResolveInfo(
register_file(), memory, trace_writer_, draw_resolution_scale_x(),
draw_resolution_scale_y(), IsFixedRG16TruncatedToMinus1To1(),
IsFixedRGBA16TruncatedToMinus1To1(), resolve_info)) {
return false;
}
// Nothing to copy/clear.
if (!resolve_info.coordinate_info.width_div_8 || !resolve_info.height_div_8) {
return true;
}
const ui::vulkan::VulkanProvider& provider =
command_processor_.GetVulkanProvider();
const ui::vulkan::VulkanProvider::DeviceFunctions& dfn = provider.dfn();
VkDevice device = provider.device();
DeferredCommandBuffer& command_buffer =
command_processor_.deferred_command_buffer();
// Copying.
bool copied = false;
if (resolve_info.copy_dest_extent_length) {
if (GetPath() == Path::kHostRenderTargets) {
// Dump the current contents of the render targets owning the affected
// range to edram_buffer_.
// TODO(Triang3l): Direct host render target -> shared memory resolve
// shaders for non-converting cases.
uint32_t dump_base;
uint32_t dump_row_length_used;
uint32_t dump_rows;
uint32_t dump_pitch;
resolve_info.GetCopyEdramTileSpan(dump_base, dump_row_length_used,
dump_rows, dump_pitch);
DumpRenderTargets(dump_base, dump_row_length_used, dump_rows, dump_pitch);
}
draw_util::ResolveCopyShaderConstants copy_shader_constants;
uint32_t copy_group_count_x, copy_group_count_y;
draw_util::ResolveCopyShaderIndex copy_shader = resolve_info.GetCopyShader(
draw_resolution_scale_x(), draw_resolution_scale_y(),
copy_shader_constants, copy_group_count_x, copy_group_count_y);
assert_true(copy_group_count_x && copy_group_count_y);
if (copy_shader != draw_util::ResolveCopyShaderIndex::kUnknown) {
const draw_util::ResolveCopyShaderInfo& copy_shader_info =
draw_util::resolve_copy_shader_info[size_t(copy_shader)];
// Make sure there is memory to write to.
bool copy_dest_committed;
// TODO(Triang3l): Resolution-scaled buffer committing.
copy_dest_committed =
shared_memory.RequestRange(resolve_info.copy_dest_extent_start,
resolve_info.copy_dest_extent_length);
if (!copy_dest_committed) {
XELOGE(
"VulkanRenderTargetCache: Failed to obtain the resolve destination "
"memory region");
} else {
// TODO(Triang3l): Switching between descriptors if exceeding
// maxStorageBufferRange.
// TODO(Triang3l): Use a single 512 MB shared memory binding if
// possible.
VkDescriptorSet descriptor_set_dest =
command_processor_.AllocateSingleTransientDescriptor(
VulkanCommandProcessor::SingleTransientDescriptorLayout ::
kStorageBufferCompute);
if (descriptor_set_dest != VK_NULL_HANDLE) {
// Write the destination descriptor.
// TODO(Triang3l): Scaled resolve buffer binding.
VkDescriptorBufferInfo write_descriptor_set_dest_buffer_info;
write_descriptor_set_dest_buffer_info.buffer = shared_memory.buffer();
write_descriptor_set_dest_buffer_info.offset =
resolve_info.copy_dest_base;
write_descriptor_set_dest_buffer_info.range =
resolve_info.copy_dest_extent_start -
resolve_info.copy_dest_base +
resolve_info.copy_dest_extent_length;
VkWriteDescriptorSet write_descriptor_set_dest;
write_descriptor_set_dest.sType =
VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
write_descriptor_set_dest.pNext = nullptr;
write_descriptor_set_dest.dstSet = descriptor_set_dest;
write_descriptor_set_dest.dstBinding = 0;
write_descriptor_set_dest.dstArrayElement = 0;
write_descriptor_set_dest.descriptorCount = 1;
write_descriptor_set_dest.descriptorType =
VK_DESCRIPTOR_TYPE_STORAGE_BUFFER;
write_descriptor_set_dest.pImageInfo = nullptr;
write_descriptor_set_dest.pBufferInfo =
&write_descriptor_set_dest_buffer_info;
write_descriptor_set_dest.pTexelBufferView = nullptr;
dfn.vkUpdateDescriptorSets(device, 1, &write_descriptor_set_dest, 0,
nullptr);
// Submit the resolve.
// TODO(Triang3l): Transition the scaled resolve buffer.
shared_memory.Use(VulkanSharedMemory::Usage::kComputeWrite,
std::pair<uint32_t, uint32_t>(
resolve_info.copy_dest_extent_start,
resolve_info.copy_dest_extent_length));
UseEdramBuffer(EdramBufferUsage::kComputeRead);
command_processor_.BindExternalComputePipeline(
resolve_copy_pipelines_[size_t(copy_shader)]);
VkDescriptorSet descriptor_sets[kResolveCopyDescriptorSetCount] = {};
descriptor_sets[kResolveCopyDescriptorSetEdram] =
edram_storage_buffer_descriptor_set_;
descriptor_sets[kResolveCopyDescriptorSetDest] = descriptor_set_dest;
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_COMPUTE, resolve_copy_pipeline_layout_, 0,
uint32_t(xe::countof(descriptor_sets)), descriptor_sets, 0,
nullptr);
if (draw_resolution_scaled) {
command_buffer.CmdVkPushConstants(
resolve_copy_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT, 0,
sizeof(copy_shader_constants.dest_relative),
&copy_shader_constants.dest_relative);
} else {
// TODO(Triang3l): Proper dest_base in case of one 512 MB shared
// memory binding, or multiple shared memory bindings in case of
// splitting due to maxStorageBufferRange overflow.
copy_shader_constants.dest_base -=
uint32_t(write_descriptor_set_dest_buffer_info.offset);
command_buffer.CmdVkPushConstants(
resolve_copy_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT, 0,
sizeof(copy_shader_constants), &copy_shader_constants);
}
command_processor_.SubmitBarriers(true);
command_buffer.CmdVkDispatch(copy_group_count_x, copy_group_count_y,
1);
// Invalidate textures and mark the range as scaled if needed.
texture_cache.MarkRangeAsResolved(
resolve_info.copy_dest_extent_start,
resolve_info.copy_dest_extent_length);
written_address_out = resolve_info.copy_dest_extent_start;
written_length_out = resolve_info.copy_dest_extent_length;
copied = true;
}
}
}
} else {
copied = true;
}
// Clearing.
bool cleared = false;
bool clear_depth = resolve_info.IsClearingDepth();
bool clear_color = resolve_info.IsClearingColor();
if (clear_depth || clear_color) {
switch (GetPath()) {
case Path::kHostRenderTargets: {
Transfer::Rectangle clear_rectangle;
RenderTarget* clear_render_targets[2];
// If PrepareHostRenderTargetsResolveClear returns false, may be just an
// empty region (success) or an error - don't care.
if (PrepareHostRenderTargetsResolveClear(
resolve_info, clear_rectangle, clear_render_targets[0],
clear_transfers_[0], clear_render_targets[1],
clear_transfers_[1])) {
uint64_t clear_values[2];
clear_values[0] = resolve_info.rb_depth_clear;
clear_values[1] = resolve_info.rb_color_clear |
(uint64_t(resolve_info.rb_color_clear_lo) << 32);
PerformTransfersAndResolveClears(2, clear_render_targets,
clear_transfers_, clear_values,
&clear_rectangle);
}
cleared = true;
} break;
case Path::kPixelShaderInterlock: {
UseEdramBuffer(EdramBufferUsage::kComputeWrite);
// Should be safe to only commit once (if was accessed as unordered or
// with fragment shader interlock previously - if there was nothing to
// copy, only to clear, for some reason, for instance), overlap of the
// depth and the color ranges is highly unlikely.
CommitEdramBufferShaderWrites();
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_COMPUTE, resolve_fsi_clear_pipeline_layout_,
0, 1, &edram_storage_buffer_descriptor_set_, 0, nullptr);
std::pair<uint32_t, uint32_t> clear_group_count =
resolve_info.GetClearShaderGroupCount(draw_resolution_scale_x(),
draw_resolution_scale_y());
assert_true(clear_group_count.first && clear_group_count.second);
if (clear_depth) {
command_processor_.BindExternalComputePipeline(
resolve_fsi_clear_32bpp_pipeline_);
draw_util::ResolveClearShaderConstants depth_clear_constants;
resolve_info.GetDepthClearShaderConstants(depth_clear_constants);
command_buffer.CmdVkPushConstants(
resolve_fsi_clear_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT,
0, sizeof(depth_clear_constants), &depth_clear_constants);
command_processor_.SubmitBarriers(true);
command_buffer.CmdVkDispatch(clear_group_count.first,
clear_group_count.second, 1);
}
if (clear_color) {
command_processor_.BindExternalComputePipeline(
resolve_info.color_edram_info.format_is_64bpp
? resolve_fsi_clear_64bpp_pipeline_
: resolve_fsi_clear_32bpp_pipeline_);
draw_util::ResolveClearShaderConstants color_clear_constants;
resolve_info.GetColorClearShaderConstants(color_clear_constants);
if (clear_depth) {
// Non-RT-specific constants have already been set.
command_buffer.CmdVkPushConstants(
resolve_fsi_clear_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT,
uint32_t(offsetof(draw_util::ResolveClearShaderConstants,
rt_specific)),
sizeof(color_clear_constants.rt_specific),
&color_clear_constants.rt_specific);
} else {
command_buffer.CmdVkPushConstants(
resolve_fsi_clear_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT,
0, sizeof(color_clear_constants), &color_clear_constants);
}
command_processor_.SubmitBarriers(true);
command_buffer.CmdVkDispatch(clear_group_count.first,
clear_group_count.second, 1);
}
MarkEdramBufferModified();
cleared = true;
} break;
default:
assert_unhandled_case(GetPath());
}
} else {
cleared = true;
}
return copied && cleared;
}
bool VulkanRenderTargetCache::Update(
bool is_rasterization_done, reg::RB_DEPTHCONTROL normalized_depth_control,
uint32_t normalized_color_mask, const Shader& vertex_shader) {
if (!RenderTargetCache::Update(is_rasterization_done,
normalized_depth_control,
normalized_color_mask, vertex_shader)) {
return false;
}
auto rb_surface_info = register_file().Get<reg::RB_SURFACE_INFO>();
RenderPassKey render_pass_key;
// Needed even with the fragment shader interlock render backend for passing
// the sample count to the pipeline cache.
render_pass_key.msaa_samples = rb_surface_info.msaa_samples;
switch (GetPath()) {
case Path::kHostRenderTargets: {
RenderTarget* const* depth_and_color_render_targets =
last_update_accumulated_render_targets();
PerformTransfersAndResolveClears(1 + xenos::kMaxColorRenderTargets,
depth_and_color_render_targets,
last_update_transfers());
uint32_t render_targets_are_srgb =
gamma_render_target_as_srgb_
? last_update_accumulated_color_targets_are_gamma()
: 0;
if (depth_and_color_render_targets[0]) {
render_pass_key.depth_and_color_used |= 1 << 0;
render_pass_key.depth_format =
depth_and_color_render_targets[0]->key().GetDepthFormat();
}
if (depth_and_color_render_targets[1]) {
render_pass_key.depth_and_color_used |= 1 << 1;
render_pass_key.color_0_view_format =
(render_targets_are_srgb & (1 << 0))
? xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA
: depth_and_color_render_targets[1]->key().GetColorFormat();
}
if (depth_and_color_render_targets[2]) {
render_pass_key.depth_and_color_used |= 1 << 2;
render_pass_key.color_1_view_format =
(render_targets_are_srgb & (1 << 1))
? xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA
: depth_and_color_render_targets[2]->key().GetColorFormat();
}
if (depth_and_color_render_targets[3]) {
render_pass_key.depth_and_color_used |= 1 << 3;
render_pass_key.color_2_view_format =
(render_targets_are_srgb & (1 << 2))
? xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA
: depth_and_color_render_targets[3]->key().GetColorFormat();
}
if (depth_and_color_render_targets[4]) {
render_pass_key.depth_and_color_used |= 1 << 4;
render_pass_key.color_3_view_format =
(render_targets_are_srgb & (1 << 3))
? xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA
: depth_and_color_render_targets[4]->key().GetColorFormat();
}
const Framebuffer* framebuffer = last_update_framebuffer_;
VkRenderPass render_pass = last_update_render_pass_key_ == render_pass_key
? last_update_render_pass_
: VK_NULL_HANDLE;
if (render_pass == VK_NULL_HANDLE) {
render_pass = GetHostRenderTargetsRenderPass(render_pass_key);
if (render_pass == VK_NULL_HANDLE) {
return false;
}
// Framebuffer for a different render pass needed now.
framebuffer = nullptr;
}
uint32_t pitch_tiles_at_32bpp =
((rb_surface_info.surface_pitch << uint32_t(
rb_surface_info.msaa_samples >= xenos::MsaaSamples::k4X)) +
(xenos::kEdramTileWidthSamples - 1)) /
xenos::kEdramTileWidthSamples;
if (framebuffer) {
if (last_update_framebuffer_pitch_tiles_at_32bpp_ !=
pitch_tiles_at_32bpp ||
std::memcmp(last_update_framebuffer_attachments_,
depth_and_color_render_targets,
sizeof(last_update_framebuffer_attachments_))) {
framebuffer = nullptr;
}
}
if (!framebuffer) {
framebuffer = GetHostRenderTargetsFramebuffer(
render_pass_key, pitch_tiles_at_32bpp,
depth_and_color_render_targets);
if (!framebuffer) {
return false;
}
}
// Successful update - write the new configuration.
last_update_render_pass_key_ = render_pass_key;
last_update_render_pass_ = render_pass;
last_update_framebuffer_pitch_tiles_at_32bpp_ = pitch_tiles_at_32bpp;
std::memcpy(last_update_framebuffer_attachments_,
depth_and_color_render_targets,
sizeof(last_update_framebuffer_attachments_));
last_update_framebuffer_ = framebuffer;
// Transition the used render targets.
for (uint32_t i = 0; i < 1 + xenos::kMaxColorRenderTargets; ++i) {
RenderTarget* rt = depth_and_color_render_targets[i];
if (!rt) {
continue;
}
auto& vulkan_rt = *static_cast<VulkanRenderTarget*>(rt);
VkPipelineStageFlags rt_dst_stage_mask;
VkAccessFlags rt_dst_access_mask;
VkImageLayout rt_new_layout;
VulkanRenderTarget::GetDrawUsage(i == 0, &rt_dst_stage_mask,
&rt_dst_access_mask, &rt_new_layout);
command_processor_.PushImageMemoryBarrier(
vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
i ? VK_IMAGE_ASPECT_COLOR_BIT
: (VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)),
vulkan_rt.current_stage_mask(), rt_dst_stage_mask,
vulkan_rt.current_access_mask(), rt_dst_access_mask,
vulkan_rt.current_layout(), rt_new_layout);
vulkan_rt.SetUsage(rt_dst_stage_mask, rt_dst_access_mask,
rt_new_layout);
}
} break;
case Path::kPixelShaderInterlock: {
// For FSI, only the barrier is needed - already scheduled if required.
// But the buffer will be used for FSI drawing now.
UseEdramBuffer(EdramBufferUsage::kFragmentReadWrite);
// Commit preceding unordered (but not FSI) writes like clears as they
// aren't synchronized with FSI accesses.
CommitEdramBufferShaderWrites(
EdramBufferModificationStatus::kViaUnordered);
// TODO(Triang3l): Check if this draw call modifies color or depth /
// stencil, at least coarsely, to prevent useless barriers.
MarkEdramBufferModified(
EdramBufferModificationStatus::kViaFragmentShaderInterlock);
last_update_render_pass_key_ = render_pass_key;
last_update_render_pass_ = fsi_render_pass_;
last_update_framebuffer_ = &fsi_framebuffer_;
} break;
default:
assert_unhandled_case(GetPath());
return false;
}
return true;
}
VkRenderPass VulkanRenderTargetCache::GetHostRenderTargetsRenderPass(
RenderPassKey key) {
assert_true(GetPath() == Path::kHostRenderTargets);
auto it = render_passes_.find(key);
if (it != render_passes_.end()) {
return it->second;
}
VkSampleCountFlagBits samples;
switch (key.msaa_samples) {
case xenos::MsaaSamples::k1X:
samples = VK_SAMPLE_COUNT_1_BIT;
break;
case xenos::MsaaSamples::k2X:
samples = IsMsaa2xSupported(key.depth_and_color_used != 0)
? VK_SAMPLE_COUNT_2_BIT
: VK_SAMPLE_COUNT_4_BIT;
break;
case xenos::MsaaSamples::k4X:
samples = VK_SAMPLE_COUNT_4_BIT;
break;
default:
return VK_NULL_HANDLE;
}
VkAttachmentDescription attachments[1 + xenos::kMaxColorRenderTargets];
if (key.depth_and_color_used & 0b1) {
VkAttachmentDescription& attachment = attachments[0];
attachment.flags = 0;
attachment.format = GetDepthVulkanFormat(key.depth_format);
attachment.samples = samples;
attachment.loadOp = VK_ATTACHMENT_LOAD_OP_LOAD;
attachment.storeOp = VK_ATTACHMENT_STORE_OP_STORE;
attachment.stencilLoadOp = VK_ATTACHMENT_LOAD_OP_LOAD;
attachment.stencilStoreOp = VK_ATTACHMENT_STORE_OP_STORE;
attachment.initialLayout = VulkanRenderTarget::kDepthDrawLayout;
attachment.finalLayout = VulkanRenderTarget::kDepthDrawLayout;
}
VkAttachmentReference color_attachments[xenos::kMaxColorRenderTargets];
xenos::ColorRenderTargetFormat color_formats[] = {
key.color_0_view_format,
key.color_1_view_format,
key.color_2_view_format,
key.color_3_view_format,
};
for (uint32_t i = 0; i < xenos::kMaxColorRenderTargets; ++i) {
VkAttachmentReference& color_attachment = color_attachments[i];
color_attachment.layout = VulkanRenderTarget::kColorDrawLayout;
uint32_t attachment_bit = uint32_t(1) << (1 + i);
if (!(key.depth_and_color_used & attachment_bit)) {
color_attachment.attachment = VK_ATTACHMENT_UNUSED;
continue;
}
uint32_t attachment_index =
xe::bit_count(key.depth_and_color_used & (attachment_bit - 1));
color_attachment.attachment = attachment_index;
VkAttachmentDescription& attachment = attachments[attachment_index];
attachment.flags = 0;
xenos::ColorRenderTargetFormat color_format = color_formats[i];
attachment.format =
key.color_rts_use_transfer_formats
? GetColorOwnershipTransferVulkanFormat(color_format)
: GetColorVulkanFormat(color_format);
attachment.samples = samples;
attachment.loadOp = VK_ATTACHMENT_LOAD_OP_LOAD;
attachment.storeOp = VK_ATTACHMENT_STORE_OP_STORE;
attachment.stencilLoadOp = VK_ATTACHMENT_LOAD_OP_DONT_CARE;
attachment.stencilStoreOp = VK_ATTACHMENT_STORE_OP_DONT_CARE;
attachment.initialLayout = VulkanRenderTarget::kColorDrawLayout;
attachment.finalLayout = VulkanRenderTarget::kColorDrawLayout;
}
VkAttachmentReference depth_stencil_attachment;
depth_stencil_attachment.attachment =
(key.depth_and_color_used & 0b1) ? 0 : VK_ATTACHMENT_UNUSED;
depth_stencil_attachment.layout = VulkanRenderTarget::kDepthDrawLayout;
VkSubpassDescription subpass;
subpass.flags = 0;
subpass.pipelineBindPoint = VK_PIPELINE_BIND_POINT_GRAPHICS;
subpass.inputAttachmentCount = 0;
subpass.pInputAttachments = nullptr;
subpass.colorAttachmentCount =
32 - xe::lzcnt(uint32_t(key.depth_and_color_used >> 1));
subpass.pColorAttachments = color_attachments;
subpass.pResolveAttachments = nullptr;
subpass.pDepthStencilAttachment =
(key.depth_and_color_used & 0b1) ? &depth_stencil_attachment : nullptr;
subpass.preserveAttachmentCount = 0;
subpass.pPreserveAttachments = nullptr;
VkPipelineStageFlags dependency_stage_mask = 0;
VkAccessFlags dependency_access_mask = 0;
if (key.depth_and_color_used & 0b1) {
dependency_stage_mask |= VulkanRenderTarget::kDepthDrawStageMask;
dependency_access_mask |= VulkanRenderTarget::kDepthDrawAccessMask;
}
if (key.depth_and_color_used >> 1) {
dependency_stage_mask |= VulkanRenderTarget::kColorDrawStageMask;
dependency_access_mask |= VulkanRenderTarget::kColorDrawAccessMask;
}
VkSubpassDependency subpass_dependencies[2];
subpass_dependencies[0].srcSubpass = VK_SUBPASS_EXTERNAL;
subpass_dependencies[0].dstSubpass = 0;
subpass_dependencies[0].srcStageMask = dependency_stage_mask;
subpass_dependencies[0].dstStageMask = dependency_stage_mask;
subpass_dependencies[0].srcAccessMask = dependency_access_mask;
subpass_dependencies[0].dstAccessMask = dependency_access_mask;
subpass_dependencies[0].dependencyFlags = VK_DEPENDENCY_BY_REGION_BIT;
subpass_dependencies[1].srcSubpass = 0;
subpass_dependencies[1].dstSubpass = VK_SUBPASS_EXTERNAL;
subpass_dependencies[1].srcStageMask = dependency_stage_mask;
subpass_dependencies[1].dstStageMask = dependency_stage_mask;
subpass_dependencies[1].srcAccessMask = dependency_access_mask;
subpass_dependencies[1].dstAccessMask = dependency_access_mask;
subpass_dependencies[1].dependencyFlags = VK_DEPENDENCY_BY_REGION_BIT;
VkRenderPassCreateInfo render_pass_create_info;
render_pass_create_info.sType = VK_STRUCTURE_TYPE_RENDER_PASS_CREATE_INFO;
render_pass_create_info.pNext = nullptr;
render_pass_create_info.flags = 0;
render_pass_create_info.attachmentCount =
xe::bit_count(key.depth_and_color_used);
render_pass_create_info.pAttachments = attachments;
render_pass_create_info.subpassCount = 1;
render_pass_create_info.pSubpasses = &subpass;
render_pass_create_info.dependencyCount =
key.depth_and_color_used ? uint32_t(xe::countof(subpass_dependencies))
: 0;
render_pass_create_info.pDependencies = subpass_dependencies;
const ui::vulkan::VulkanProvider& provider =
command_processor_.GetVulkanProvider();
const ui::vulkan::VulkanProvider::DeviceFunctions& dfn = provider.dfn();
VkDevice device = provider.device();
VkRenderPass render_pass;
if (dfn.vkCreateRenderPass(device, &render_pass_create_info, nullptr,
&render_pass) != VK_SUCCESS) {
XELOGE("VulkanRenderTargetCache: Failed to create a render pass");
render_passes_.emplace(key, VK_NULL_HANDLE);
return VK_NULL_HANDLE;
}
render_passes_.emplace(key, render_pass);
return render_pass;
}
VkFormat VulkanRenderTargetCache::GetDepthVulkanFormat(
xenos::DepthRenderTargetFormat format) const {
if (format == xenos::DepthRenderTargetFormat::kD24S8 &&
depth_unorm24_vulkan_format_supported()) {
return VK_FORMAT_D24_UNORM_S8_UINT;
}
return VK_FORMAT_D32_SFLOAT_S8_UINT;
}
VkFormat VulkanRenderTargetCache::GetColorVulkanFormat(
xenos::ColorRenderTargetFormat format) const {
switch (format) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
return VK_FORMAT_R8G8B8A8_UNORM;
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA:
return gamma_render_target_as_srgb_ ? VK_FORMAT_R8G8B8A8_SRGB
: VK_FORMAT_R8G8B8A8_UNORM;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10:
return VK_FORMAT_A8B8G8R8_UNORM_PACK32;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT_AS_16_16_16_16:
return VK_FORMAT_R16G16B16A16_SFLOAT;
case xenos::ColorRenderTargetFormat::k_16_16:
// TODO(Triang3l): Fallback to float16 (disregarding clearing correctness
// likely) - possibly on render target gathering, treating them entirely
// as float16.
return VK_FORMAT_R16G16_SNORM;
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
// TODO(Triang3l): Fallback to float16 (disregarding clearing correctness
// likely) - possibly on render target gathering, treating them entirely
// as float16.
return VK_FORMAT_R16G16B16A16_SNORM;
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT:
return VK_FORMAT_R16G16_SFLOAT;
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT:
return VK_FORMAT_R16G16B16A16_SFLOAT;
case xenos::ColorRenderTargetFormat::k_32_FLOAT:
return VK_FORMAT_R32_SFLOAT;
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT:
return VK_FORMAT_R32G32_SFLOAT;
default:
assert_unhandled_case(format);
return VK_FORMAT_UNDEFINED;
}
}
VkFormat VulkanRenderTargetCache::GetColorOwnershipTransferVulkanFormat(
xenos::ColorRenderTargetFormat format, bool* is_integer_out) const {
if (is_integer_out) {
*is_integer_out = true;
}
// Floating-point numbers have NaNs that need to be propagated without
// modifications to the bit representation, and SNORM has two representations
// of -1.
switch (format) {
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT:
return VK_FORMAT_R16G16_UINT;
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT:
return VK_FORMAT_R16G16B16A16_UINT;
case xenos::ColorRenderTargetFormat::k_32_FLOAT:
return VK_FORMAT_R32_UINT;
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT:
return VK_FORMAT_R32G32_UINT;
default:
if (is_integer_out) {
*is_integer_out = false;
}
return GetColorVulkanFormat(format);
}
}
VulkanRenderTargetCache::VulkanRenderTarget::~VulkanRenderTarget() {
const ui::vulkan::VulkanProvider& provider =
render_target_cache_.command_processor_.GetVulkanProvider();
const ui::vulkan::VulkanProvider::DeviceFunctions& dfn = provider.dfn();
VkDevice device = provider.device();
ui::vulkan::SingleLayoutDescriptorSetPool& descriptor_set_pool =
key().is_depth
? *render_target_cache_.descriptor_set_pool_sampled_image_x2_
: *render_target_cache_.descriptor_set_pool_sampled_image_;
descriptor_set_pool.Free(descriptor_set_index_transfer_source_);
if (view_color_transfer_separate_ != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_color_transfer_separate_, nullptr);
}
if (view_srgb_ != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_srgb_, nullptr);
}
if (view_stencil_ != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_stencil_, nullptr);
}
if (view_depth_stencil_ != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_depth_stencil_, nullptr);
}
dfn.vkDestroyImageView(device, view_depth_color_, nullptr);
dfn.vkDestroyImage(device, image_, nullptr);
dfn.vkFreeMemory(device, memory_, nullptr);
}
uint32_t VulkanRenderTargetCache::GetMaxRenderTargetWidth() const {
const VkPhysicalDeviceLimits& device_limits =
command_processor_.GetVulkanProvider().device_properties().limits;
return std::min(device_limits.maxFramebufferWidth,
device_limits.maxImageDimension2D);
}
uint32_t VulkanRenderTargetCache::GetMaxRenderTargetHeight() const {
const VkPhysicalDeviceLimits& device_limits =
command_processor_.GetVulkanProvider().device_properties().limits;
return std::min(device_limits.maxFramebufferHeight,
device_limits.maxImageDimension2D);
}
RenderTargetCache::RenderTarget* VulkanRenderTargetCache::CreateRenderTarget(
RenderTargetKey key) {
const ui::vulkan::VulkanProvider& provider =
command_processor_.GetVulkanProvider();
const ui::vulkan::VulkanProvider::DeviceFunctions& dfn = provider.dfn();
VkDevice device = provider.device();
// Create the image.
VkImageCreateInfo image_create_info;
image_create_info.sType = VK_STRUCTURE_TYPE_IMAGE_CREATE_INFO;
image_create_info.pNext = nullptr;
image_create_info.flags = 0;
image_create_info.imageType = VK_IMAGE_TYPE_2D;
image_create_info.extent.width = key.GetWidth() * draw_resolution_scale_x();
image_create_info.extent.height =
GetRenderTargetHeight(key.pitch_tiles_at_32bpp, key.msaa_samples) *
draw_resolution_scale_y();
image_create_info.extent.depth = 1;
image_create_info.mipLevels = 1;
image_create_info.arrayLayers = 1;
if (key.msaa_samples == xenos::MsaaSamples::k2X &&
!msaa_2x_attachments_supported_) {
image_create_info.samples = VK_SAMPLE_COUNT_4_BIT;
} else {
image_create_info.samples =
VkSampleCountFlagBits(uint32_t(1) << uint32_t(key.msaa_samples));
}
image_create_info.tiling = VK_IMAGE_TILING_OPTIMAL;
image_create_info.usage = VK_IMAGE_USAGE_SAMPLED_BIT;
image_create_info.sharingMode = VK_SHARING_MODE_EXCLUSIVE;
image_create_info.queueFamilyIndexCount = 0;
image_create_info.pQueueFamilyIndices = nullptr;
image_create_info.initialLayout = VK_IMAGE_LAYOUT_UNDEFINED;
VkFormat transfer_format;
bool is_srgb_view_needed = false;
if (key.is_depth) {
image_create_info.format = GetDepthVulkanFormat(key.GetDepthFormat());
transfer_format = image_create_info.format;
image_create_info.usage |= VK_IMAGE_USAGE_DEPTH_STENCIL_ATTACHMENT_BIT;
} else {
xenos::ColorRenderTargetFormat color_format = key.GetColorFormat();
image_create_info.format = GetColorVulkanFormat(color_format);
transfer_format = GetColorOwnershipTransferVulkanFormat(color_format);
is_srgb_view_needed =
gamma_render_target_as_srgb_ &&
(color_format == xenos::ColorRenderTargetFormat::k_8_8_8_8 ||
color_format == xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA);
if (image_create_info.format != transfer_format || is_srgb_view_needed) {
image_create_info.flags |= VK_IMAGE_CREATE_MUTABLE_FORMAT_BIT;
}
image_create_info.usage |= VK_IMAGE_USAGE_COLOR_ATTACHMENT_BIT;
}
if (image_create_info.format == VK_FORMAT_UNDEFINED) {
XELOGE("VulkanRenderTargetCache: Unknown {} render target format {}",
key.is_depth ? "depth" : "color", key.resource_format);
return nullptr;
}
VkImage image;
VkDeviceMemory memory;
if (!ui::vulkan::util::CreateDedicatedAllocationImage(
provider, image_create_info,
ui::vulkan::util::MemoryPurpose::kDeviceLocal, image, memory)) {
XELOGE(
"VulkanRenderTarget: Failed to create a {}x{} {}xMSAA {} render target "
"image",
image_create_info.extent.width, image_create_info.extent.height,
uint32_t(1) << uint32_t(key.msaa_samples), key.GetFormatName());
return nullptr;
}
// Create the image views.
VkImageViewCreateInfo view_create_info;
view_create_info.sType = VK_STRUCTURE_TYPE_IMAGE_VIEW_CREATE_INFO;
view_create_info.pNext = nullptr;
view_create_info.flags = 0;
view_create_info.image = image;
view_create_info.viewType = VK_IMAGE_VIEW_TYPE_2D;
view_create_info.format = image_create_info.format;
view_create_info.components.r = VK_COMPONENT_SWIZZLE_IDENTITY;
view_create_info.components.g = VK_COMPONENT_SWIZZLE_IDENTITY;
view_create_info.components.b = VK_COMPONENT_SWIZZLE_IDENTITY;
view_create_info.components.a = VK_COMPONENT_SWIZZLE_IDENTITY;
view_create_info.subresourceRange =
ui::vulkan::util::InitializeSubresourceRange(
key.is_depth ? VK_IMAGE_ASPECT_DEPTH_BIT : VK_IMAGE_ASPECT_COLOR_BIT);
VkImageView view_depth_color;
if (dfn.vkCreateImageView(device, &view_create_info, nullptr,
&view_depth_color) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTarget: Failed to create a {} view for a {}x{} {}xMSAA {} "
"render target",
key.is_depth ? "depth" : "color", image_create_info.extent.width,
image_create_info.extent.height,
uint32_t(1) << uint32_t(key.msaa_samples), key.GetFormatName());
dfn.vkDestroyImage(device, image, nullptr);
dfn.vkFreeMemory(device, memory, nullptr);
return nullptr;
}
VkImageView view_depth_stencil = VK_NULL_HANDLE;
VkImageView view_stencil = VK_NULL_HANDLE;
VkImageView view_srgb = VK_NULL_HANDLE;
VkImageView view_color_transfer_separate = VK_NULL_HANDLE;
if (key.is_depth) {
view_create_info.subresourceRange.aspectMask =
VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT;
if (dfn.vkCreateImageView(device, &view_create_info, nullptr,
&view_depth_stencil) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTarget: Failed to create a depth / stencil view for a "
"{}x{} {}xMSAA {} render target",
image_create_info.extent.width, image_create_info.extent.height,
uint32_t(1) << uint32_t(key.msaa_samples),
xenos::GetDepthRenderTargetFormatName(key.GetDepthFormat()));
dfn.vkDestroyImageView(device, view_depth_color, nullptr);
dfn.vkDestroyImage(device, image, nullptr);
dfn.vkFreeMemory(device, memory, nullptr);
return nullptr;
}
view_create_info.subresourceRange.aspectMask = VK_IMAGE_ASPECT_STENCIL_BIT;
if (dfn.vkCreateImageView(device, &view_create_info, nullptr,
&view_stencil) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTarget: Failed to create a stencil view for a {}x{} "
"{}xMSAA render target",
image_create_info.extent.width, image_create_info.extent.height,
uint32_t(1) << uint32_t(key.msaa_samples),
xenos::GetDepthRenderTargetFormatName(key.GetDepthFormat()));
dfn.vkDestroyImageView(device, view_depth_stencil, nullptr);
dfn.vkDestroyImageView(device, view_depth_color, nullptr);
dfn.vkDestroyImage(device, image, nullptr);
dfn.vkFreeMemory(device, memory, nullptr);
return nullptr;
}
} else {
if (is_srgb_view_needed) {
view_create_info.format = VK_FORMAT_R8G8B8A8_SRGB;
if (dfn.vkCreateImageView(device, &view_create_info, nullptr,
&view_srgb) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTarget: Failed to create an sRGB view for a {}x{} "
"{}xMSAA render target",
image_create_info.extent.width, image_create_info.extent.height,
uint32_t(1) << uint32_t(key.msaa_samples),
xenos::GetColorRenderTargetFormatName(key.GetColorFormat()));
dfn.vkDestroyImageView(device, view_depth_color, nullptr);
dfn.vkDestroyImage(device, image, nullptr);
dfn.vkFreeMemory(device, memory, nullptr);
return nullptr;
}
}
if (transfer_format != image_create_info.format) {
view_create_info.format = transfer_format;
if (dfn.vkCreateImageView(device, &view_create_info, nullptr,
&view_color_transfer_separate) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTarget: Failed to create a transfer view for a {}x{} "
"{}xMSAA {} render target",
image_create_info.extent.width, image_create_info.extent.height,
uint32_t(1) << uint32_t(key.msaa_samples), key.GetFormatName());
if (view_srgb != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_srgb, nullptr);
}
dfn.vkDestroyImageView(device, view_depth_color, nullptr);
dfn.vkDestroyImage(device, image, nullptr);
dfn.vkFreeMemory(device, memory, nullptr);
return nullptr;
}
}
}
ui::vulkan::SingleLayoutDescriptorSetPool& descriptor_set_pool =
key.is_depth ? *descriptor_set_pool_sampled_image_x2_
: *descriptor_set_pool_sampled_image_;
size_t descriptor_set_index_transfer_source = descriptor_set_pool.Allocate();
if (descriptor_set_index_transfer_source == SIZE_MAX) {
XELOGE(
"VulkanRenderTargetCache: Failed to allocate sampled image descriptors "
"for a {} render target",
key.is_depth ? "depth/stencil" : "color");
if (view_color_transfer_separate != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_color_transfer_separate, nullptr);
}
if (view_srgb != VK_NULL_HANDLE) {
dfn.vkDestroyImageView(device, view_srgb, nullptr);
}
dfn.vkDestroyImageView(device, view_depth_color, nullptr);
dfn.vkDestroyImage(device, image, nullptr);
dfn.vkFreeMemory(device, memory, nullptr);
return nullptr;
}
VkDescriptorSet descriptor_set_transfer_source =
descriptor_set_pool.Get(descriptor_set_index_transfer_source);
VkWriteDescriptorSet descriptor_set_write[2];
VkDescriptorImageInfo descriptor_set_write_depth_color;
descriptor_set_write_depth_color.sampler = VK_NULL_HANDLE;
descriptor_set_write_depth_color.imageView =
view_color_transfer_separate != VK_NULL_HANDLE
? view_color_transfer_separate
: view_depth_color;
descriptor_set_write_depth_color.imageLayout =
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
descriptor_set_write[0].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
descriptor_set_write[0].pNext = nullptr;
descriptor_set_write[0].dstSet = descriptor_set_transfer_source;
descriptor_set_write[0].dstBinding = 0;
descriptor_set_write[0].dstArrayElement = 0;
descriptor_set_write[0].descriptorCount = 1;
descriptor_set_write[0].descriptorType = VK_DESCRIPTOR_TYPE_SAMPLED_IMAGE;
descriptor_set_write[0].pImageInfo = &descriptor_set_write_depth_color;
descriptor_set_write[0].pBufferInfo = nullptr;
descriptor_set_write[0].pTexelBufferView = nullptr;
VkDescriptorImageInfo descriptor_set_write_stencil;
if (key.is_depth) {
descriptor_set_write_stencil.sampler = VK_NULL_HANDLE;
descriptor_set_write_stencil.imageView = view_stencil;
descriptor_set_write_stencil.imageLayout =
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
descriptor_set_write[1].sType = VK_STRUCTURE_TYPE_WRITE_DESCRIPTOR_SET;
descriptor_set_write[1].pNext = nullptr;
descriptor_set_write[1].dstSet = descriptor_set_transfer_source;
descriptor_set_write[1].dstBinding = 1;
descriptor_set_write[1].dstArrayElement = 0;
descriptor_set_write[1].descriptorCount = 1;
descriptor_set_write[1].descriptorType = VK_DESCRIPTOR_TYPE_SAMPLED_IMAGE;
descriptor_set_write[1].pImageInfo = &descriptor_set_write_stencil;
descriptor_set_write[1].pBufferInfo = nullptr;
descriptor_set_write[1].pTexelBufferView = nullptr;
}
dfn.vkUpdateDescriptorSets(device, key.is_depth ? 2 : 1, descriptor_set_write,
0, nullptr);
return new VulkanRenderTarget(key, *this, image, memory, view_depth_color,
view_depth_stencil, view_stencil, view_srgb,
view_color_transfer_separate,
descriptor_set_index_transfer_source);
}
bool VulkanRenderTargetCache::IsHostDepthEncodingDifferent(
xenos::DepthRenderTargetFormat format) const {
// TODO(Triang3l): Conversion directly in shaders.
switch (format) {
case xenos::DepthRenderTargetFormat::kD24S8:
return !depth_unorm24_vulkan_format_supported();
case xenos::DepthRenderTargetFormat::kD24FS8:
return true;
}
return false;
}
void VulkanRenderTargetCache::RequestPixelShaderInterlockBarrier() {
if (edram_buffer_usage_ == EdramBufferUsage::kFragmentReadWrite) {
CommitEdramBufferShaderWrites();
}
}
void VulkanRenderTargetCache::GetEdramBufferUsageMasks(
EdramBufferUsage usage, VkPipelineStageFlags& stage_mask_out,
VkAccessFlags& access_mask_out) {
switch (usage) {
case EdramBufferUsage::kFragmentRead:
stage_mask_out = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT;
access_mask_out = VK_ACCESS_SHADER_READ_BIT;
break;
case EdramBufferUsage::kFragmentReadWrite:
stage_mask_out = VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT;
access_mask_out = VK_ACCESS_SHADER_READ_BIT | VK_ACCESS_SHADER_WRITE_BIT;
break;
case EdramBufferUsage::kComputeRead:
stage_mask_out = VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT;
access_mask_out = VK_ACCESS_SHADER_READ_BIT;
break;
case EdramBufferUsage::kComputeWrite:
stage_mask_out = VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT;
access_mask_out = VK_ACCESS_SHADER_WRITE_BIT;
break;
case EdramBufferUsage::kTransferRead:
stage_mask_out = VK_PIPELINE_STAGE_TRANSFER_BIT;
access_mask_out = VK_ACCESS_TRANSFER_READ_BIT;
break;
case EdramBufferUsage::kTransferWrite:
stage_mask_out = VK_PIPELINE_STAGE_TRANSFER_BIT;
access_mask_out = VK_ACCESS_TRANSFER_WRITE_BIT;
break;
default:
assert_unhandled_case(usage);
}
}
void VulkanRenderTargetCache::UseEdramBuffer(EdramBufferUsage new_usage) {
if (edram_buffer_usage_ == new_usage) {
return;
}
VkPipelineStageFlags src_stage_mask, dst_stage_mask;
VkAccessFlags src_access_mask, dst_access_mask;
GetEdramBufferUsageMasks(edram_buffer_usage_, src_stage_mask,
src_access_mask);
GetEdramBufferUsageMasks(new_usage, dst_stage_mask, dst_access_mask);
if (command_processor_.PushBufferMemoryBarrier(
edram_buffer_, 0, VK_WHOLE_SIZE, src_stage_mask, dst_stage_mask,
src_access_mask, dst_access_mask)) {
// Resetting edram_buffer_modification_status_ only if the barrier has been
// truly inserted.
edram_buffer_modification_status_ =
EdramBufferModificationStatus::kUnmodified;
}
edram_buffer_usage_ = new_usage;
}
void VulkanRenderTargetCache::MarkEdramBufferModified(
EdramBufferModificationStatus modification_status) {
assert_true(modification_status !=
EdramBufferModificationStatus::kUnmodified);
switch (edram_buffer_usage_) {
case EdramBufferUsage::kFragmentReadWrite:
// max because being modified via unordered access requires stricter
// synchronization than via fragment shader interlocks.
edram_buffer_modification_status_ =
std::max(edram_buffer_modification_status_, modification_status);
break;
case EdramBufferUsage::kComputeWrite:
assert_true(modification_status ==
EdramBufferModificationStatus::kViaUnordered);
modification_status = EdramBufferModificationStatus::kViaUnordered;
break;
default:
assert_always(
"While changing the usage of the EDRAM buffer before marking it as "
"modified is handled safely (but will cause spurious marking as "
"modified after the changes have been implicitly committed by the "
"usage switch), normally that shouldn't be done and is an "
"indication of architectural mistakes. Alternatively, this may "
"indicate that the usage switch has been forgotten before writing, "
"which is a clearly invalid situation.");
}
}
void VulkanRenderTargetCache::CommitEdramBufferShaderWrites(
EdramBufferModificationStatus commit_status) {
assert_true(commit_status != EdramBufferModificationStatus::kUnmodified);
if (edram_buffer_modification_status_ < commit_status) {
return;
}
VkPipelineStageFlags stage_mask;
VkAccessFlags access_mask;
GetEdramBufferUsageMasks(edram_buffer_usage_, stage_mask, access_mask);
assert_not_zero(access_mask & VK_ACCESS_SHADER_WRITE_BIT);
command_processor_.PushBufferMemoryBarrier(
edram_buffer_, 0, VK_WHOLE_SIZE, stage_mask, stage_mask, access_mask,
access_mask, VK_QUEUE_FAMILY_IGNORED, VK_QUEUE_FAMILY_IGNORED, false);
edram_buffer_modification_status_ =
EdramBufferModificationStatus::kUnmodified;
PixelShaderInterlockFullEdramBarrierPlaced();
}
const VulkanRenderTargetCache::Framebuffer*
VulkanRenderTargetCache::GetHostRenderTargetsFramebuffer(
RenderPassKey render_pass_key, uint32_t pitch_tiles_at_32bpp,
const RenderTarget* const* depth_and_color_render_targets) {
FramebufferKey key;
key.render_pass_key = render_pass_key;
key.pitch_tiles_at_32bpp = pitch_tiles_at_32bpp;
if (render_pass_key.depth_and_color_used & (1 << 0)) {
key.depth_base_tiles = depth_and_color_render_targets[0]->key().base_tiles;
}
if (render_pass_key.depth_and_color_used & (1 << 1)) {
key.color_0_base_tiles =
depth_and_color_render_targets[1]->key().base_tiles;
}
if (render_pass_key.depth_and_color_used & (1 << 2)) {
key.color_1_base_tiles =
depth_and_color_render_targets[2]->key().base_tiles;
}
if (render_pass_key.depth_and_color_used & (1 << 3)) {
key.color_2_base_tiles =
depth_and_color_render_targets[3]->key().base_tiles;
}
if (render_pass_key.depth_and_color_used & (1 << 4)) {
key.color_3_base_tiles =
depth_and_color_render_targets[4]->key().base_tiles;
}
auto it = framebuffers_.find(key);
if (it != framebuffers_.end()) {
return &it->second;
}
const ui::vulkan::VulkanProvider& provider =
command_processor_.GetVulkanProvider();
const ui::vulkan::VulkanProvider::DeviceFunctions& dfn = provider.dfn();
VkDevice device = provider.device();
const VkPhysicalDeviceLimits& device_limits =
provider.device_properties().limits;
VkRenderPass render_pass = GetHostRenderTargetsRenderPass(render_pass_key);
if (render_pass == VK_NULL_HANDLE) {
return nullptr;
}
VkImageView attachments[1 + xenos::kMaxColorRenderTargets];
uint32_t attachment_count = 0;
uint32_t depth_and_color_rts_remaining = render_pass_key.depth_and_color_used;
uint32_t rt_index;
while (xe::bit_scan_forward(depth_and_color_rts_remaining, &rt_index)) {
depth_and_color_rts_remaining &= ~(uint32_t(1) << rt_index);
const auto& vulkan_rt = *static_cast<const VulkanRenderTarget*>(
depth_and_color_render_targets[rt_index]);
VkImageView attachment;
if (rt_index) {
attachment = render_pass_key.color_rts_use_transfer_formats
? vulkan_rt.view_color_transfer()
: vulkan_rt.view_depth_color();
} else {
attachment = vulkan_rt.view_depth_stencil();
}
attachments[attachment_count++] = attachment;
}
VkFramebufferCreateInfo framebuffer_create_info;
framebuffer_create_info.sType = VK_STRUCTURE_TYPE_FRAMEBUFFER_CREATE_INFO;
framebuffer_create_info.pNext = nullptr;
framebuffer_create_info.flags = 0;
framebuffer_create_info.renderPass = render_pass;
framebuffer_create_info.attachmentCount = attachment_count;
framebuffer_create_info.pAttachments = attachments;
VkExtent2D host_extent;
if (pitch_tiles_at_32bpp) {
host_extent.width = RenderTargetKey::GetWidth(pitch_tiles_at_32bpp,
render_pass_key.msaa_samples);
host_extent.height = GetRenderTargetHeight(pitch_tiles_at_32bpp,
render_pass_key.msaa_samples);
} else {
assert_zero(render_pass_key.depth_and_color_used);
// Still needed for occlusion queries.
host_extent.width = xenos::kTexture2DCubeMaxWidthHeight;
host_extent.height = xenos::kTexture2DCubeMaxWidthHeight;
}
// Limiting to the device limit for the case of no attachments, for which
// there's no limit imposed by the sizes of the attachments that have been
// created successfully.
host_extent.width = std::min(host_extent.width * draw_resolution_scale_x(),
device_limits.maxFramebufferWidth);
host_extent.height = std::min(host_extent.height * draw_resolution_scale_y(),
device_limits.maxFramebufferHeight);
framebuffer_create_info.width = host_extent.width;
framebuffer_create_info.height = host_extent.height;
framebuffer_create_info.layers = 1;
VkFramebuffer framebuffer;
if (dfn.vkCreateFramebuffer(device, &framebuffer_create_info, nullptr,
&framebuffer) != VK_SUCCESS) {
return nullptr;
}
// Creates at a persistent location - safe to use pointers.
return &framebuffers_
.emplace(std::piecewise_construct, std::forward_as_tuple(key),
std::forward_as_tuple(framebuffer, host_extent))
.first->second;
}
VkShaderModule VulkanRenderTargetCache::GetTransferShader(
TransferShaderKey key) {
auto shader_it = transfer_shaders_.find(key);
if (shader_it != transfer_shaders_.end()) {
return shader_it->second;
}
const ui::vulkan::VulkanProvider& provider =
command_processor_.GetVulkanProvider();
const VkPhysicalDeviceFeatures& device_features = provider.device_features();
std::vector<spv::Id> id_vector_temp;
std::vector<unsigned int> uint_vector_temp;
spv::Builder builder(spv::Spv_1_0,
(SpirvShaderTranslator::kSpirvMagicToolId << 16) | 1,
nullptr);
spv::Id ext_inst_glsl_std_450 = builder.import("GLSL.std.450");
builder.addCapability(spv::CapabilityShader);
builder.setMemoryModel(spv::AddressingModelLogical, spv::MemoryModelGLSL450);
builder.setSource(spv::SourceLanguageUnknown, 0);
spv::Id type_void = builder.makeVoidType();
spv::Id type_bool = builder.makeBoolType();
spv::Id type_int = builder.makeIntType(32);
spv::Id type_int2 = builder.makeVectorType(type_int, 2);
spv::Id type_uint = builder.makeUintType(32);
spv::Id type_uint2 = builder.makeVectorType(type_uint, 2);
spv::Id type_uint4 = builder.makeVectorType(type_uint, 4);
spv::Id type_float = builder.makeFloatType(32);
spv::Id type_float2 = builder.makeVectorType(type_float, 2);
spv::Id type_float4 = builder.makeVectorType(type_float, 4);
const TransferModeInfo& mode = kTransferModes[size_t(key.mode)];
const TransferPipelineLayoutInfo& pipeline_layout_info =
kTransferPipelineLayoutInfos[size_t(mode.pipeline_layout)];
// If not dest_is_color, it's depth, or stencil bit - 40-sample columns are
// swapped as opposed to color source.
bool dest_is_color = (mode.output == TransferOutput::kColor);
xenos::ColorRenderTargetFormat dest_color_format =
xenos::ColorRenderTargetFormat(key.dest_resource_format);
xenos::DepthRenderTargetFormat dest_depth_format =
xenos::DepthRenderTargetFormat(key.dest_resource_format);
bool dest_is_64bpp =
dest_is_color && xenos::IsColorRenderTargetFormat64bpp(dest_color_format);
xenos::ColorRenderTargetFormat source_color_format =
xenos::ColorRenderTargetFormat(key.source_resource_format);
xenos::DepthRenderTargetFormat source_depth_format =
xenos::DepthRenderTargetFormat(key.source_resource_format);
// If not source_is_color, it's depth / stencil - 40-sample columns are
// swapped as opposed to color destination.
bool source_is_color = (pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetColorTextureBit) != 0;
bool source_is_64bpp;
uint32_t source_color_format_component_count;
uint32_t source_color_texture_component_mask;
bool source_color_is_uint;
spv::Id source_color_component_type;
if (source_is_color) {
assert_zero(pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetDepthStencilTexturesBit);
source_is_64bpp =
xenos::IsColorRenderTargetFormat64bpp(source_color_format);
source_color_format_component_count =
xenos::GetColorRenderTargetFormatComponentCount(source_color_format);
if (mode.output == TransferOutput::kStencilBit) {
if (source_is_64bpp && !dest_is_64bpp) {
// Need one component, but choosing from the two 32bpp halves of the
// 64bpp sample.
source_color_texture_component_mask =
0b1 | (0b1 << (source_color_format_component_count >> 1));
} else {
// Red is at least 8 bits per component in all formats.
source_color_texture_component_mask = 0b1;
}
} else {
source_color_texture_component_mask =
(uint32_t(1) << source_color_format_component_count) - 1;
}
GetColorOwnershipTransferVulkanFormat(source_color_format,
&source_color_is_uint);
source_color_component_type = source_color_is_uint ? type_uint : type_float;
} else {
source_is_64bpp = false;
source_color_format_component_count = 0;
source_color_texture_component_mask = 0;
source_color_is_uint = false;
source_color_component_type = spv::NoType;
}
std::vector<spv::Id> main_interface;
// Outputs.
bool shader_uses_stencil_reference_output =
mode.output == TransferOutput::kDepth &&
provider.device_extensions().ext_shader_stencil_export;
bool dest_color_is_uint = false;
uint32_t dest_color_component_count = 0;
spv::Id type_fragment_data_component = spv::NoResult;
spv::Id type_fragment_data = spv::NoResult;
spv::Id output_fragment_data = spv::NoResult;
spv::Id output_fragment_depth = spv::NoResult;
spv::Id output_fragment_stencil_ref = spv::NoResult;
switch (mode.output) {
case TransferOutput::kColor:
GetColorOwnershipTransferVulkanFormat(dest_color_format,
&dest_color_is_uint);
dest_color_component_count =
xenos::GetColorRenderTargetFormatComponentCount(dest_color_format);
type_fragment_data_component =
dest_color_is_uint ? type_uint : type_float;
type_fragment_data =
dest_color_component_count > 1
? builder.makeVectorType(type_fragment_data_component,
dest_color_component_count)
: type_fragment_data_component;
output_fragment_data = builder.createVariable(
spv::NoPrecision, spv::StorageClassOutput, type_fragment_data,
"xe_transfer_fragment_data");
builder.addDecoration(output_fragment_data, spv::DecorationLocation,
key.dest_color_rt_index);
main_interface.push_back(output_fragment_data);
break;
case TransferOutput::kDepth:
output_fragment_depth =
builder.createVariable(spv::NoPrecision, spv::StorageClassOutput,
type_float, "gl_FragDepth");
builder.addDecoration(output_fragment_depth, spv::DecorationBuiltIn,
spv::BuiltInFragDepth);
main_interface.push_back(output_fragment_depth);
if (shader_uses_stencil_reference_output) {
builder.addExtension("SPV_EXT_shader_stencil_export");
builder.addCapability(spv::CapabilityStencilExportEXT);
output_fragment_stencil_ref =
builder.createVariable(spv::NoPrecision, spv::StorageClassOutput,
type_int, "gl_FragStencilRefARB");
builder.addDecoration(output_fragment_stencil_ref,
spv::DecorationBuiltIn,
spv::BuiltInFragStencilRefEXT);
main_interface.push_back(output_fragment_stencil_ref);
}
break;
default:
break;
}
// Bindings.
// Generating SPIR-V 1.0, no need to add bindings to the entry point's
// interface until SPIR-V 1.4.
// Color source.
bool source_is_multisampled =
key.source_msaa_samples != xenos::MsaaSamples::k1X;
spv::Id source_color_texture = spv::NoResult;
if (pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetColorTextureBit) {
source_color_texture = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniformConstant,
builder.makeImageType(source_color_component_type, spv::Dim2D, false,
false, source_is_multisampled, 1,
spv::ImageFormatUnknown),
"xe_transfer_color");
builder.addDecoration(
source_color_texture, spv::DecorationDescriptorSet,
xe::bit_count(pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetColorTextureBit - 1)));
builder.addDecoration(source_color_texture, spv::DecorationBinding, 0);
}
// Depth / stencil source.
spv::Id source_depth_texture = spv::NoResult;
spv::Id source_stencil_texture = spv::NoResult;
if (pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetDepthStencilTexturesBit) {
uint32_t source_depth_stencil_descriptor_set =
xe::bit_count(pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetDepthStencilTexturesBit - 1));
// Using `depth == false` in makeImageType because comparisons are not
// required, and other values of `depth` are causing issues in drivers.
// https://github.com/microsoft/DirectXShaderCompiler/issues/1107
if (mode.output != TransferOutput::kStencilBit) {
source_depth_texture = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniformConstant,
builder.makeImageType(type_float, spv::Dim2D, false, false,
source_is_multisampled, 1,
spv::ImageFormatUnknown),
"xe_transfer_depth");
builder.addDecoration(source_depth_texture, spv::DecorationDescriptorSet,
source_depth_stencil_descriptor_set);
builder.addDecoration(source_depth_texture, spv::DecorationBinding, 0);
}
if (mode.output != TransferOutput::kDepth ||
shader_uses_stencil_reference_output) {
source_stencil_texture = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniformConstant,
builder.makeImageType(type_uint, spv::Dim2D, false, false,
source_is_multisampled, 1,
spv::ImageFormatUnknown),
"xe_transfer_stencil");
builder.addDecoration(source_stencil_texture,
spv::DecorationDescriptorSet,
source_depth_stencil_descriptor_set);
builder.addDecoration(source_stencil_texture, spv::DecorationBinding, 1);
}
}
// Host depth source buffer.
spv::Id host_depth_source_buffer = spv::NoResult;
if (pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetHostDepthBufferBit) {
id_vector_temp.clear();
id_vector_temp.push_back(builder.makeRuntimeArray(type_uint));
// Storage buffers have std430 packing, no padding to 4-component vectors.
builder.addDecoration(id_vector_temp.back(), spv::DecorationArrayStride,
sizeof(uint32_t));
spv::Id type_host_depth_source_buffer =
builder.makeStructType(id_vector_temp, "XeTransferHostDepthBuffer");
builder.addMemberName(type_host_depth_source_buffer, 0, "host_depth");
builder.addMemberDecoration(type_host_depth_source_buffer, 0,
spv::DecorationNonWritable);
builder.addMemberDecoration(type_host_depth_source_buffer, 0,
spv::DecorationOffset, 0);
// Block since SPIR-V 1.3, but since SPIR-V 1.0 is generated, it's
// BufferBlock.
builder.addDecoration(type_host_depth_source_buffer,
spv::DecorationBufferBlock);
// StorageBuffer since SPIR-V 1.3, but since SPIR-V 1.0 is generated, it's
// Uniform.
host_depth_source_buffer = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniform,
type_host_depth_source_buffer, "xe_transfer_host_depth_buffer");
builder.addDecoration(
host_depth_source_buffer, spv::DecorationDescriptorSet,
xe::bit_count(pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetHostDepthBufferBit - 1)));
builder.addDecoration(host_depth_source_buffer, spv::DecorationBinding, 0);
}
// Host depth source texture (the depth / stencil descriptor set is reused,
// but stencil is not needed).
spv::Id host_depth_source_texture = spv::NoResult;
if (pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetHostDepthStencilTexturesBit) {
host_depth_source_texture = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniformConstant,
builder.makeImageType(
type_float, spv::Dim2D, false, false,
key.host_depth_source_msaa_samples != xenos::MsaaSamples::k1X, 1,
spv::ImageFormatUnknown),
"xe_transfer_host_depth");
builder.addDecoration(
host_depth_source_texture, spv::DecorationDescriptorSet,
xe::bit_count(
pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetHostDepthStencilTexturesBit - 1)));
builder.addDecoration(host_depth_source_texture, spv::DecorationBinding, 0);
}
// Push constants.
id_vector_temp.clear();
uint32_t push_constants_member_host_depth_address = UINT32_MAX;
if (pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordHostDepthAddressBit) {
push_constants_member_host_depth_address = uint32_t(id_vector_temp.size());
id_vector_temp.push_back(type_uint);
}
uint32_t push_constants_member_address = UINT32_MAX;
if (pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordAddressBit) {
push_constants_member_address = uint32_t(id_vector_temp.size());
id_vector_temp.push_back(type_uint);
}
uint32_t push_constants_member_stencil_mask = UINT32_MAX;
if (pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordStencilMaskBit) {
push_constants_member_stencil_mask = uint32_t(id_vector_temp.size());
id_vector_temp.push_back(type_uint);
}
spv::Id push_constants = spv::NoResult;
if (!id_vector_temp.empty()) {
spv::Id type_push_constants =
builder.makeStructType(id_vector_temp, "XeTransferPushConstants");
if (pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordHostDepthAddressBit) {
assert_true(push_constants_member_host_depth_address != UINT32_MAX);
builder.addMemberName(type_push_constants,
push_constants_member_host_depth_address,
"host_depth_address");
builder.addMemberDecoration(
type_push_constants, push_constants_member_host_depth_address,
spv::DecorationOffset,
sizeof(uint32_t) *
xe::bit_count(
pipeline_layout_info.used_push_constant_dwords &
(kTransferUsedPushConstantDwordHostDepthAddressBit - 1)));
}
if (pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordAddressBit) {
assert_true(push_constants_member_address != UINT32_MAX);
builder.addMemberName(type_push_constants, push_constants_member_address,
"address");
builder.addMemberDecoration(
type_push_constants, push_constants_member_address,
spv::DecorationOffset,
sizeof(uint32_t) *
xe::bit_count(pipeline_layout_info.used_push_constant_dwords &
(kTransferUsedPushConstantDwordAddressBit - 1)));
}
if (pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordStencilMaskBit) {
assert_true(push_constants_member_stencil_mask != UINT32_MAX);
builder.addMemberName(type_push_constants,
push_constants_member_stencil_mask, "stencil_mask");
builder.addMemberDecoration(
type_push_constants, push_constants_member_stencil_mask,
spv::DecorationOffset,
sizeof(uint32_t) *
xe::bit_count(
pipeline_layout_info.used_push_constant_dwords &
(kTransferUsedPushConstantDwordStencilMaskBit - 1)));
}
builder.addDecoration(type_push_constants, spv::DecorationBlock);
push_constants = builder.createVariable(
spv::NoPrecision, spv::StorageClassPushConstant, type_push_constants,
"xe_transfer_push_constants");
}
// Coordinate inputs.
spv::Id input_fragment_coord = builder.createVariable(
spv::NoPrecision, spv::StorageClassInput, type_float4, "gl_FragCoord");
builder.addDecoration(input_fragment_coord, spv::DecorationBuiltIn,
spv::BuiltInFragCoord);
main_interface.push_back(input_fragment_coord);
spv::Id input_sample_id = spv::NoResult;
spv::Id spec_const_sample_id = spv::NoResult;
if (key.dest_msaa_samples != xenos::MsaaSamples::k1X) {
if (device_features.sampleRateShading) {
// One draw for all samples.
builder.addCapability(spv::CapabilitySampleRateShading);
input_sample_id = builder.createVariable(
spv::NoPrecision, spv::StorageClassInput, type_int, "gl_SampleID");
builder.addDecoration(input_sample_id, spv::DecorationFlat);
builder.addDecoration(input_sample_id, spv::DecorationBuiltIn,
spv::BuiltInSampleId);
main_interface.push_back(input_sample_id);
} else {
// One sample per draw, with different sample masks.
spec_const_sample_id = builder.makeUintConstant(0, true);
builder.addName(spec_const_sample_id, "xe_transfer_sample_id");
builder.addDecoration(spec_const_sample_id, spv::DecorationSpecId, 0);
}
}
// Begin the main function.
std::vector<spv::Id> main_param_types;
std::vector<std::vector<spv::Decoration>> main_precisions;
spv::Block* main_entry;
spv::Function* main_function =
builder.makeFunctionEntry(spv::NoPrecision, type_void, "main",
main_param_types, main_precisions, &main_entry);
// Working with unsigned numbers for simplicity now, bitcasting to signed will
// be done at texture fetch.
uint32_t tile_width_samples =
xenos::kEdramTileWidthSamples * draw_resolution_scale_x();
uint32_t tile_height_samples =
xenos::kEdramTileHeightSamples * draw_resolution_scale_y();
// Split the destination pixel index into 32bpp tile and 32bpp-tile-relative
// pixel index.
// Note that division by non-power-of-two constants will include a 4-cycle
// 32*32 multiplication on AMD, even though so many bits are not needed for
// the pixel position - however, if an OpUnreachable path is inserted for the
// case when the position has upper bits set, for some reason, the code for it
// is not eliminated when compiling the shader for AMD via RenderDoc on
// Windows, as of June 2022.
uint_vector_temp.clear();
uint_vector_temp.reserve(2);
uint_vector_temp.push_back(0);
uint_vector_temp.push_back(1);
spv::Id dest_pixel_coord = builder.createUnaryOp(
spv::OpConvertFToU, type_uint2,
builder.createRvalueSwizzle(
spv::NoPrecision, type_float2,
builder.createLoad(input_fragment_coord, spv::NoPrecision),
uint_vector_temp));
spv::Id dest_pixel_x =
builder.createCompositeExtract(dest_pixel_coord, type_uint, 0);
spv::Id const_dest_tile_width_pixels = builder.makeUintConstant(
tile_width_samples >>
(uint32_t(dest_is_64bpp) +
uint32_t(key.dest_msaa_samples >= xenos::MsaaSamples::k4X)));
spv::Id dest_tile_index_x = builder.createBinOp(
spv::OpUDiv, type_uint, dest_pixel_x, const_dest_tile_width_pixels);
spv::Id dest_tile_pixel_x = builder.createBinOp(
spv::OpUMod, type_uint, dest_pixel_x, const_dest_tile_width_pixels);
spv::Id dest_pixel_y =
builder.createCompositeExtract(dest_pixel_coord, type_uint, 1);
spv::Id const_dest_tile_height_pixels = builder.makeUintConstant(
tile_height_samples >>
uint32_t(key.dest_msaa_samples >= xenos::MsaaSamples::k2X));
spv::Id dest_tile_index_y = builder.createBinOp(
spv::OpUDiv, type_uint, dest_pixel_y, const_dest_tile_height_pixels);
spv::Id dest_tile_pixel_y = builder.createBinOp(
spv::OpUMod, type_uint, dest_pixel_y, const_dest_tile_height_pixels);
assert_true(push_constants_member_address != UINT32_MAX);
id_vector_temp.clear();
id_vector_temp.push_back(
builder.makeIntConstant(int32_t(push_constants_member_address)));
spv::Id address_constant = builder.createLoad(
builder.createAccessChain(spv::StorageClassPushConstant, push_constants,
id_vector_temp),
spv::NoPrecision);
// Calculate the 32bpp tile index from its X and Y parts.
spv::Id dest_tile_index = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, address_constant,
builder.makeUintConstant(0),
builder.makeUintConstant(xenos::kEdramPitchTilesBits)),
dest_tile_index_y),
dest_tile_index_x);
// Load the destination sample index.
spv::Id dest_sample_id = spv::NoResult;
if (key.dest_msaa_samples != xenos::MsaaSamples::k1X) {
if (device_features.sampleRateShading) {
assert_true(input_sample_id != spv::NoResult);
dest_sample_id = builder.createUnaryOp(
spv::OpBitcast, type_uint,
builder.createLoad(input_sample_id, spv::NoPrecision));
} else {
assert_true(spec_const_sample_id != spv::NoResult);
// Already uint.
dest_sample_id = spec_const_sample_id;
}
}
// Transform the destination framebuffer pixel and sample coordinates into the
// source texture pixel and sample coordinates.
// First sample bit at 4x with Vulkan standard locations - horizontal sample.
// Second sample bit at 4x with Vulkan standard locations - vertical sample.
// At 2x:
// - Native 2x: top is 1 in Vulkan, bottom is 0.
// - 2x as 4x: top is 0, bottom is 3.
spv::Id source_sample_id = dest_sample_id;
spv::Id source_tile_pixel_x = dest_tile_pixel_x;
spv::Id source_tile_pixel_y = dest_tile_pixel_y;
spv::Id source_color_half = spv::NoResult;
if (!source_is_64bpp && dest_is_64bpp) {
// 32bpp -> 64bpp, need two samples of the source.
if (key.source_msaa_samples >= xenos::MsaaSamples::k4X) {
// 32bpp -> 64bpp, 4x ->.
// Source has 32bpp halves in two adjacent samples.
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// 32bpp -> 64bpp, 4x -> 4x.
// 1 destination horizontal sample = 2 source horizontal samples.
// D p0,0 s0,0 = S p0,0 s0,0 | S p0,0 s1,0
// D p0,0 s1,0 = S p1,0 s0,0 | S p1,0 s1,0
// D p0,0 s0,1 = S p0,0 s0,1 | S p0,0 s1,1
// D p0,0 s1,1 = S p1,0 s0,1 | S p1,0 s1,1
// Thus destination horizontal sample -> source horizontal pixel,
// vertical samples are 1:1.
source_sample_id =
builder.createBinOp(spv::OpBitwiseAnd, type_uint, dest_sample_id,
builder.makeUintConstant(1 << 1));
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(dest_sample_id);
id_vector_temp.push_back(dest_tile_pixel_x);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
source_tile_pixel_x =
builder.createOp(spv::OpBitFieldInsert, type_uint, id_vector_temp);
} else if (key.dest_msaa_samples == xenos::MsaaSamples::k2X) {
// 32bpp -> 64bpp, 4x -> 2x.
// 1 destination horizontal pixel = 2 source horizontal samples.
// D p0,0 s0 = S p0,0 s0,0 | S p0,0 s1,0
// D p0,0 s1 = S p0,0 s0,1 | S p0,0 s1,1
// D p1,0 s0 = S p1,0 s0,0 | S p1,0 s1,0
// D p1,0 s1 = S p1,0 s0,1 | S p1,0 s1,1
// Pixel index can be reused. Sample 1 (for native 2x) or 0 (for 2x as
// 4x) should become samples 01, sample 0 or 3 should become samples 23.
if (msaa_2x_attachments_supported_) {
source_sample_id = builder.createBinOp(
spv::OpShiftLeftLogical, type_uint,
builder.createBinOp(spv::OpBitwiseXor, type_uint, dest_sample_id,
builder.makeUintConstant(1)),
builder.makeUintConstant(1));
} else {
source_sample_id =
builder.createBinOp(spv::OpBitwiseAnd, type_uint, dest_sample_id,
builder.makeUintConstant(1 << 1));
}
} else {
// 32bpp -> 64bpp, 4x -> 1x.
// 1 destination horizontal pixel = 2 source horizontal samples.
// D p0,0 = S p0,0 s0,0 | S p0,0 s1,0
// D p0,1 = S p0,0 s0,1 | S p0,0 s1,1
// Horizontal pixel index can be reused. Vertical pixel 1 should
// become sample 2.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(builder.makeUintConstant(0));
id_vector_temp.push_back(dest_tile_pixel_y);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(1));
source_sample_id =
builder.createOp(spv::OpBitFieldInsert, type_uint, id_vector_temp);
source_tile_pixel_y =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_tile_pixel_y, builder.makeUintConstant(1));
}
} else {
// 32bpp -> 64bpp, 1x/2x ->.
// Source has 32bpp halves in two adjacent pixels.
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// 32bpp -> 64bpp, 1x/2x -> 4x.
// The X part.
// 1 destination horizontal sample = 2 source horizontal pixels.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(builder.createBinOp(
spv::OpShiftLeftLogical, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(2)));
id_vector_temp.push_back(dest_sample_id);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(1));
source_tile_pixel_x =
builder.createOp(spv::OpBitFieldInsert, type_uint, id_vector_temp);
// Y is handled by common code.
} else {
// 32bpp -> 64bpp, 1x/2x -> 1x/2x.
// The X part.
// 1 destination horizontal pixel = 2 source horizontal pixels.
source_tile_pixel_x =
builder.createBinOp(spv::OpShiftLeftLogical, type_uint,
dest_tile_pixel_x, builder.makeUintConstant(1));
// Y is handled by common code.
}
}
} else if (source_is_64bpp && !dest_is_64bpp) {
// 64bpp -> 32bpp, also the half to load.
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// 64bpp -> 32bpp, -> 4x.
// The needed half is in the destination horizontal sample index.
if (key.source_msaa_samples >= xenos::MsaaSamples::k4X) {
// 64bpp -> 32bpp, 4x -> 4x.
// D p0,0 s0,0 = S s0,0 low
// D p0,0 s1,0 = S s0,0 high
// D p1,0 s0,0 = S s1,0 low
// D p1,0 s1,0 = S s1,0 high
// Vertical pixel and sample (second bit) addressing is the same.
// However, 1 horizontal destination pixel = 1 horizontal source sample.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(dest_sample_id);
id_vector_temp.push_back(dest_tile_pixel_x);
id_vector_temp.push_back(builder.makeUintConstant(0));
id_vector_temp.push_back(builder.makeUintConstant(1));
source_sample_id =
builder.createOp(spv::OpBitFieldInsert, type_uint, id_vector_temp);
// 2 destination horizontal samples = 1 source horizontal sample, thus
// 2 destination horizontal pixels = 1 source horizontal pixel.
source_tile_pixel_x =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_tile_pixel_x, builder.makeUintConstant(1));
} else {
// 64bpp -> 32bpp, 1x/2x -> 4x.
// 2 destination horizontal samples = 1 source horizontal pixel, thus
// 1 destination horizontal pixel = 1 source horizontal pixel. Can reuse
// horizontal pixel index.
// Y is handled by common code.
}
// Half from the destination horizontal sample index.
source_color_half =
builder.createBinOp(spv::OpBitwiseAnd, type_uint, dest_sample_id,
builder.makeUintConstant(1));
} else {
// 64bpp -> 32bpp, -> 1x/2x.
// The needed half is in the destination horizontal pixel index.
if (key.source_msaa_samples >= xenos::MsaaSamples::k4X) {
// 64bpp -> 32bpp, 4x -> 1x/2x.
// (Destination horizontal pixel >> 1) & 1 = source horizontal sample
// (first bit).
source_sample_id = builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1), builder.makeUintConstant(1));
if (key.dest_msaa_samples == xenos::MsaaSamples::k2X) {
// 64bpp -> 32bpp, 4x -> 2x.
// Destination vertical samples (1/0 in the first bit for native 2x or
// 0/1 in the second bit for 2x as 4x) = source vertical samples
// (second bit).
if (msaa_2x_attachments_supported_) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(source_sample_id);
id_vector_temp.push_back(builder.createBinOp(
spv::OpBitwiseXor, type_uint, dest_sample_id,
builder.makeUintConstant(1)));
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(1));
source_sample_id = builder.createOp(spv::OpBitFieldInsert,
type_uint, id_vector_temp);
} else {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(dest_sample_id);
id_vector_temp.push_back(source_sample_id);
id_vector_temp.push_back(builder.makeUintConstant(0));
id_vector_temp.push_back(builder.makeUintConstant(1));
source_sample_id = builder.createOp(spv::OpBitFieldInsert,
type_uint, id_vector_temp);
}
} else {
// 64bpp -> 32bpp, 4x -> 1x.
// 1 destination vertical pixel = 1 source vertical sample.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(source_sample_id);
id_vector_temp.push_back(source_tile_pixel_y);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(1));
source_sample_id = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
source_tile_pixel_y = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_y,
builder.makeUintConstant(1));
}
// 2 destination horizontal pixels = 1 source horizontal sample.
// 4 destination horizontal pixels = 1 source horizontal pixel.
source_tile_pixel_x =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_tile_pixel_x, builder.makeUintConstant(2));
} else {
// 64bpp -> 32bpp, 1x/2x -> 1x/2x.
// The X part.
// 2 destination horizontal pixels = 1 destination source pixel.
source_tile_pixel_x =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_tile_pixel_x, builder.makeUintConstant(1));
// Y is handled by common code.
}
// Half from the destination horizontal pixel index.
source_color_half =
builder.createBinOp(spv::OpBitwiseAnd, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1));
}
assert_true(source_color_half != spv::NoResult);
} else {
// Same bit count.
if (key.source_msaa_samples != key.dest_msaa_samples) {
if (key.source_msaa_samples >= xenos::MsaaSamples::k4X) {
// Same BPP, 4x -> 1x/2x.
if (key.dest_msaa_samples == xenos::MsaaSamples::k2X) {
// Same BPP, 4x -> 2x.
// Horizontal pixels to samples. Vertical sample (1/0 in the first bit
// for native 2x or 0/1 in the second bit for 2x as 4x) to second
// sample bit.
if (msaa_2x_attachments_supported_) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(dest_tile_pixel_x);
id_vector_temp.push_back(builder.createBinOp(
spv::OpBitwiseXor, type_uint, dest_sample_id,
builder.makeUintConstant(1)));
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
source_sample_id = builder.createOp(spv::OpBitFieldInsert,
type_uint, id_vector_temp);
} else {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(dest_sample_id);
id_vector_temp.push_back(dest_tile_pixel_x);
id_vector_temp.push_back(builder.makeUintConstant(0));
id_vector_temp.push_back(builder.makeUintConstant(1));
source_sample_id = builder.createOp(spv::OpBitFieldInsert,
type_uint, id_vector_temp);
}
source_tile_pixel_x = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1));
} else {
// Same BPP, 4x -> 1x.
// Pixels to samples.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(builder.createBinOp(
spv::OpBitwiseAnd, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1)));
id_vector_temp.push_back(dest_tile_pixel_y);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(1));
source_sample_id = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
source_tile_pixel_x = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1));
source_tile_pixel_y = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_y,
builder.makeUintConstant(1));
}
} else {
// Same BPP, 1x/2x -> 1x/2x/4x (as long as they're different).
// Only the X part - Y is handled by common code.
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// Horizontal samples to pixels.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(dest_sample_id);
id_vector_temp.push_back(dest_tile_pixel_x);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
source_tile_pixel_x = builder.createOp(spv::OpBitFieldInsert,
type_uint, id_vector_temp);
}
}
}
}
// Common source Y and sample index for 1x/2x AA sources, independent of bits
// per sample.
if (key.source_msaa_samples < xenos::MsaaSamples::k4X &&
key.source_msaa_samples != key.dest_msaa_samples) {
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// 1x/2x -> 4x.
if (key.source_msaa_samples == xenos::MsaaSamples::k2X) {
// 2x -> 4x.
// Vertical samples (second bit) of 4x destination to vertical sample
// (1, 0 for native 2x, or 0, 3 for 2x as 4x) of 2x source.
source_sample_id =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_sample_id, builder.makeUintConstant(1));
if (msaa_2x_attachments_supported_) {
source_sample_id = builder.createBinOp(spv::OpBitwiseXor, type_uint,
source_sample_id,
builder.makeUintConstant(1));
} else {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(source_sample_id);
id_vector_temp.push_back(source_sample_id);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(1));
source_sample_id = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
} else {
// 1x -> 4x.
// Vertical samples (second bit) to Y pixels.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_sample_id, builder.makeUintConstant(1)));
id_vector_temp.push_back(dest_tile_pixel_y);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
source_tile_pixel_y =
builder.createOp(spv::OpBitFieldInsert, type_uint, id_vector_temp);
}
} else {
// 1x/2x -> different 1x/2x.
if (key.source_msaa_samples == xenos::MsaaSamples::k2X) {
// 2x -> 1x.
// Vertical pixels of 2x destination to vertical samples (1, 0 for
// native 2x, or 0, 3 for 2x as 4x) of 1x source.
source_sample_id =
builder.createBinOp(spv::OpBitwiseAnd, type_uint, dest_tile_pixel_y,
builder.makeUintConstant(1));
if (msaa_2x_attachments_supported_) {
source_sample_id = builder.createBinOp(spv::OpBitwiseXor, type_uint,
source_sample_id,
builder.makeUintConstant(1));
} else {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(source_sample_id);
id_vector_temp.push_back(source_sample_id);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(1));
source_sample_id = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
source_tile_pixel_y =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_tile_pixel_y, builder.makeUintConstant(1));
} else {
// 1x -> 2x.
// Vertical samples (1/0 in the first bit for native 2x or 0/1 in the
// second bit for 2x as 4x) of 2x destination to vertical pixels of 1x
// source.
if (msaa_2x_attachments_supported_) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(
builder.createBinOp(spv::OpBitwiseXor, type_uint, dest_sample_id,
builder.makeUintConstant(1)));
id_vector_temp.push_back(dest_tile_pixel_y);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
source_tile_pixel_y = builder.createOp(spv::OpBitFieldInsert,
type_uint, id_vector_temp);
} else {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
dest_sample_id, builder.makeUintConstant(1)));
id_vector_temp.push_back(dest_tile_pixel_y);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
source_tile_pixel_y = builder.createOp(spv::OpBitFieldInsert,
type_uint, id_vector_temp);
}
}
}
}
uint32_t source_pixel_width_dwords_log2 =
uint32_t(key.source_msaa_samples >= xenos::MsaaSamples::k4X) +
uint32_t(source_is_64bpp);
if (source_is_color != dest_is_color) {
// Copying between color and depth / stencil - swap 40-32bpp-sample columns
// in the pixel index within the source 32bpp tile.
uint32_t source_32bpp_tile_half_pixels =
tile_width_samples >> (1 + source_pixel_width_dwords_log2);
source_tile_pixel_x = builder.createUnaryOp(
spv::OpBitcast, type_uint,
builder.createBinOp(
spv::OpIAdd, type_int,
builder.createUnaryOp(spv::OpBitcast, type_int,
source_tile_pixel_x),
builder.createTriOp(
spv::OpSelect, type_int,
builder.createBinOp(
spv::OpULessThan, builder.makeBoolType(),
source_tile_pixel_x,
builder.makeUintConstant(source_32bpp_tile_half_pixels)),
builder.makeIntConstant(int32_t(source_32bpp_tile_half_pixels)),
builder.makeIntConstant(
-int32_t(source_32bpp_tile_half_pixels)))));
}
// Transform the destination 32bpp tile index into the source. After the
// addition, it may be negative - in which case, the transfer is done across
// EDRAM addressing wrapping, and xenos::kEdramTileCount must be added to it,
// but `& (xenos::kEdramTileCount - 1)` handles that regardless of the sign.
spv::Id source_tile_index = builder.createBinOp(
spv::OpBitwiseAnd, type_uint,
builder.createUnaryOp(
spv::OpBitcast, type_uint,
builder.createBinOp(
spv::OpIAdd, type_int,
builder.createUnaryOp(spv::OpBitcast, type_int, dest_tile_index),
builder.createTriOp(
spv::OpBitFieldSExtract, type_int,
builder.createUnaryOp(spv::OpBitcast, type_int,
address_constant),
builder.makeUintConstant(xenos::kEdramPitchTilesBits * 2),
builder.makeUintConstant(xenos::kEdramBaseTilesBits + 1)))),
builder.makeUintConstant(xenos::kEdramTileCount - 1));
// Split the source 32bpp tile index into X and Y tile index within the source
// image.
spv::Id source_pitch_tiles = builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, address_constant,
builder.makeUintConstant(xenos::kEdramPitchTilesBits),
builder.makeUintConstant(xenos::kEdramPitchTilesBits));
spv::Id source_tile_index_y = builder.createBinOp(
spv::OpUDiv, type_uint, source_tile_index, source_pitch_tiles);
spv::Id source_tile_index_x = builder.createBinOp(
spv::OpUMod, type_uint, source_tile_index, source_pitch_tiles);
// Finally calculate the source texture coordinates.
spv::Id source_pixel_x_int = builder.createUnaryOp(
spv::OpBitcast, type_int,
builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.makeUintConstant(tile_width_samples >>
source_pixel_width_dwords_log2),
source_tile_index_x),
source_tile_pixel_x));
spv::Id source_pixel_y_int = builder.createUnaryOp(
spv::OpBitcast, type_int,
builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.makeUintConstant(
tile_height_samples >>
uint32_t(key.source_msaa_samples >= xenos::MsaaSamples::k2X)),
source_tile_index_y),
source_tile_pixel_y));
// Load the source.
spv::Builder::TextureParameters source_texture_parameters = {};
id_vector_temp.clear();
id_vector_temp.reserve(2);
id_vector_temp.push_back(source_pixel_x_int);
id_vector_temp.push_back(source_pixel_y_int);
spv::Id source_coordinates[2] = {
builder.createCompositeConstruct(type_int2, id_vector_temp),
};
spv::Id source_sample_ids_int[2] = {};
if (key.source_msaa_samples != xenos::MsaaSamples::k1X) {
source_sample_ids_int[0] =
builder.createUnaryOp(spv::OpBitcast, type_int, source_sample_id);
} else {
source_texture_parameters.lod = builder.makeIntConstant(0);
}
// Go to the next sample or pixel along X if need to load two dwords.
bool source_load_is_two_32bpp_samples = !source_is_64bpp && dest_is_64bpp;
if (source_load_is_two_32bpp_samples) {
if (key.source_msaa_samples >= xenos::MsaaSamples::k4X) {
source_coordinates[1] = source_coordinates[0];
source_sample_ids_int[1] = builder.createBinOp(
spv::OpBitwiseOr, type_int, source_sample_ids_int[0],
builder.makeIntConstant(1));
} else {
id_vector_temp.clear();
id_vector_temp.reserve(2);
id_vector_temp.push_back(builder.createBinOp(spv::OpBitwiseOr, type_int,
source_pixel_x_int,
builder.makeIntConstant(1)));
id_vector_temp.push_back(source_pixel_y_int);
source_coordinates[1] =
builder.createCompositeConstruct(type_int2, id_vector_temp);
source_sample_ids_int[1] = source_sample_ids_int[0];
}
}
spv::Id source_color[2][4] = {};
if (source_color_texture != spv::NoResult) {
source_texture_parameters.sampler =
builder.createLoad(source_color_texture, spv::NoPrecision);
assert_true(source_color_component_type != spv::NoType);
spv::Id source_color_vec4_type =
builder.makeVectorType(source_color_component_type, 4);
for (uint32_t i = 0; i <= uint32_t(source_load_is_two_32bpp_samples); ++i) {
source_texture_parameters.coords = source_coordinates[i];
source_texture_parameters.sample = source_sample_ids_int[i];
spv::Id source_color_vec4 = builder.createTextureCall(
spv::NoPrecision, source_color_vec4_type, false, true, false, false,
false, source_texture_parameters, spv::ImageOperandsMaskNone);
uint32_t source_color_components_remaining =
source_color_texture_component_mask;
uint32_t source_color_component_index;
while (xe::bit_scan_forward(source_color_components_remaining,
&source_color_component_index)) {
source_color_components_remaining &=
~(uint32_t(1) << source_color_component_index);
source_color[i][source_color_component_index] =
builder.createCompositeExtract(source_color_vec4,
source_color_component_type,
source_color_component_index);
}
}
}
spv::Id source_depth_float[2] = {};
if (source_depth_texture != spv::NoResult) {
source_texture_parameters.sampler =
builder.createLoad(source_depth_texture, spv::NoPrecision);
for (uint32_t i = 0; i <= uint32_t(source_load_is_two_32bpp_samples); ++i) {
source_texture_parameters.coords = source_coordinates[i];
source_texture_parameters.sample = source_sample_ids_int[i];
source_depth_float[i] = builder.createCompositeExtract(
builder.createTextureCall(
spv::NoPrecision, type_float4, false, true, false, false, false,
source_texture_parameters, spv::ImageOperandsMaskNone),
type_float, 0);
}
}
spv::Id source_stencil[2] = {};
if (source_stencil_texture != spv::NoResult) {
source_texture_parameters.sampler =
builder.createLoad(source_stencil_texture, spv::NoPrecision);
for (uint32_t i = 0; i <= uint32_t(source_load_is_two_32bpp_samples); ++i) {
source_texture_parameters.coords = source_coordinates[i];
source_texture_parameters.sample = source_sample_ids_int[i];
source_stencil[i] = builder.createCompositeExtract(
builder.createTextureCall(
spv::NoPrecision, type_uint4, false, true, false, false, false,
source_texture_parameters, spv::ImageOperandsMaskNone),
type_uint, 0);
}
}
// Pick the needed 32bpp half of the 64bpp color.
if (source_is_64bpp && !dest_is_64bpp) {
uint32_t source_color_half_component_count =
source_color_format_component_count >> 1;
assert_true(source_color_half != spv::NoResult);
spv::Id source_color_is_second_half =
builder.createBinOp(spv::OpINotEqual, type_bool, source_color_half,
builder.makeUintConstant(0));
if (mode.output == TransferOutput::kStencilBit) {
source_color[0][0] = builder.createTriOp(
spv::OpSelect, source_color_component_type,
source_color_is_second_half,
source_color[0][source_color_half_component_count],
source_color[0][0]);
} else {
for (uint32_t i = 0; i < source_color_half_component_count; ++i) {
source_color[0][i] = builder.createTriOp(
spv::OpSelect, source_color_component_type,
source_color_is_second_half,
source_color[0][source_color_half_component_count + i],
source_color[0][i]);
}
}
}
if (output_fragment_stencil_ref != spv::NoResult &&
source_stencil[0] != spv::NoResult) {
// For the depth -> depth case, write the stencil directly to the output.
assert_true(mode.output == TransferOutput::kDepth);
builder.createStore(source_stencil[0], output_fragment_stencil_ref);
}
if (dest_is_64bpp) {
// Construct the 64bpp color from two 32-bit samples or one 64-bit sample.
// If `packed` (two uints) are created, use the generic path involving
// unpacking.
// Otherwise, the fragment data output must be written to directly by the
// reached control flow path.
spv::Id packed[2] = {};
if (source_is_color) {
switch (source_color_format) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA: {
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale = builder.makeFloatConstant(255.0f);
spv::Id component_width = builder.makeUintConstant(8);
for (uint32_t i = 0; i < 2; ++i) {
packed[i] = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
source_color[i][0], unorm_scale),
unorm_round_offset));
for (uint32_t j = 1; j < 4; ++j) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed[i]);
id_vector_temp.push_back(builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
source_color[i][j], unorm_scale),
unorm_round_offset)));
id_vector_temp.push_back(builder.makeUintConstant(8 * j));
id_vector_temp.push_back(component_width);
packed[i] = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10: {
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale_rgb = builder.makeFloatConstant(1023.0f);
spv::Id width_rgb = builder.makeUintConstant(10);
spv::Id unorm_scale_a = builder.makeFloatConstant(3.0f);
spv::Id width_a = builder.makeUintConstant(2);
for (uint32_t i = 0; i < 2; ++i) {
packed[i] = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
source_color[i][0], unorm_scale_rgb),
unorm_round_offset));
for (uint32_t j = 1; j < 4; ++j) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed[i]);
id_vector_temp.push_back(builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float, source_color[i][j],
j == 3 ? unorm_scale_a : unorm_scale_rgb),
unorm_round_offset)));
id_vector_temp.push_back(builder.makeUintConstant(10 * j));
id_vector_temp.push_back(j == 3 ? width_a : width_rgb);
packed[i] = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::
k_2_10_10_10_FLOAT_AS_16_16_16_16: {
spv::Id width_rgb = builder.makeUintConstant(10);
spv::Id float_0 = builder.makeFloatConstant(0.0f);
spv::Id float_1 = builder.makeFloatConstant(1.0f);
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale_a = builder.makeFloatConstant(3.0f);
spv::Id offset_a = builder.makeUintConstant(30);
spv::Id width_a = builder.makeUintConstant(2);
for (uint32_t i = 0; i < 2; ++i) {
// Float16 has a wider range for both color and alpha, also NaNs -
// clamp and convert.
packed[i] = SpirvShaderTranslator::UnclampedFloat32To7e3(
builder, source_color[i][0], ext_inst_glsl_std_450);
for (uint32_t j = 1; j < 3; ++j) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed[i]);
id_vector_temp.push_back(
SpirvShaderTranslator::UnclampedFloat32To7e3(
builder, source_color[i][j], ext_inst_glsl_std_450));
id_vector_temp.push_back(builder.makeUintConstant(10 * j));
id_vector_temp.push_back(width_rgb);
packed[i] = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
// Saturate and convert the alpha.
id_vector_temp.clear();
id_vector_temp.reserve(3);
id_vector_temp.push_back(source_color[i][3]);
id_vector_temp.push_back(float_0);
id_vector_temp.push_back(float_1);
spv::Id alpha_saturated =
builder.createBuiltinCall(type_float, ext_inst_glsl_std_450,
GLSLstd450NClamp, id_vector_temp);
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed[i]);
id_vector_temp.push_back(builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
alpha_saturated, unorm_scale_a),
unorm_round_offset)));
id_vector_temp.push_back(offset_a);
id_vector_temp.push_back(width_a);
packed[i] = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
} break;
// All 64bpp formats, and all 16 bits per component formats, are
// represented as integers in ownership transfer for safe handling of
// NaN encodings and -32768 / -32767.
// TODO(Triang3l): Handle the case when that's not true (no multisampled
// sampled images, no 16-bit UNORM, no cross-packing 32bpp aliasing on a
// portability subset device or a 64bpp format where that wouldn't help
// anyway).
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT: {
if (dest_color_format ==
xenos::ColorRenderTargetFormat::k_32_32_FLOAT) {
spv::Id component_offset_width = builder.makeUintConstant(16);
spv::Id color_16_in_32[2];
for (uint32_t i = 0; i < 2; ++i) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(source_color[i][0]);
id_vector_temp.push_back(source_color[i][1]);
id_vector_temp.push_back(component_offset_width);
id_vector_temp.push_back(component_offset_width);
color_16_in_32[i] = builder.createOp(spv::OpBitFieldInsert,
type_uint, id_vector_temp);
}
id_vector_temp.clear();
id_vector_temp.reserve(2);
id_vector_temp.push_back(color_16_in_32[0]);
id_vector_temp.push_back(color_16_in_32[1]);
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} else {
id_vector_temp.clear();
id_vector_temp.reserve(4);
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(source_color[i >> 1][i & 1]);
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
}
} break;
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT: {
if (dest_color_format ==
xenos::ColorRenderTargetFormat::k_32_32_FLOAT) {
spv::Id component_offset_width = builder.makeUintConstant(16);
spv::Id color_16_in_32[2];
for (uint32_t i = 0; i < 2; ++i) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(source_color[0][i << 1]);
id_vector_temp.push_back(source_color[0][(i << 1) + 1]);
id_vector_temp.push_back(component_offset_width);
id_vector_temp.push_back(component_offset_width);
color_16_in_32[i] = builder.createOp(spv::OpBitFieldInsert,
type_uint, id_vector_temp);
}
id_vector_temp.clear();
id_vector_temp.reserve(2);
id_vector_temp.push_back(color_16_in_32[0]);
id_vector_temp.push_back(color_16_in_32[1]);
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} else {
id_vector_temp.clear();
id_vector_temp.reserve(4);
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(source_color[0][i]);
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
}
} break;
// Float32 is transferred as uint32 to preserve NaN encodings. However,
// multisampled sampled image support is optional in Vulkan.
case xenos::ColorRenderTargetFormat::k_32_FLOAT: {
for (uint32_t i = 0; i < 2; ++i) {
packed[i] = source_color[i][0];
if (!source_color_is_uint) {
packed[i] =
builder.createUnaryOp(spv::OpBitcast, type_uint, packed[i]);
}
}
} break;
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT: {
for (uint32_t i = 0; i < 2; ++i) {
packed[i] = source_color[0][i];
if (!source_color_is_uint) {
packed[i] =
builder.createUnaryOp(spv::OpBitcast, type_uint, packed[i]);
}
}
} break;
}
} else {
assert_true(source_depth_texture != spv::NoResult);
assert_true(source_stencil_texture != spv::NoResult);
spv::Id depth_offset = builder.makeUintConstant(8);
spv::Id depth_width = builder.makeUintConstant(24);
for (uint32_t i = 0; i < 2; ++i) {
spv::Id depth24 = spv::NoResult;
switch (source_depth_format) {
case xenos::DepthRenderTargetFormat::kD24S8: {
// Round to the nearest even integer. This seems to be the
// correct conversion, adding +0.5 and rounding towards zero results
// in red instead of black in the 4D5307E6 clear shader.
id_vector_temp.clear();
id_vector_temp.push_back(builder.createBinOp(
spv::OpFMul, type_float, source_depth_float[i],
builder.makeFloatConstant(float(0xFFFFFF))));
depth24 = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBuiltinCall(type_float, ext_inst_glsl_std_450,
GLSLstd450RoundEven, id_vector_temp));
} break;
case xenos::DepthRenderTargetFormat::kD24FS8: {
depth24 = SpirvShaderTranslator::PreClampedDepthTo20e4(
builder, source_depth_float[i], depth_float24_round(), true,
ext_inst_glsl_std_450);
} break;
}
// Merge depth and stencil.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(source_stencil[i]);
id_vector_temp.push_back(depth24);
id_vector_temp.push_back(depth_offset);
id_vector_temp.push_back(depth_width);
packed[i] =
builder.createOp(spv::OpBitFieldInsert, type_uint, id_vector_temp);
}
}
// Common path unless there was a specialized one - unpack two packed 32-bit
// parts.
if (packed[0] != spv::NoResult) {
assert_true(packed[1] != spv::NoResult);
if (dest_color_format == xenos::ColorRenderTargetFormat::k_32_32_FLOAT) {
id_vector_temp.clear();
id_vector_temp.reserve(2);
id_vector_temp.push_back(packed[0]);
id_vector_temp.push_back(packed[1]);
// Multisampled sampled images are optional in Vulkan, and image views
// of different formats can't be created separately for sampled image
// and color attachment usages, so no multisampled integer sampled image
// support implies no multisampled integer framebuffer attachment
// support in Xenia.
if (!dest_color_is_uint) {
for (spv::Id& float32 : id_vector_temp) {
float32 =
builder.createUnaryOp(spv::OpBitcast, type_float, float32);
}
}
builder.createStore(builder.createCompositeConstruct(type_fragment_data,
id_vector_temp),
output_fragment_data);
} else {
spv::Id const_uint_0 = builder.makeUintConstant(0);
spv::Id const_uint_16 = builder.makeUintConstant(16);
id_vector_temp.clear();
id_vector_temp.reserve(4);
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, packed[i >> 1],
(i & 1) ? const_uint_16 : const_uint_0, const_uint_16));
}
// TODO(Triang3l): Handle the case when that's not true (no multisampled
// sampled images, no 16-bit UNORM, no cross-packing 32bpp aliasing on a
// portability subset device or a 64bpp format where that wouldn't help
// anyway).
builder.createStore(builder.createCompositeConstruct(type_fragment_data,
id_vector_temp),
output_fragment_data);
}
}
} else {
// If `packed` is created, use the generic path involving unpacking.
// - For a color destination, the packed 32bpp color.
// - For a depth / stencil destination, stencil in 0:7, depth in 8:31
// normally, or depth in 0:23 and zeros in 24:31 with packed_only_depth.
// - For a stencil bit, stencil in 0:7.
// Otherwise, the fragment data or fragment depth / stencil output must be
// written to directly by the reached control flow path.
spv::Id packed = spv::NoResult;
bool packed_only_depth = false;
if (source_is_color) {
switch (source_color_format) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA: {
if (dest_is_color &&
(dest_color_format == xenos::ColorRenderTargetFormat::k_8_8_8_8 ||
dest_color_format ==
xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA)) {
// Same format - passthrough.
id_vector_temp.clear();
id_vector_temp.reserve(4);
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(source_color[0][i]);
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} else {
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale = builder.makeFloatConstant(255.0f);
uint32_t packed_component_offset = 0;
if (mode.output == TransferOutput::kDepth) {
// When need only depth, not stencil, skip the red component, and
// put the depth from GBA directly in the lower bits.
packed_component_offset = 1;
packed_only_depth = true;
if (output_fragment_stencil_ref != spv::NoResult) {
builder.createStore(
builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
source_color[0][0],
unorm_scale),
unorm_round_offset)),
output_fragment_stencil_ref);
}
}
packed = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float,
source_color[0][packed_component_offset], unorm_scale),
unorm_round_offset));
if (mode.output != TransferOutput::kStencilBit) {
spv::Id component_width = builder.makeUintConstant(8);
for (uint32_t i = 1; i < 4 - packed_component_offset; ++i) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed);
id_vector_temp.push_back(builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float,
source_color[0][packed_component_offset + i],
unorm_scale),
unorm_round_offset)));
id_vector_temp.push_back(builder.makeUintConstant(8 * i));
id_vector_temp.push_back(component_width);
packed = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
}
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10: {
if (dest_is_color &&
(dest_color_format ==
xenos::ColorRenderTargetFormat::k_2_10_10_10 ||
dest_color_format == xenos::ColorRenderTargetFormat::
k_2_10_10_10_AS_10_10_10_10)) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(source_color[0][i]);
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} else {
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale_rgb = builder.makeFloatConstant(1023.0f);
packed = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
source_color[0][0], unorm_scale_rgb),
unorm_round_offset));
if (mode.output != TransferOutput::kStencilBit) {
spv::Id width_rgb = builder.makeUintConstant(10);
spv::Id unorm_scale_a = builder.makeFloatConstant(3.0f);
spv::Id width_a = builder.makeUintConstant(2);
for (uint32_t i = 1; i < 4; ++i) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed);
id_vector_temp.push_back(builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float, source_color[0][i],
i == 3 ? unorm_scale_a : unorm_scale_rgb),
unorm_round_offset)));
id_vector_temp.push_back(builder.makeUintConstant(10 * i));
id_vector_temp.push_back(i == 3 ? width_a : width_rgb);
packed = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
}
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::
k_2_10_10_10_FLOAT_AS_16_16_16_16: {
if (dest_is_color &&
(dest_color_format ==
xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT ||
dest_color_format == xenos::ColorRenderTargetFormat::
k_2_10_10_10_FLOAT_AS_16_16_16_16)) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(source_color[0][i]);
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} else {
// Float16 has a wider range for both color and alpha, also NaNs -
// clamp and convert.
packed = SpirvShaderTranslator::UnclampedFloat32To7e3(
builder, source_color[0][0], ext_inst_glsl_std_450);
if (mode.output != TransferOutput::kStencilBit) {
spv::Id width_rgb = builder.makeUintConstant(10);
for (uint32_t i = 1; i < 3; ++i) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed);
id_vector_temp.push_back(
SpirvShaderTranslator::UnclampedFloat32To7e3(
builder, source_color[0][i], ext_inst_glsl_std_450));
id_vector_temp.push_back(builder.makeUintConstant(10 * i));
id_vector_temp.push_back(width_rgb);
packed = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
// Saturate and convert the alpha.
id_vector_temp.clear();
id_vector_temp.reserve(3);
id_vector_temp.push_back(source_color[0][3]);
id_vector_temp.push_back(builder.makeFloatConstant(0.0f));
id_vector_temp.push_back(builder.makeFloatConstant(1.0f));
spv::Id alpha_saturated =
builder.createBuiltinCall(type_float, ext_inst_glsl_std_450,
GLSLstd450NClamp, id_vector_temp);
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed);
id_vector_temp.push_back(builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
alpha_saturated,
builder.makeFloatConstant(3.0f)),
builder.makeFloatConstant(0.5f))));
id_vector_temp.push_back(builder.makeUintConstant(30));
id_vector_temp.push_back(builder.makeUintConstant(2));
packed = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
}
} break;
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT:
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT: {
// All 64bpp formats, and all 16 bits per component formats, are
// represented as integers in ownership transfer for safe handling of
// NaN encodings and -32768 / -32767.
// TODO(Triang3l): Handle the case when that's not true (no
// multisampled sampled images, no 16-bit UNORM, no cross-packing
// 32bpp aliasing on a portability subset device or a 64bpp format
// where that wouldn't help anyway).
if (dest_is_color &&
(dest_color_format == xenos::ColorRenderTargetFormat::k_16_16 ||
dest_color_format ==
xenos::ColorRenderTargetFormat::k_16_16_FLOAT)) {
id_vector_temp.clear();
id_vector_temp.reserve(2);
for (uint32_t i = 0; i < 2; ++i) {
id_vector_temp.push_back(source_color[0][i]);
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} else {
packed = source_color[0][0];
if (mode.output != TransferOutput::kStencilBit) {
spv::Id component_offset_width = builder.makeUintConstant(16);
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed);
id_vector_temp.push_back(source_color[0][1]);
id_vector_temp.push_back(component_offset_width);
id_vector_temp.push_back(component_offset_width);
packed = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
}
} break;
// Float32 is transferred as uint32 to preserve NaN encodings. However,
// multisampled sampled image support is optional in Vulkan.
case xenos::ColorRenderTargetFormat::k_32_FLOAT:
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT: {
packed = source_color[0][0];
if (!source_color_is_uint) {
packed = builder.createUnaryOp(spv::OpBitcast, type_uint, packed);
}
} break;
}
} else if (source_depth_float[0] != spv::NoResult) {
if (mode.output == TransferOutput::kDepth &&
dest_depth_format == source_depth_format) {
builder.createStore(source_depth_float[0], output_fragment_depth);
} else {
switch (source_depth_format) {
case xenos::DepthRenderTargetFormat::kD24S8: {
// Round to the nearest even integer. This seems to be the correct
// conversion, adding +0.5 and rounding towards zero results in red
// instead of black in the 4D5307E6 clear shader.
id_vector_temp.clear();
id_vector_temp.push_back(builder.createBinOp(
spv::OpFMul, type_float, source_depth_float[0],
builder.makeFloatConstant(float(0xFFFFFF))));
packed = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBuiltinCall(type_float, ext_inst_glsl_std_450,
GLSLstd450RoundEven, id_vector_temp));
} break;
case xenos::DepthRenderTargetFormat::kD24FS8: {
packed = SpirvShaderTranslator::PreClampedDepthTo20e4(
builder, source_depth_float[0], depth_float24_round(), true,
ext_inst_glsl_std_450);
} break;
}
if (mode.output == TransferOutput::kDepth) {
packed_only_depth = true;
} else {
// Merge depth and stencil.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(source_stencil[0]);
id_vector_temp.push_back(packed);
id_vector_temp.push_back(builder.makeUintConstant(8));
id_vector_temp.push_back(builder.makeUintConstant(24));
packed = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
}
}
switch (mode.output) {
case TransferOutput::kColor: {
// Unless a special path was taken, unpack the raw 32bpp value into the
// 32bpp color output.
if (packed != spv::NoResult) {
switch (dest_color_format) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA: {
spv::Id component_width = builder.makeUintConstant(8);
spv::Id unorm_scale = builder.makeFloatConstant(1.0f / 255.0f);
id_vector_temp.clear();
id_vector_temp.reserve(4);
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(builder.createBinOp(
spv::OpFMul, type_float,
builder.createUnaryOp(
spv::OpConvertUToF, type_float,
builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, packed,
builder.makeUintConstant(8 * i), component_width)),
unorm_scale));
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10: {
spv::Id width_rgb = builder.makeUintConstant(10);
spv::Id unorm_scale_rgb =
builder.makeFloatConstant(1.0f / 1023.0f);
spv::Id width_a = builder.makeUintConstant(2);
spv::Id unorm_scale_a = builder.makeFloatConstant(1.0f / 3.0f);
id_vector_temp.clear();
id_vector_temp.reserve(4);
for (uint32_t i = 0; i < 4; ++i) {
id_vector_temp.push_back(builder.createBinOp(
spv::OpFMul, type_float,
builder.createUnaryOp(
spv::OpConvertUToF, type_float,
builder.createTriOp(spv::OpBitFieldUExtract, type_uint,
packed,
builder.makeUintConstant(10 * i),
i == 3 ? width_a : width_rgb)),
i == 3 ? unorm_scale_a : unorm_scale_rgb));
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::
k_2_10_10_10_FLOAT_AS_16_16_16_16: {
id_vector_temp.clear();
id_vector_temp.reserve(4);
// Color.
spv::Id width_rgb = builder.makeUintConstant(10);
for (uint32_t i = 0; i < 3; ++i) {
id_vector_temp.push_back(SpirvShaderTranslator::Float7e3To32(
builder, packed, 10 * i, false, ext_inst_glsl_std_450));
}
// Alpha.
id_vector_temp.push_back(builder.createBinOp(
spv::OpFMul, type_float,
builder.createUnaryOp(
spv::OpConvertUToF, type_float,
builder.createTriOp(spv::OpBitFieldUExtract, type_uint,
packed, builder.makeUintConstant(30),
builder.makeUintConstant(2))),
builder.makeFloatConstant(1.0f / 3.0f)));
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} break;
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT: {
// All 16 bits per component formats are represented as integers
// in ownership transfer for safe handling of NaN encodings and
// -32768 / -32767.
// TODO(Triang3l): Handle the case when that's not true (no
// multisampled sampled images, no 16-bit UNORM, no cross-packing
// 32bpp aliasing on a portability subset device or a 64bpp format
// where that wouldn't help anyway).
spv::Id component_offset_width = builder.makeUintConstant(16);
id_vector_temp.clear();
id_vector_temp.reserve(2);
for (uint32_t i = 0; i < 2; ++i) {
id_vector_temp.push_back(builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, packed,
i ? component_offset_width : builder.makeUintConstant(0),
component_offset_width));
}
builder.createStore(builder.createCompositeConstruct(
type_fragment_data, id_vector_temp),
output_fragment_data);
} break;
case xenos::ColorRenderTargetFormat::k_32_FLOAT: {
// Float32 is transferred as uint32 to preserve NaN encodings.
// However, multisampled sampled images are optional in Vulkan,
// and image views of different formats can't be created
// separately for sampled image and color attachment usages, so no
// multisampled integer sampled image support implies no
// multisampled integer framebuffer attachment support in Xenia.
spv::Id float32 = packed;
if (!dest_color_is_uint) {
float32 =
builder.createUnaryOp(spv::OpBitcast, type_float, float32);
}
builder.createStore(float32, output_fragment_data);
} break;
default:
// A 64bpp format (handled separately) or an invalid one.
assert_unhandled_case(dest_color_format);
}
}
} break;
case TransferOutput::kDepth: {
if (packed) {
spv::Id guest_depth24 = packed;
if (!packed_only_depth) {
// Extract the depth bits.
guest_depth24 =
builder.createBinOp(spv::OpShiftRightLogical, type_uint,
guest_depth24, builder.makeUintConstant(8));
}
// Load the host float32 depth, check if, when converted to the guest
// format, it's the same as the guest source, thus up to date, and if
// it is, write host float32 depth, otherwise do the guest -> host
// conversion.
spv::Id host_depth32 = spv::NoResult;
if (host_depth_source_texture != spv::NoResult) {
// Convert position and sample index from within the destination
// tile to within the host depth source tile, like for the guest
// render target, but for 32bpp -> 32bpp only.
spv::Id host_depth_source_sample_id = dest_sample_id;
spv::Id host_depth_source_tile_pixel_x = dest_tile_pixel_x;
spv::Id host_depth_source_tile_pixel_y = dest_tile_pixel_y;
if (key.host_depth_source_msaa_samples != key.dest_msaa_samples) {
if (key.host_depth_source_msaa_samples >=
xenos::MsaaSamples::k4X) {
// 4x -> 1x/2x.
if (key.dest_msaa_samples == xenos::MsaaSamples::k2X) {
// 4x -> 2x.
// Horizontal pixels to samples. Vertical sample (1/0 in the
// first bit for native 2x or 0/1 in the second bit for 2x as
// 4x) to second sample bit.
if (msaa_2x_attachments_supported_) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(dest_tile_pixel_x);
id_vector_temp.push_back(builder.createBinOp(
spv::OpBitwiseXor, type_uint, dest_sample_id,
builder.makeUintConstant(1)));
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
host_depth_source_sample_id = builder.createOp(
spv::OpBitFieldInsert, type_uint, id_vector_temp);
} else {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(dest_sample_id);
id_vector_temp.push_back(dest_tile_pixel_x);
id_vector_temp.push_back(builder.makeUintConstant(0));
id_vector_temp.push_back(builder.makeUintConstant(1));
host_depth_source_sample_id = builder.createOp(
spv::OpBitFieldInsert, type_uint, id_vector_temp);
}
host_depth_source_tile_pixel_x = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1));
} else {
// 4x -> 1x.
// Pixels to samples.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(builder.createBinOp(
spv::OpBitwiseAnd, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1)));
id_vector_temp.push_back(dest_tile_pixel_y);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(1));
host_depth_source_sample_id = builder.createOp(
spv::OpBitFieldInsert, type_uint, id_vector_temp);
host_depth_source_tile_pixel_x = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_x,
builder.makeUintConstant(1));
host_depth_source_tile_pixel_y = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_y,
builder.makeUintConstant(1));
}
} else {
// 1x/2x -> 1x/2x/4x (as long as they're different).
// Only the X part - Y is handled by common code.
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// Horizontal samples to pixels.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(dest_sample_id);
id_vector_temp.push_back(dest_tile_pixel_x);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
host_depth_source_tile_pixel_x = builder.createOp(
spv::OpBitFieldInsert, type_uint, id_vector_temp);
}
}
// Host depth source Y and sample index for 1x/2x AA sources.
if (key.host_depth_source_msaa_samples <
xenos::MsaaSamples::k4X) {
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// 1x/2x -> 4x.
if (key.host_depth_source_msaa_samples ==
xenos::MsaaSamples::k2X) {
// 2x -> 4x.
// Vertical samples (second bit) of 4x destination to
// vertical sample (1, 0 for native 2x, or 0, 3 for 2x as
// 4x) of 2x source.
host_depth_source_sample_id = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_sample_id,
builder.makeUintConstant(1));
if (msaa_2x_attachments_supported_) {
host_depth_source_sample_id =
builder.createBinOp(spv::OpBitwiseXor, type_uint,
host_depth_source_sample_id,
builder.makeUintConstant(1));
} else {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(host_depth_source_sample_id);
id_vector_temp.push_back(host_depth_source_sample_id);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(1));
host_depth_source_sample_id = builder.createOp(
spv::OpBitFieldInsert, type_uint, id_vector_temp);
}
} else {
// 1x -> 4x.
// Vertical samples (second bit) to Y pixels.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_sample_id,
builder.makeUintConstant(1)));
id_vector_temp.push_back(dest_tile_pixel_y);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
host_depth_source_tile_pixel_y = builder.createOp(
spv::OpBitFieldInsert, type_uint, id_vector_temp);
}
} else {
// 1x/2x -> different 1x/2x.
if (key.host_depth_source_msaa_samples ==
xenos::MsaaSamples::k2X) {
// 2x -> 1x.
// Vertical pixels of 2x destination to vertical samples (1,
// 0 for native 2x, or 0, 3 for 2x as 4x) of 1x source.
host_depth_source_sample_id = builder.createBinOp(
spv::OpBitwiseAnd, type_uint, dest_tile_pixel_y,
builder.makeUintConstant(1));
if (msaa_2x_attachments_supported_) {
host_depth_source_sample_id =
builder.createBinOp(spv::OpBitwiseXor, type_uint,
host_depth_source_sample_id,
builder.makeUintConstant(1));
} else {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(host_depth_source_sample_id);
id_vector_temp.push_back(host_depth_source_sample_id);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(1));
host_depth_source_sample_id = builder.createOp(
spv::OpBitFieldInsert, type_uint, id_vector_temp);
}
host_depth_source_tile_pixel_y = builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_tile_pixel_y,
builder.makeUintConstant(1));
} else {
// 1x -> 2x.
// Vertical samples (1/0 in the first bit for native 2x or
// 0/1 in the second bit for 2x as 4x) of 2x destination to
// vertical pixels of 1x source.
if (msaa_2x_attachments_supported_) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(builder.createBinOp(
spv::OpBitwiseXor, type_uint, dest_sample_id,
builder.makeUintConstant(1)));
id_vector_temp.push_back(dest_tile_pixel_y);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
host_depth_source_tile_pixel_y = builder.createOp(
spv::OpBitFieldInsert, type_uint, id_vector_temp);
} else {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(builder.createBinOp(
spv::OpShiftRightLogical, type_uint, dest_sample_id,
builder.makeUintConstant(1)));
id_vector_temp.push_back(dest_tile_pixel_y);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
host_depth_source_tile_pixel_y = builder.createOp(
spv::OpBitFieldInsert, type_uint, id_vector_temp);
}
}
}
}
}
assert_true(push_constants_member_host_depth_address != UINT32_MAX);
id_vector_temp.clear();
id_vector_temp.push_back(builder.makeIntConstant(
int32_t(push_constants_member_host_depth_address)));
spv::Id host_depth_address_constant = builder.createLoad(
builder.createAccessChain(spv::StorageClassPushConstant,
push_constants, id_vector_temp),
spv::NoPrecision);
// Transform the destination tile index into the host depth source.
// After the addition, it may be negative - in which case, the
// transfer is done across EDRAM addressing wrapping, and
// xenos::kEdramTileCount must be added to it, but
// `& (xenos::kEdramTileCount - 1)` handles that regardless of the
// sign.
spv::Id host_depth_source_tile_index = builder.createBinOp(
spv::OpBitwiseAnd, type_uint,
builder.createUnaryOp(
spv::OpBitcast, type_uint,
builder.createBinOp(
spv::OpIAdd, type_int,
builder.createUnaryOp(spv::OpBitcast, type_int,
dest_tile_index),
builder.createTriOp(
spv::OpBitFieldSExtract, type_int,
builder.createUnaryOp(spv::OpBitcast, type_int,
host_depth_address_constant),
builder.makeUintConstant(
xenos::kEdramPitchTilesBits * 2),
builder.makeUintConstant(
xenos::kEdramBaseTilesBits + 1)))),
builder.makeUintConstant(xenos::kEdramTileCount - 1));
// Split the host depth source tile index into X and Y tile index
// within the source image.
spv::Id host_depth_source_pitch_tiles = builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, host_depth_address_constant,
builder.makeUintConstant(xenos::kEdramPitchTilesBits),
builder.makeUintConstant(xenos::kEdramPitchTilesBits));
spv::Id host_depth_source_tile_index_y = builder.createBinOp(
spv::OpUDiv, type_uint, host_depth_source_tile_index,
host_depth_source_pitch_tiles);
spv::Id host_depth_source_tile_index_x = builder.createBinOp(
spv::OpUMod, type_uint, host_depth_source_tile_index,
host_depth_source_pitch_tiles);
// Finally calculate the host depth source texture coordinates.
spv::Id host_depth_source_pixel_x_int = builder.createUnaryOp(
spv::OpBitcast, type_int,
builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(spv::OpIMul, type_uint,
builder.makeUintConstant(
tile_width_samples >>
uint32_t(key.source_msaa_samples >=
xenos::MsaaSamples::k4X)),
host_depth_source_tile_index_x),
host_depth_source_tile_pixel_x));
spv::Id host_depth_source_pixel_y_int = builder.createUnaryOp(
spv::OpBitcast, type_int,
builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(spv::OpIMul, type_uint,
builder.makeUintConstant(
tile_height_samples >>
uint32_t(key.source_msaa_samples >=
xenos::MsaaSamples::k2X)),
host_depth_source_tile_index_y),
host_depth_source_tile_pixel_y));
// Load the host depth source.
spv::Builder::TextureParameters
host_depth_source_texture_parameters = {};
host_depth_source_texture_parameters.sampler =
builder.createLoad(host_depth_source_texture, spv::NoPrecision);
id_vector_temp.clear();
id_vector_temp.reserve(2);
id_vector_temp.push_back(host_depth_source_pixel_x_int);
id_vector_temp.push_back(host_depth_source_pixel_y_int);
host_depth_source_texture_parameters.coords =
builder.createCompositeConstruct(type_int2, id_vector_temp);
if (key.host_depth_source_msaa_samples != xenos::MsaaSamples::k1X) {
host_depth_source_texture_parameters.sample =
builder.createUnaryOp(spv::OpBitcast, type_int,
host_depth_source_sample_id);
} else {
host_depth_source_texture_parameters.lod =
builder.makeIntConstant(0);
}
host_depth32 = builder.createCompositeExtract(
builder.createTextureCall(spv::NoPrecision, type_float4, false,
true, false, false, false,
host_depth_source_texture_parameters,
spv::ImageOperandsMaskNone),
type_float, 0);
} else if (host_depth_source_buffer != spv::NoResult) {
// Get the address in the EDRAM scratch buffer and load from there.
// The beginning of the buffer is (0, 0) of the destination.
// 40-sample columns are not swapped for addressing simplicity
// (because this is used for depth -> depth transfers, where
// swapping isn't needed).
// Convert samples to pixels.
assert_true(key.host_depth_source_msaa_samples ==
xenos::MsaaSamples::k1X);
spv::Id dest_tile_sample_x = dest_tile_pixel_x;
spv::Id dest_tile_sample_y = dest_tile_pixel_y;
if (key.dest_msaa_samples >= xenos::MsaaSamples::k2X) {
if (key.dest_msaa_samples >= xenos::MsaaSamples::k4X) {
// Horizontal sample index in bit 0.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(dest_sample_id);
id_vector_temp.push_back(dest_tile_pixel_x);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
dest_tile_sample_x = builder.createOp(
spv::OpBitFieldInsert, type_uint, id_vector_temp);
}
// Vertical sample index as 1 or 0 in bit 0 for true 2x or as 0
// or 1 in bit 1 for 4x or for 2x emulated as 4x.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(builder.createBinOp(
(key.dest_msaa_samples == xenos::MsaaSamples::k2X &&
msaa_2x_attachments_supported_)
? spv::OpBitwiseXor
: spv::OpShiftRightLogical,
type_uint, dest_sample_id, builder.makeUintConstant(1)));
id_vector_temp.push_back(dest_tile_pixel_y);
id_vector_temp.push_back(builder.makeUintConstant(1));
id_vector_temp.push_back(builder.makeUintConstant(31));
dest_tile_sample_y = builder.createOp(spv::OpBitFieldInsert,
type_uint, id_vector_temp);
}
// Combine the tile sample index and the tile index.
// The tile index doesn't need to be wrapped, as the host depth is
// written to the beginning of the buffer, without the base offset.
spv::Id host_depth_offset = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.makeUintConstant(tile_width_samples *
tile_height_samples),
dest_tile_index),
builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.makeUintConstant(tile_width_samples),
dest_tile_sample_y),
dest_tile_sample_x));
id_vector_temp.clear();
id_vector_temp.reserve(2);
// The only SSBO structure member.
id_vector_temp.push_back(builder.makeIntConstant(0));
id_vector_temp.push_back(builder.createUnaryOp(
spv::OpBitcast, type_int, host_depth_offset));
// StorageBuffer since SPIR-V 1.3, but since SPIR-V 1.0 is
// generated, it's Uniform.
host_depth32 = builder.createUnaryOp(
spv::OpBitcast, type_float,
builder.createLoad(
builder.createAccessChain(spv::StorageClassUniform,
host_depth_source_buffer,
id_vector_temp),
spv::NoPrecision));
}
spv::Block* depth24_to_depth32_header = builder.getBuildPoint();
spv::Id depth24_to_depth32_convert_id = spv::NoResult;
spv::Block* depth24_to_depth32_merge = nullptr;
spv::Id host_depth24 = spv::NoResult;
if (host_depth32 != spv::NoResult) {
// Convert the host depth value to the guest format and check if it
// matches the value in the currently owning guest render target.
switch (dest_depth_format) {
case xenos::DepthRenderTargetFormat::kD24S8: {
// Round to the nearest even integer. This seems to be the
// correct conversion, adding +0.5 and rounding towards zero
// results in red instead of black in the 4D5307E6 clear shader.
id_vector_temp.clear();
id_vector_temp.push_back(builder.createBinOp(
spv::OpFMul, type_float, host_depth32,
builder.makeFloatConstant(float(0xFFFFFF))));
host_depth24 = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBuiltinCall(type_float, ext_inst_glsl_std_450,
GLSLstd450RoundEven,
id_vector_temp));
} break;
case xenos::DepthRenderTargetFormat::kD24FS8: {
host_depth24 = SpirvShaderTranslator::PreClampedDepthTo20e4(
builder, host_depth32, depth_float24_round(), true,
ext_inst_glsl_std_450);
} break;
}
assert_true(host_depth24 != spv::NoResult);
// Update the header block pointer after the conversion (to avoid
// assuming that the conversion doesn't branch).
depth24_to_depth32_header = builder.getBuildPoint();
spv::Id host_depth_outdated = builder.createBinOp(
spv::OpINotEqual, type_bool, guest_depth24, host_depth24);
spv::Block& depth24_to_depth32_convert_entry =
builder.makeNewBlock();
{
spv::Block& depth24_to_depth32_merge_block =
builder.makeNewBlock();
depth24_to_depth32_merge = &depth24_to_depth32_merge_block;
}
{
std::unique_ptr<spv::Instruction> depth24_to_depth32_merge_op =
std::make_unique<spv::Instruction>(spv::OpSelectionMerge);
depth24_to_depth32_merge_op->addIdOperand(
depth24_to_depth32_merge->getId());
depth24_to_depth32_merge_op->addImmediateOperand(
spv::SelectionControlMaskNone);
builder.getBuildPoint()->addInstruction(
std::move(depth24_to_depth32_merge_op));
}
builder.createConditionalBranch(host_depth_outdated,
&depth24_to_depth32_convert_entry,
depth24_to_depth32_merge);
builder.setBuildPoint(&depth24_to_depth32_convert_entry);
}
// Convert the guest 24-bit depth to float32 (in an open conditional
// if the host depth is also loaded).
spv::Id guest_depth32 = spv::NoResult;
switch (dest_depth_format) {
case xenos::DepthRenderTargetFormat::kD24S8: {
// Multiplying by 1.0 / 0xFFFFFF produces an incorrect result (for
// 0xC00000, for instance - which is 2_10_10_10 clear to 0001) -
// rescale from 0...0xFFFFFF to 0...0x1000000 doing what true
// float division followed by multiplication does (on x86-64 MSVC
// with default SSE rounding) - values starting from 0x800000
// become bigger by 1; then accurately bias the result's exponent.
guest_depth32 = builder.createBinOp(
spv::OpFMul, type_float,
builder.createUnaryOp(
spv::OpConvertUToF, type_float,
builder.createBinOp(
spv::OpIAdd, type_uint, guest_depth24,
builder.createBinOp(spv::OpShiftRightLogical,
type_uint, guest_depth24,
builder.makeUintConstant(23)))),
builder.makeFloatConstant(1.0f / float(1 << 24)));
} break;
case xenos::DepthRenderTargetFormat::kD24FS8: {
guest_depth32 = SpirvShaderTranslator::Depth20e4To32(
builder, guest_depth24, 0, true, false,
ext_inst_glsl_std_450);
} break;
}
assert_true(guest_depth32 != spv::NoResult);
spv::Id fragment_depth32 = guest_depth32;
if (host_depth32 != spv::NoResult) {
assert_not_null(depth24_to_depth32_merge);
spv::Id depth24_to_depth32_result_block_id =
builder.getBuildPoint()->getId();
builder.createBranch(depth24_to_depth32_merge);
builder.setBuildPoint(depth24_to_depth32_merge);
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(guest_depth32);
id_vector_temp.push_back(depth24_to_depth32_result_block_id);
id_vector_temp.push_back(host_depth32);
id_vector_temp.push_back(depth24_to_depth32_header->getId());
fragment_depth32 =
builder.createOp(spv::OpPhi, type_float, id_vector_temp);
}
builder.createStore(fragment_depth32, output_fragment_depth);
}
} break;
case TransferOutput::kStencilBit: {
if (packed) {
// Kill the sample if the needed stencil bit is not set.
assert_true(push_constants_member_stencil_mask != UINT32_MAX);
id_vector_temp.clear();
id_vector_temp.push_back(builder.makeIntConstant(
int32_t(push_constants_member_stencil_mask)));
spv::Id stencil_mask_constant = builder.createLoad(
builder.createAccessChain(spv::StorageClassPushConstant,
push_constants, id_vector_temp),
spv::NoPrecision);
spv::Id stencil_sample_passed = builder.createBinOp(
spv::OpINotEqual, type_bool,
builder.createBinOp(spv::OpBitwiseAnd, type_uint, packed,
stencil_mask_constant),
builder.makeUintConstant(0));
spv::Block& stencil_bit_kill_block = builder.makeNewBlock();
spv::Block& stencil_bit_merge_block = builder.makeNewBlock();
{
std::unique_ptr<spv::Instruction> stencil_bit_merge_op =
std::make_unique<spv::Instruction>(spv::OpSelectionMerge);
stencil_bit_merge_op->addIdOperand(stencil_bit_merge_block.getId());
stencil_bit_merge_op->addImmediateOperand(
spv::SelectionControlMaskNone);
builder.getBuildPoint()->addInstruction(
std::move(stencil_bit_merge_op));
}
builder.createConditionalBranch(stencil_sample_passed,
&stencil_bit_merge_block,
&stencil_bit_kill_block);
builder.setBuildPoint(&stencil_bit_kill_block);
builder.createNoResultOp(spv::OpKill);
builder.setBuildPoint(&stencil_bit_merge_block);
}
} break;
}
}
// End the main function and make it the entry point.
builder.leaveFunction();
builder.addExecutionMode(main_function, spv::ExecutionModeOriginUpperLeft);
if (output_fragment_depth != spv::NoResult) {
builder.addExecutionMode(main_function, spv::ExecutionModeDepthReplacing);
}
if (output_fragment_stencil_ref != spv::NoResult) {
builder.addExecutionMode(main_function,
spv::ExecutionModeStencilRefReplacingEXT);
}
spv::Instruction* entry_point =
builder.addEntryPoint(spv::ExecutionModelFragment, main_function, "main");
for (spv::Id interface_id : main_interface) {
entry_point->addIdOperand(interface_id);
}
// Serialize the shader code.
std::vector<unsigned int> shader_code;
builder.dump(shader_code);
// Create the shader module, and store the handle even if creation fails not
// to try to create it again later.
VkShaderModule shader_module = ui::vulkan::util::CreateShaderModule(
provider, reinterpret_cast<const uint32_t*>(shader_code.data()),
sizeof(uint32_t) * shader_code.size());
if (shader_module == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the render target ownership "
"transfer shader 0x{:08X}",
key.key);
}
transfer_shaders_.emplace(key, shader_module);
return shader_module;
}
VkPipeline const* VulkanRenderTargetCache::GetTransferPipelines(
TransferPipelineKey key) {
auto pipeline_it = transfer_pipelines_.find(key);
if (pipeline_it != transfer_pipelines_.end()) {
return pipeline_it->second[0] != VK_NULL_HANDLE ? pipeline_it->second.data()
: nullptr;
}
VkRenderPass render_pass =
GetHostRenderTargetsRenderPass(key.render_pass_key);
VkShaderModule fragment_shader_module = GetTransferShader(key.shader_key);
if (render_pass == VK_NULL_HANDLE ||
fragment_shader_module == VK_NULL_HANDLE) {
transfer_pipelines_.emplace(key, std::array<VkPipeline, 4>{});
return nullptr;
}
const TransferModeInfo& mode = kTransferModes[size_t(key.shader_key.mode)];
const ui::vulkan::VulkanProvider& provider =
command_processor_.GetVulkanProvider();
const ui::vulkan::VulkanProvider::DeviceFunctions& dfn = provider.dfn();
VkDevice device = provider.device();
const VkPhysicalDeviceFeatures& device_features = provider.device_features();
uint32_t dest_sample_count = uint32_t(1)
<< uint32_t(key.shader_key.dest_msaa_samples);
bool dest_is_masked_sample =
dest_sample_count > 1 && !device_features.sampleRateShading;
VkPipelineShaderStageCreateInfo shader_stages[2];
shader_stages[0].sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO;
shader_stages[0].pNext = nullptr;
shader_stages[0].flags = 0;
shader_stages[0].stage = VK_SHADER_STAGE_VERTEX_BIT;
shader_stages[0].module = transfer_passthrough_vertex_shader_;
shader_stages[0].pName = "main";
shader_stages[0].pSpecializationInfo = nullptr;
shader_stages[1].sType = VK_STRUCTURE_TYPE_PIPELINE_SHADER_STAGE_CREATE_INFO;
shader_stages[1].pNext = nullptr;
shader_stages[1].flags = 0;
shader_stages[1].stage = VK_SHADER_STAGE_FRAGMENT_BIT;
shader_stages[1].module = fragment_shader_module;
shader_stages[1].pName = "main";
shader_stages[1].pSpecializationInfo = nullptr;
VkSpecializationMapEntry sample_id_specialization_map_entry;
uint32_t sample_id_specialization_constant;
VkSpecializationInfo sample_id_specialization_info;
if (dest_is_masked_sample) {
sample_id_specialization_map_entry.constantID = 0;
sample_id_specialization_map_entry.offset = 0;
sample_id_specialization_map_entry.size = sizeof(uint32_t);
sample_id_specialization_constant = 0;
sample_id_specialization_info.mapEntryCount = 1;
sample_id_specialization_info.pMapEntries =
&sample_id_specialization_map_entry;
sample_id_specialization_info.dataSize =
sizeof(sample_id_specialization_constant);
sample_id_specialization_info.pData = &sample_id_specialization_constant;
shader_stages[1].pSpecializationInfo = &sample_id_specialization_info;
}
VkVertexInputBindingDescription vertex_input_binding;
vertex_input_binding.binding = 0;
vertex_input_binding.stride = sizeof(float) * 2;
vertex_input_binding.inputRate = VK_VERTEX_INPUT_RATE_VERTEX;
VkVertexInputAttributeDescription vertex_input_attribute;
vertex_input_attribute.location = 0;
vertex_input_attribute.binding = 0;
vertex_input_attribute.format = VK_FORMAT_R32G32_SFLOAT;
vertex_input_attribute.offset = 0;
VkPipelineVertexInputStateCreateInfo vertex_input_state;
vertex_input_state.sType =
VK_STRUCTURE_TYPE_PIPELINE_VERTEX_INPUT_STATE_CREATE_INFO;
vertex_input_state.pNext = nullptr;
vertex_input_state.flags = 0;
vertex_input_state.vertexBindingDescriptionCount = 1;
vertex_input_state.pVertexBindingDescriptions = &vertex_input_binding;
vertex_input_state.vertexAttributeDescriptionCount = 1;
vertex_input_state.pVertexAttributeDescriptions = &vertex_input_attribute;
VkPipelineInputAssemblyStateCreateInfo input_assembly_state;
input_assembly_state.sType =
VK_STRUCTURE_TYPE_PIPELINE_INPUT_ASSEMBLY_STATE_CREATE_INFO;
input_assembly_state.pNext = nullptr;
input_assembly_state.flags = 0;
input_assembly_state.topology = VK_PRIMITIVE_TOPOLOGY_TRIANGLE_LIST;
input_assembly_state.primitiveRestartEnable = VK_FALSE;
// Dynamic, to stay within maxViewportDimensions while preferring a
// power-of-two factor for converting from pixel coordinates to NDC for exact
// precision.
VkPipelineViewportStateCreateInfo viewport_state;
viewport_state.sType = VK_STRUCTURE_TYPE_PIPELINE_VIEWPORT_STATE_CREATE_INFO;
viewport_state.pNext = nullptr;
viewport_state.flags = 0;
viewport_state.viewportCount = 1;
viewport_state.pViewports = nullptr;
viewport_state.scissorCount = 1;
viewport_state.pScissors = nullptr;
VkPipelineRasterizationStateCreateInfo rasterization_state = {};
rasterization_state.sType =
VK_STRUCTURE_TYPE_PIPELINE_RASTERIZATION_STATE_CREATE_INFO;
rasterization_state.polygonMode = VK_POLYGON_MODE_FILL;
rasterization_state.cullMode = VK_CULL_MODE_NONE;
rasterization_state.frontFace = VK_FRONT_FACE_COUNTER_CLOCKWISE;
rasterization_state.lineWidth = 1.0f;
// For samples other than the first, will be changed for the pipelines for
// other samples.
VkSampleMask sample_mask = UINT32_MAX;
VkPipelineMultisampleStateCreateInfo multisample_state = {};
multisample_state.sType =
VK_STRUCTURE_TYPE_PIPELINE_MULTISAMPLE_STATE_CREATE_INFO;
multisample_state.rasterizationSamples =
(dest_sample_count == 2 && !msaa_2x_attachments_supported_)
? VK_SAMPLE_COUNT_4_BIT
: VkSampleCountFlagBits(dest_sample_count);
if (dest_sample_count > 1) {
if (device_features.sampleRateShading) {
multisample_state.sampleShadingEnable = VK_TRUE;
multisample_state.minSampleShading = 1.0f;
if (dest_sample_count == 2 && !msaa_2x_attachments_supported_) {
// Emulating 2x MSAA as samples 0 and 3 of 4x MSAA when 2x is not
// supported.
sample_mask = 0b1001;
}
} else {
sample_mask = 0b1;
}
if (sample_mask != UINT32_MAX) {
multisample_state.pSampleMask = &sample_mask;
}
}
// Whether the depth / stencil state is used depends on the presence of a
// depth attachment in the render pass - but not making assumptions about
// whether the render pass contains any specific attachments, so setting up
// valid depth / stencil state unconditionally.
VkPipelineDepthStencilStateCreateInfo depth_stencil_state = {};
depth_stencil_state.sType =
VK_STRUCTURE_TYPE_PIPELINE_DEPTH_STENCIL_STATE_CREATE_INFO;
if (mode.output == TransferOutput::kDepth) {
depth_stencil_state.depthTestEnable = VK_TRUE;
depth_stencil_state.depthWriteEnable = VK_TRUE;
depth_stencil_state.depthCompareOp = cvars::depth_transfer_not_equal_test
? VK_COMPARE_OP_NOT_EQUAL
: VK_COMPARE_OP_ALWAYS;
}
if ((mode.output == TransferOutput::kDepth &&
provider.device_extensions().ext_shader_stencil_export) ||
mode.output == TransferOutput::kStencilBit) {
depth_stencil_state.stencilTestEnable = VK_TRUE;
depth_stencil_state.front.failOp = VK_STENCIL_OP_KEEP;
depth_stencil_state.front.passOp = VK_STENCIL_OP_REPLACE;
depth_stencil_state.front.depthFailOp = VK_STENCIL_OP_REPLACE;
// Using ALWAYS, not NOT_EQUAL, so depth writing is unaffected by stencil
// being different.
depth_stencil_state.front.compareOp = VK_COMPARE_OP_ALWAYS;
// Will be dynamic for stencil bit output.
depth_stencil_state.front.writeMask = UINT8_MAX;
depth_stencil_state.front.reference = UINT8_MAX;
depth_stencil_state.back = depth_stencil_state.front;
}
// Whether the color blend state is used depends on the presence of color
// attachments in the render pass - but not making assumptions about whether
// the render pass contains any specific attachments, so setting up valid
// color blend state unconditionally.
VkPipelineColorBlendAttachmentState
color_blend_attachments[xenos::kMaxColorRenderTargets] = {};
VkPipelineColorBlendStateCreateInfo color_blend_state = {};
color_blend_state.sType =
VK_STRUCTURE_TYPE_PIPELINE_COLOR_BLEND_STATE_CREATE_INFO;
color_blend_state.attachmentCount =
32 - xe::lzcnt(key.render_pass_key.depth_and_color_used >> 1);
color_blend_state.pAttachments = color_blend_attachments;
if (mode.output == TransferOutput::kColor) {
if (device_features.independentBlend) {
// State the intention more explicitly.
color_blend_attachments[key.shader_key.dest_color_rt_index]
.colorWriteMask = VK_COLOR_COMPONENT_R_BIT |
VK_COLOR_COMPONENT_G_BIT |
VK_COLOR_COMPONENT_B_BIT | VK_COLOR_COMPONENT_A_BIT;
} else {
// The blend state for all attachments must be identical, but other render
// targets are not written to by the shader.
for (uint32_t i = 0; i < color_blend_state.attachmentCount; ++i) {
color_blend_attachments[i].colorWriteMask =
VK_COLOR_COMPONENT_R_BIT | VK_COLOR_COMPONENT_G_BIT |
VK_COLOR_COMPONENT_B_BIT | VK_COLOR_COMPONENT_A_BIT;
}
}
}
std::array<VkDynamicState, 3> dynamic_states;
VkPipelineDynamicStateCreateInfo dynamic_state;
dynamic_state.sType = VK_STRUCTURE_TYPE_PIPELINE_DYNAMIC_STATE_CREATE_INFO;
dynamic_state.pNext = nullptr;
dynamic_state.flags = 0;
dynamic_state.dynamicStateCount = 0;
dynamic_state.pDynamicStates = dynamic_states.data();
dynamic_states[dynamic_state.dynamicStateCount++] = VK_DYNAMIC_STATE_VIEWPORT;
dynamic_states[dynamic_state.dynamicStateCount++] = VK_DYNAMIC_STATE_SCISSOR;
if (mode.output == TransferOutput::kStencilBit) {
dynamic_states[dynamic_state.dynamicStateCount++] =
VK_DYNAMIC_STATE_STENCIL_WRITE_MASK;
}
std::array<VkPipeline, 4> pipelines{};
VkGraphicsPipelineCreateInfo pipeline_create_info;
pipeline_create_info.sType = VK_STRUCTURE_TYPE_GRAPHICS_PIPELINE_CREATE_INFO;
pipeline_create_info.pNext = nullptr;
pipeline_create_info.flags = 0;
if (dest_is_masked_sample) {
pipeline_create_info.flags |= VK_PIPELINE_CREATE_ALLOW_DERIVATIVES_BIT;
}
pipeline_create_info.stageCount = uint32_t(xe::countof(shader_stages));
pipeline_create_info.pStages = shader_stages;
pipeline_create_info.pVertexInputState = &vertex_input_state;
pipeline_create_info.pInputAssemblyState = &input_assembly_state;
pipeline_create_info.pTessellationState = nullptr;
pipeline_create_info.pViewportState = &viewport_state;
pipeline_create_info.pRasterizationState = &rasterization_state;
pipeline_create_info.pMultisampleState = &multisample_state;
pipeline_create_info.pDepthStencilState = &depth_stencil_state;
pipeline_create_info.pColorBlendState = &color_blend_state;
pipeline_create_info.pDynamicState = &dynamic_state;
pipeline_create_info.layout =
transfer_pipeline_layouts_[size_t(mode.pipeline_layout)];
pipeline_create_info.renderPass = render_pass;
pipeline_create_info.subpass = 0;
pipeline_create_info.basePipelineHandle = VK_NULL_HANDLE;
pipeline_create_info.basePipelineIndex = -1;
if (dfn.vkCreateGraphicsPipelines(device, VK_NULL_HANDLE, 1,
&pipeline_create_info, nullptr,
&pipelines[0]) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the render target ownership "
"transfer pipeline for render pass 0x{:08X}, shader 0x{:08X}",
key.render_pass_key.key, key.shader_key.key);
transfer_pipelines_.emplace(key, std::array<VkPipeline, 4>{});
return nullptr;
}
if (dest_is_masked_sample) {
assert_true(multisample_state.pSampleMask == &sample_mask);
pipeline_create_info.flags = (pipeline_create_info.flags &
~VK_PIPELINE_CREATE_ALLOW_DERIVATIVES_BIT) |
VK_PIPELINE_CREATE_DERIVATIVE_BIT;
pipeline_create_info.basePipelineHandle = pipelines[0];
for (uint32_t i = 1; i < dest_sample_count; ++i) {
// Emulating 2x MSAA as samples 0 and 3 of 4x MSAA when 2x is not
// supported.
uint32_t host_sample_index =
(dest_sample_count == 2 && !msaa_2x_attachments_supported_ && i == 1)
? 3
: i;
sample_id_specialization_constant = host_sample_index;
sample_mask = uint32_t(1) << host_sample_index;
if (dfn.vkCreateGraphicsPipelines(device, VK_NULL_HANDLE, 1,
&pipeline_create_info, nullptr,
&pipelines[i]) != VK_SUCCESS) {
XELOGE(
"VulkanRenderTargetCache: Failed to create the render target "
"ownership transfer pipeline for render pass 0x{:08X}, shader "
"0x{:08X}, sample {}",
key.render_pass_key.key, key.shader_key.key, i);
for (uint32_t j = 0; j < i; ++j) {
dfn.vkDestroyPipeline(device, pipelines[j], nullptr);
}
transfer_pipelines_.emplace(key, std::array<VkPipeline, 4>{});
return nullptr;
}
}
}
return transfer_pipelines_.emplace(key, pipelines).first->second.data();
}
void VulkanRenderTargetCache::PerformTransfersAndResolveClears(
uint32_t render_target_count, RenderTarget* const* render_targets,
const std::vector<Transfer>* render_target_transfers,
const uint64_t* render_target_resolve_clear_values,
const Transfer::Rectangle* resolve_clear_rectangle) {
assert_true(GetPath() == Path::kHostRenderTargets);
const ui::vulkan::VulkanProvider& provider =
command_processor_.GetVulkanProvider();
const VkPhysicalDeviceLimits& device_limits =
provider.device_properties().limits;
const VkPhysicalDeviceFeatures& device_features = provider.device_features();
bool shader_stencil_export =
provider.device_extensions().ext_shader_stencil_export;
uint64_t current_submission = command_processor_.GetCurrentSubmission();
DeferredCommandBuffer& command_buffer =
command_processor_.deferred_command_buffer();
bool resolve_clear_needed =
render_target_resolve_clear_values && resolve_clear_rectangle;
VkClearRect resolve_clear_rect;
if (resolve_clear_needed) {
// Assuming the rectangle is already clamped by the setup function from the
// common render target cache.
resolve_clear_rect.rect.offset.x =
int32_t(resolve_clear_rectangle->x_pixels * draw_resolution_scale_x());
resolve_clear_rect.rect.offset.y =
int32_t(resolve_clear_rectangle->y_pixels * draw_resolution_scale_y());
resolve_clear_rect.rect.extent.width =
resolve_clear_rectangle->width_pixels * draw_resolution_scale_x();
resolve_clear_rect.rect.extent.height =
resolve_clear_rectangle->height_pixels * draw_resolution_scale_y();
resolve_clear_rect.baseArrayLayer = 0;
resolve_clear_rect.layerCount = 1;
}
// Do host depth storing for the depth destination (assuming there can be only
// one depth destination) where depth destination == host depth source.
bool host_depth_store_set_up = false;
for (uint32_t i = 0; i < render_target_count; ++i) {
RenderTarget* dest_rt = render_targets[i];
if (!dest_rt) {
continue;
}
auto& dest_vulkan_rt = *static_cast<VulkanRenderTarget*>(dest_rt);
RenderTargetKey dest_rt_key = dest_vulkan_rt.key();
if (!dest_rt_key.is_depth) {
continue;
}
const std::vector<Transfer>& depth_transfers = render_target_transfers[i];
for (const Transfer& transfer : depth_transfers) {
if (transfer.host_depth_source != dest_rt) {
continue;
}
if (!host_depth_store_set_up) {
// Pipeline.
command_processor_.BindExternalComputePipeline(
host_depth_store_pipelines_[size_t(dest_rt_key.msaa_samples)]);
// Descriptor set bindings.
VkDescriptorSet host_depth_store_descriptor_sets[] = {
edram_storage_buffer_descriptor_set_,
dest_vulkan_rt.GetDescriptorSetTransferSource(),
};
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_COMPUTE, host_depth_store_pipeline_layout_,
0, uint32_t(xe::countof(host_depth_store_descriptor_sets)),
host_depth_store_descriptor_sets, 0, nullptr);
// Render target constant.
HostDepthStoreRenderTargetConstant
host_depth_store_render_target_constant =
GetHostDepthStoreRenderTargetConstant(
dest_rt_key.pitch_tiles_at_32bpp,
msaa_2x_attachments_supported_);
command_buffer.CmdVkPushConstants(
host_depth_store_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT,
uint32_t(offsetof(HostDepthStoreConstants, render_target)),
sizeof(host_depth_store_render_target_constant),
&host_depth_store_render_target_constant);
// Barriers - don't need to try to combine them with the rest of
// render target transfer barriers now - if this happens, after host
// depth storing, SHADER_READ -> DEPTH_STENCIL_ATTACHMENT_WRITE will be
// done anyway even in the best case, so it's not possible to have all
// the barriers in one place here.
UseEdramBuffer(EdramBufferUsage::kComputeWrite);
// Always transitioning both depth and stencil, not storing separate
// usage flags for depth and stencil.
command_processor_.PushImageMemoryBarrier(
dest_vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT),
dest_vulkan_rt.current_stage_mask(),
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
dest_vulkan_rt.current_access_mask(), VK_ACCESS_SHADER_READ_BIT,
dest_vulkan_rt.current_layout(),
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL);
dest_vulkan_rt.SetUsage(VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
VK_ACCESS_SHADER_READ_BIT,
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL);
host_depth_store_set_up = true;
}
Transfer::Rectangle
transfer_rectangles[Transfer::kMaxRectanglesWithCutout];
uint32_t transfer_rectangle_count = transfer.GetRectangles(
dest_rt_key.base_tiles, dest_rt_key.pitch_tiles_at_32bpp,
dest_rt_key.msaa_samples, false, transfer_rectangles,
resolve_clear_rectangle);
assert_not_zero(transfer_rectangle_count);
HostDepthStoreRectangleConstant host_depth_store_rectangle_constant;
for (uint32_t j = 0; j < transfer_rectangle_count; ++j) {
uint32_t group_count_x, group_count_y;
GetHostDepthStoreRectangleInfo(
transfer_rectangles[j], dest_rt_key.msaa_samples,
host_depth_store_rectangle_constant, group_count_x, group_count_y);
command_buffer.CmdVkPushConstants(
host_depth_store_pipeline_layout_, VK_SHADER_STAGE_COMPUTE_BIT,
uint32_t(offsetof(HostDepthStoreConstants, rectangle)),
sizeof(host_depth_store_rectangle_constant),
&host_depth_store_rectangle_constant);
command_processor_.SubmitBarriers(true);
command_buffer.CmdVkDispatch(group_count_x, group_count_y, 1);
MarkEdramBufferModified();
}
}
break;
}
constexpr VkPipelineStageFlags kSourceStageMask =
VK_PIPELINE_STAGE_FRAGMENT_SHADER_BIT;
constexpr VkAccessFlags kSourceAccessMask = VK_ACCESS_SHADER_READ_BIT;
constexpr VkImageLayout kSourceLayout =
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
// Try to insert as many barriers as possible in one place, hoping that in the
// best case (no cross-copying between current render targets), barriers will
// need to be only inserted here, not between transfers. In case of
// cross-copying, if the destination use is going to happen before the source
// use, choose the destination state, otherwise the source state - to match
// the order in which transfers will actually happen (otherwise there will be
// just a useless switch back and forth).
for (uint32_t i = 0; i < render_target_count; ++i) {
RenderTarget* dest_rt = render_targets[i];
if (!dest_rt) {
continue;
}
const std::vector<Transfer>& dest_transfers = render_target_transfers[i];
if (!resolve_clear_needed && dest_transfers.empty()) {
continue;
}
// Transition the destination, only if not going to be used as a source
// earlier.
bool dest_used_previously_as_source = false;
for (uint32_t j = 0; j < i; ++j) {
for (const Transfer& previous_transfer : render_target_transfers[j]) {
if (previous_transfer.source == dest_rt ||
previous_transfer.host_depth_source == dest_rt) {
dest_used_previously_as_source = true;
break;
}
}
}
if (!dest_used_previously_as_source) {
auto& dest_vulkan_rt = *static_cast<VulkanRenderTarget*>(dest_rt);
VkPipelineStageFlags dest_dst_stage_mask;
VkAccessFlags dest_dst_access_mask;
VkImageLayout dest_new_layout;
dest_vulkan_rt.GetDrawUsage(&dest_dst_stage_mask, &dest_dst_access_mask,
&dest_new_layout);
command_processor_.PushImageMemoryBarrier(
dest_vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
dest_vulkan_rt.key().is_depth
? (VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)
: VK_IMAGE_ASPECT_COLOR_BIT),
dest_vulkan_rt.current_stage_mask(), dest_dst_stage_mask,
dest_vulkan_rt.current_access_mask(), dest_dst_access_mask,
dest_vulkan_rt.current_layout(), dest_new_layout);
dest_vulkan_rt.SetUsage(dest_dst_stage_mask, dest_dst_access_mask,
dest_new_layout);
}
// Transition the sources, only if not going to be used as destinations
// earlier.
for (const Transfer& transfer : dest_transfers) {
bool source_previously_used_as_dest = false;
bool host_depth_source_previously_used_as_dest = false;
for (uint32_t j = 0; j < i; ++j) {
if (render_target_transfers[j].empty()) {
continue;
}
const RenderTarget* previous_rt = render_targets[j];
if (transfer.source == previous_rt) {
source_previously_used_as_dest = true;
}
if (transfer.host_depth_source == previous_rt) {
host_depth_source_previously_used_as_dest = true;
}
}
if (!source_previously_used_as_dest) {
auto& source_vulkan_rt =
*static_cast<VulkanRenderTarget*>(transfer.source);
command_processor_.PushImageMemoryBarrier(
source_vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
source_vulkan_rt.key().is_depth
? (VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)
: VK_IMAGE_ASPECT_COLOR_BIT),
source_vulkan_rt.current_stage_mask(), kSourceStageMask,
source_vulkan_rt.current_access_mask(), kSourceAccessMask,
source_vulkan_rt.current_layout(), kSourceLayout);
source_vulkan_rt.SetUsage(kSourceStageMask, kSourceAccessMask,
kSourceLayout);
}
// transfer.host_depth_source == dest_rt means the EDRAM buffer will be
// used instead, no need to transition.
if (transfer.host_depth_source && transfer.host_depth_source != dest_rt &&
!host_depth_source_previously_used_as_dest) {
auto& host_depth_source_vulkan_rt =
*static_cast<VulkanRenderTarget*>(transfer.host_depth_source);
command_processor_.PushImageMemoryBarrier(
host_depth_source_vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT),
host_depth_source_vulkan_rt.current_stage_mask(), kSourceStageMask,
host_depth_source_vulkan_rt.current_access_mask(),
kSourceAccessMask, host_depth_source_vulkan_rt.current_layout(),
kSourceLayout);
host_depth_source_vulkan_rt.SetUsage(kSourceStageMask,
kSourceAccessMask, kSourceLayout);
}
}
}
if (host_depth_store_set_up) {
// Will be reading copied host depth from the EDRAM buffer.
UseEdramBuffer(EdramBufferUsage::kFragmentRead);
}
// Perform the transfers and clears.
TransferPipelineLayoutIndex last_transfer_pipeline_layout_index =
TransferPipelineLayoutIndex::kCount;
uint32_t transfer_descriptor_sets_bound = 0;
uint32_t transfer_push_constants_set = 0;
VkDescriptorSet last_descriptor_set_host_depth_stencil_textures =
VK_NULL_HANDLE;
VkDescriptorSet last_descriptor_set_depth_stencil_textures = VK_NULL_HANDLE;
VkDescriptorSet last_descriptor_set_color_texture = VK_NULL_HANDLE;
TransferAddressConstant last_host_depth_address_constant;
TransferAddressConstant last_address_constant;
for (uint32_t i = 0; i < render_target_count; ++i) {
RenderTarget* dest_rt = render_targets[i];
if (!dest_rt) {
continue;
}
const std::vector<Transfer>& current_transfers = render_target_transfers[i];
if (current_transfers.empty() && !resolve_clear_needed) {
continue;
}
auto& dest_vulkan_rt = *static_cast<VulkanRenderTarget*>(dest_rt);
RenderTargetKey dest_rt_key = dest_vulkan_rt.key();
// Late barriers in case there was cross-copying that prevented merging of
// barriers.
{
VkPipelineStageFlags dest_dst_stage_mask;
VkAccessFlags dest_dst_access_mask;
VkImageLayout dest_new_layout;
dest_vulkan_rt.GetDrawUsage(&dest_dst_stage_mask, &dest_dst_access_mask,
&dest_new_layout);
command_processor_.PushImageMemoryBarrier(
dest_vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
dest_rt_key.is_depth
? (VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)
: VK_IMAGE_ASPECT_COLOR_BIT),
dest_vulkan_rt.current_stage_mask(), dest_dst_stage_mask,
dest_vulkan_rt.current_access_mask(), dest_dst_access_mask,
dest_vulkan_rt.current_layout(), dest_new_layout);
dest_vulkan_rt.SetUsage(dest_dst_stage_mask, dest_dst_access_mask,
dest_new_layout);
}
// Get the objects needed for transfers to the destination.
// TODO(Triang3l): Reuse the guest render pass for transfers where possible
// (if the Vulkan format used for drawing is also usable for transfers - for
// instance, R8G8B8A8_UNORM can be used for both, so the guest pass can be
// reused, but R16G16B16A16_SFLOAT render targets use R16G16B16A16_UINT for
// transfers, so the transfer pass has to be separate) to avoid stores and
// loads on tile-based devices to make this actually applicable. Also
// overall perform all non-cross-copying transfers for the current
// framebuffer configuration in a single pass, to load / store only once.
RenderPassKey transfer_render_pass_key;
transfer_render_pass_key.msaa_samples = dest_rt_key.msaa_samples;
if (dest_rt_key.is_depth) {
transfer_render_pass_key.depth_and_color_used = 0b1;
transfer_render_pass_key.depth_format = dest_rt_key.GetDepthFormat();
} else {
transfer_render_pass_key.depth_and_color_used = 0b1 << 1;
transfer_render_pass_key.color_0_view_format =
dest_rt_key.GetColorFormat();
transfer_render_pass_key.color_rts_use_transfer_formats = 1;
}
VkRenderPass transfer_render_pass =
GetHostRenderTargetsRenderPass(transfer_render_pass_key);
if (transfer_render_pass == VK_NULL_HANDLE) {
continue;
}
const RenderTarget*
transfer_framebuffer_render_targets[1 + xenos::kMaxColorRenderTargets] =
{};
transfer_framebuffer_render_targets[dest_rt_key.is_depth ? 0 : 1] = dest_rt;
const Framebuffer* transfer_framebuffer = GetHostRenderTargetsFramebuffer(
transfer_render_pass_key, dest_rt_key.pitch_tiles_at_32bpp,
transfer_framebuffer_render_targets);
if (!transfer_framebuffer) {
continue;
}
// Don't enter the render pass immediately - may still insert source
// barriers later.
if (!current_transfers.empty()) {
uint32_t dest_pitch_tiles = dest_rt_key.GetPitchTiles();
bool dest_is_64bpp = dest_rt_key.Is64bpp();
// Gather shader keys and sort to reduce pipeline state and binding
// switches. Also gather stencil rectangles to clear if needed.
bool need_stencil_bit_draws =
dest_rt_key.is_depth && !shader_stencil_export;
current_transfer_invocations_.clear();
current_transfer_invocations_.reserve(
current_transfers.size() << uint32_t(need_stencil_bit_draws));
uint32_t rt_sort_index = 0;
TransferShaderKey new_transfer_shader_key;
new_transfer_shader_key.dest_msaa_samples = dest_rt_key.msaa_samples;
new_transfer_shader_key.dest_resource_format =
dest_rt_key.resource_format;
uint32_t stencil_clear_rectangle_count = 0;
for (uint32_t j = 0; j <= uint32_t(need_stencil_bit_draws); ++j) {
// j == 0 - color or depth.
// j == 1 - stencil bits.
// Stencil bit writing always requires a different root signature,
// handle these separately. Stencil never has a host depth source.
// Clear previously set sort indices.
for (const Transfer& transfer : current_transfers) {
auto host_depth_source_vulkan_rt =
static_cast<VulkanRenderTarget*>(transfer.host_depth_source);
if (host_depth_source_vulkan_rt) {
host_depth_source_vulkan_rt->SetTemporarySortIndex(UINT32_MAX);
}
assert_not_null(transfer.source);
auto& source_vulkan_rt =
*static_cast<VulkanRenderTarget*>(transfer.source);
source_vulkan_rt.SetTemporarySortIndex(UINT32_MAX);
}
for (const Transfer& transfer : current_transfers) {
assert_not_null(transfer.source);
auto& source_vulkan_rt =
*static_cast<VulkanRenderTarget*>(transfer.source);
VulkanRenderTarget* host_depth_source_vulkan_rt =
j ? nullptr
: static_cast<VulkanRenderTarget*>(transfer.host_depth_source);
if (host_depth_source_vulkan_rt &&
host_depth_source_vulkan_rt->temporary_sort_index() ==
UINT32_MAX) {
host_depth_source_vulkan_rt->SetTemporarySortIndex(rt_sort_index++);
}
if (source_vulkan_rt.temporary_sort_index() == UINT32_MAX) {
source_vulkan_rt.SetTemporarySortIndex(rt_sort_index++);
}
RenderTargetKey source_rt_key = source_vulkan_rt.key();
new_transfer_shader_key.source_msaa_samples =
source_rt_key.msaa_samples;
new_transfer_shader_key.source_resource_format =
source_rt_key.resource_format;
bool host_depth_source_is_copy =
host_depth_source_vulkan_rt == &dest_vulkan_rt;
// The host depth copy buffer has only raw samples.
new_transfer_shader_key.host_depth_source_msaa_samples =
(host_depth_source_vulkan_rt && !host_depth_source_is_copy)
? host_depth_source_vulkan_rt->key().msaa_samples
: xenos::MsaaSamples::k1X;
if (j) {
new_transfer_shader_key.mode =
source_rt_key.is_depth ? TransferMode::kDepthToStencilBit
: TransferMode::kColorToStencilBit;
stencil_clear_rectangle_count +=
transfer.GetRectangles(dest_rt_key.base_tiles, dest_pitch_tiles,
dest_rt_key.msaa_samples, dest_is_64bpp,
nullptr, resolve_clear_rectangle);
} else {
if (dest_rt_key.is_depth) {
if (host_depth_source_vulkan_rt) {
if (host_depth_source_is_copy) {
new_transfer_shader_key.mode =
source_rt_key.is_depth
? TransferMode::kDepthAndHostDepthCopyToDepth
: TransferMode::kColorAndHostDepthCopyToDepth;
} else {
new_transfer_shader_key.mode =
source_rt_key.is_depth
? TransferMode::kDepthAndHostDepthToDepth
: TransferMode::kColorAndHostDepthToDepth;
}
} else {
new_transfer_shader_key.mode =
source_rt_key.is_depth ? TransferMode::kDepthToDepth
: TransferMode::kColorToDepth;
}
} else {
new_transfer_shader_key.mode = source_rt_key.is_depth
? TransferMode::kDepthToColor
: TransferMode::kColorToColor;
}
}
current_transfer_invocations_.emplace_back(transfer,
new_transfer_shader_key);
if (j) {
current_transfer_invocations_.back().transfer.host_depth_source =
nullptr;
}
}
}
std::sort(current_transfer_invocations_.begin(),
current_transfer_invocations_.end());
for (auto it = current_transfer_invocations_.cbegin();
it != current_transfer_invocations_.cend(); ++it) {
assert_not_null(it->transfer.source);
auto& source_vulkan_rt =
*static_cast<VulkanRenderTarget*>(it->transfer.source);
command_processor_.PushImageMemoryBarrier(
source_vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
source_vulkan_rt.key().is_depth
? (VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)
: VK_IMAGE_ASPECT_COLOR_BIT),
source_vulkan_rt.current_stage_mask(), kSourceStageMask,
source_vulkan_rt.current_access_mask(), kSourceAccessMask,
source_vulkan_rt.current_layout(), kSourceLayout);
source_vulkan_rt.SetUsage(kSourceStageMask, kSourceAccessMask,
kSourceLayout);
auto host_depth_source_vulkan_rt =
static_cast<VulkanRenderTarget*>(it->transfer.host_depth_source);
if (host_depth_source_vulkan_rt) {
TransferShaderKey transfer_shader_key = it->shader_key;
if (transfer_shader_key.mode ==
TransferMode::kDepthAndHostDepthCopyToDepth ||
transfer_shader_key.mode ==
TransferMode::kColorAndHostDepthCopyToDepth) {
// Reading copied host depth from the EDRAM buffer.
UseEdramBuffer(EdramBufferUsage::kFragmentRead);
} else {
// Reading host depth from the texture.
command_processor_.PushImageMemoryBarrier(
host_depth_source_vulkan_rt->image(),
ui::vulkan::util::InitializeSubresourceRange(
VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT),
host_depth_source_vulkan_rt->current_stage_mask(),
kSourceStageMask,
host_depth_source_vulkan_rt->current_access_mask(),
kSourceAccessMask,
host_depth_source_vulkan_rt->current_layout(), kSourceLayout);
host_depth_source_vulkan_rt->SetUsage(
kSourceStageMask, kSourceAccessMask, kSourceLayout);
}
}
}
// Perform the transfers for the render target.
command_processor_.SubmitBarriersAndEnterRenderTargetCacheRenderPass(
transfer_render_pass, transfer_framebuffer);
if (stencil_clear_rectangle_count) {
VkClearAttachment* stencil_clear_attachment;
VkClearRect* stencil_clear_rect_write_ptr;
command_buffer.CmdClearAttachmentsEmplace(1, stencil_clear_attachment,
stencil_clear_rectangle_count,
stencil_clear_rect_write_ptr);
stencil_clear_attachment->aspectMask = VK_IMAGE_ASPECT_STENCIL_BIT;
stencil_clear_attachment->colorAttachment = 0;
stencil_clear_attachment->clearValue.depthStencil.depth = 0.0f;
stencil_clear_attachment->clearValue.depthStencil.stencil = 0;
for (const Transfer& transfer : current_transfers) {
Transfer::Rectangle transfer_stencil_clear_rectangles
[Transfer::kMaxRectanglesWithCutout];
uint32_t transfer_stencil_clear_rectangle_count =
transfer.GetRectangles(dest_rt_key.base_tiles, dest_pitch_tiles,
dest_rt_key.msaa_samples, dest_is_64bpp,
transfer_stencil_clear_rectangles,
resolve_clear_rectangle);
for (uint32_t j = 0; j < transfer_stencil_clear_rectangle_count;
++j) {
const Transfer::Rectangle& stencil_clear_rectangle =
transfer_stencil_clear_rectangles[j];
stencil_clear_rect_write_ptr->rect.offset.x = int32_t(
stencil_clear_rectangle.x_pixels * draw_resolution_scale_x());
stencil_clear_rect_write_ptr->rect.offset.y = int32_t(
stencil_clear_rectangle.y_pixels * draw_resolution_scale_y());
stencil_clear_rect_write_ptr->rect.extent.width =
stencil_clear_rectangle.width_pixels *
draw_resolution_scale_x();
stencil_clear_rect_write_ptr->rect.extent.height =
stencil_clear_rectangle.height_pixels *
draw_resolution_scale_y();
stencil_clear_rect_write_ptr->baseArrayLayer = 0;
stencil_clear_rect_write_ptr->layerCount = 1;
++stencil_clear_rect_write_ptr;
}
}
}
// Prefer power of two viewports for exact division by simply biasing the
// exponent.
VkViewport transfer_viewport;
transfer_viewport.x = 0.0f;
transfer_viewport.y = 0.0f;
transfer_viewport.width =
float(std::min(xe::next_pow2(transfer_framebuffer->host_extent.width),
device_limits.maxViewportDimensions[0]));
transfer_viewport.height = float(
std::min(xe::next_pow2(transfer_framebuffer->host_extent.height),
device_limits.maxViewportDimensions[1]));
transfer_viewport.minDepth = 0.0f;
transfer_viewport.maxDepth = 1.0f;
command_processor_.SetViewport(transfer_viewport);
float pixels_to_ndc_x = 2.0f / transfer_viewport.width;
float pixels_to_ndc_y = 2.0f / transfer_viewport.height;
VkRect2D transfer_scissor;
transfer_scissor.offset.x = 0;
transfer_scissor.offset.y = 0;
transfer_scissor.extent = transfer_framebuffer->host_extent;
command_processor_.SetScissor(transfer_scissor);
for (auto it = current_transfer_invocations_.cbegin();
it != current_transfer_invocations_.cend(); ++it) {
const TransferInvocation& transfer_invocation_first = *it;
// Will be merging transfers from the same source into one mesh.
auto it_merged_first = it, it_merged_last = it;
uint32_t transfer_rectangle_count =
transfer_invocation_first.transfer.GetRectangles(
dest_rt_key.base_tiles, dest_pitch_tiles,
dest_rt_key.msaa_samples, dest_is_64bpp, nullptr,
resolve_clear_rectangle);
for (auto it_merge = std::next(it_merged_first);
it_merge != current_transfer_invocations_.cend(); ++it_merge) {
if (!transfer_invocation_first.CanBeMergedIntoOneDraw(*it_merge)) {
break;
}
transfer_rectangle_count += it_merge->transfer.GetRectangles(
dest_rt_key.base_tiles, dest_pitch_tiles,
dest_rt_key.msaa_samples, dest_is_64bpp, nullptr,
resolve_clear_rectangle);
it_merged_last = it_merge;
}
assert_not_zero(transfer_rectangle_count);
// Skip the merged transfers in the subsequent iterations.
it = it_merged_last;
assert_not_null(it->transfer.source);
auto& source_vulkan_rt =
*static_cast<VulkanRenderTarget*>(it->transfer.source);
auto host_depth_source_vulkan_rt =
static_cast<VulkanRenderTarget*>(it->transfer.host_depth_source);
TransferShaderKey transfer_shader_key = it->shader_key;
const TransferModeInfo& transfer_mode_info =
kTransferModes[size_t(transfer_shader_key.mode)];
TransferPipelineLayoutIndex transfer_pipeline_layout_index =
transfer_mode_info.pipeline_layout;
const TransferPipelineLayoutInfo& transfer_pipeline_layout_info =
kTransferPipelineLayoutInfos[size_t(
transfer_pipeline_layout_index)];
uint32_t transfer_sample_pipeline_count =
device_features.sampleRateShading
? 1
: uint32_t(1) << uint32_t(dest_rt_key.msaa_samples);
bool transfer_is_stencil_bit =
(transfer_pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordStencilMaskBit) != 0;
uint32_t transfer_vertex_count = 6 * transfer_rectangle_count;
VkBuffer transfer_vertex_buffer;
VkDeviceSize transfer_vertex_buffer_offset;
float* transfer_rectangle_write_ptr =
reinterpret_cast<float*>(transfer_vertex_buffer_pool_->Request(
current_submission, sizeof(float) * 2 * transfer_vertex_count,
sizeof(float), transfer_vertex_buffer,
transfer_vertex_buffer_offset));
if (!transfer_rectangle_write_ptr) {
continue;
}
for (auto it_merged = it_merged_first; it_merged <= it_merged_last;
++it_merged) {
Transfer::Rectangle transfer_invocation_rectangles
[Transfer::kMaxRectanglesWithCutout];
uint32_t transfer_invocation_rectangle_count =
it_merged->transfer.GetRectangles(
dest_rt_key.base_tiles, dest_pitch_tiles,
dest_rt_key.msaa_samples, dest_is_64bpp,
transfer_invocation_rectangles, resolve_clear_rectangle);
assert_not_zero(transfer_invocation_rectangle_count);
for (uint32_t j = 0; j < transfer_invocation_rectangle_count; ++j) {
const Transfer::Rectangle& transfer_rectangle =
transfer_invocation_rectangles[j];
float transfer_rectangle_x0 =
-1.0f + transfer_rectangle.x_pixels * pixels_to_ndc_x;
float transfer_rectangle_y0 =
-1.0f + transfer_rectangle.y_pixels * pixels_to_ndc_y;
float transfer_rectangle_x1 =
transfer_rectangle_x0 +
transfer_rectangle.width_pixels * pixels_to_ndc_x;
float transfer_rectangle_y1 =
transfer_rectangle_y0 +
transfer_rectangle.height_pixels * pixels_to_ndc_y;
// O-*
// |/
// *
*(transfer_rectangle_write_ptr++) = transfer_rectangle_x0;
*(transfer_rectangle_write_ptr++) = transfer_rectangle_y0;
// *-*
// |/
// O
*(transfer_rectangle_write_ptr++) = transfer_rectangle_x0;
*(transfer_rectangle_write_ptr++) = transfer_rectangle_y1;
// *-O
// |/
// *
*(transfer_rectangle_write_ptr++) = transfer_rectangle_x1;
*(transfer_rectangle_write_ptr++) = transfer_rectangle_y0;
// O
// /|
// *-*
*(transfer_rectangle_write_ptr++) = transfer_rectangle_x1;
*(transfer_rectangle_write_ptr++) = transfer_rectangle_y0;
// *
// /|
// O-*
*(transfer_rectangle_write_ptr++) = transfer_rectangle_x0;
*(transfer_rectangle_write_ptr++) = transfer_rectangle_y1;
// *
// /|
// *-O
*(transfer_rectangle_write_ptr++) = transfer_rectangle_x1;
*(transfer_rectangle_write_ptr++) = transfer_rectangle_y1;
}
}
command_buffer.CmdVkBindVertexBuffers(0, 1, &transfer_vertex_buffer,
&transfer_vertex_buffer_offset);
const VkPipeline* transfer_pipelines = GetTransferPipelines(
TransferPipelineKey(transfer_render_pass_key, transfer_shader_key));
if (!transfer_pipelines) {
continue;
}
command_processor_.BindExternalGraphicsPipeline(transfer_pipelines[0]);
if (last_transfer_pipeline_layout_index !=
transfer_pipeline_layout_index) {
last_transfer_pipeline_layout_index = transfer_pipeline_layout_index;
transfer_descriptor_sets_bound = 0;
transfer_push_constants_set = 0;
}
// Invalidate outdated bindings.
if (transfer_pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetHostDepthStencilTexturesBit) {
assert_not_null(host_depth_source_vulkan_rt);
VkDescriptorSet descriptor_set_host_depth_stencil_textures =
host_depth_source_vulkan_rt->GetDescriptorSetTransferSource();
if (last_descriptor_set_host_depth_stencil_textures !=
descriptor_set_host_depth_stencil_textures) {
last_descriptor_set_host_depth_stencil_textures =
descriptor_set_host_depth_stencil_textures;
transfer_descriptor_sets_bound &=
~kTransferUsedDescriptorSetHostDepthStencilTexturesBit;
}
}
if (transfer_pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetDepthStencilTexturesBit) {
VkDescriptorSet descriptor_set_depth_stencil_textures =
source_vulkan_rt.GetDescriptorSetTransferSource();
if (last_descriptor_set_depth_stencil_textures !=
descriptor_set_depth_stencil_textures) {
last_descriptor_set_depth_stencil_textures =
descriptor_set_depth_stencil_textures;
transfer_descriptor_sets_bound &=
~kTransferUsedDescriptorSetDepthStencilTexturesBit;
}
}
if (transfer_pipeline_layout_info.used_descriptor_sets &
kTransferUsedDescriptorSetColorTextureBit) {
VkDescriptorSet descriptor_set_color_texture =
source_vulkan_rt.GetDescriptorSetTransferSource();
if (last_descriptor_set_color_texture !=
descriptor_set_color_texture) {
last_descriptor_set_color_texture = descriptor_set_color_texture;
transfer_descriptor_sets_bound &=
~kTransferUsedDescriptorSetColorTextureBit;
}
}
if (transfer_pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordHostDepthAddressBit) {
assert_not_null(host_depth_source_vulkan_rt);
RenderTargetKey host_depth_source_rt_key =
host_depth_source_vulkan_rt->key();
TransferAddressConstant host_depth_address_constant;
host_depth_address_constant.dest_pitch = dest_pitch_tiles;
host_depth_address_constant.source_pitch =
host_depth_source_rt_key.GetPitchTiles();
host_depth_address_constant.source_to_dest =
int32_t(dest_rt_key.base_tiles) -
int32_t(host_depth_source_rt_key.base_tiles);
if (last_host_depth_address_constant != host_depth_address_constant) {
last_host_depth_address_constant = host_depth_address_constant;
transfer_push_constants_set &=
~kTransferUsedPushConstantDwordHostDepthAddressBit;
}
}
if (transfer_pipeline_layout_info.used_push_constant_dwords &
kTransferUsedPushConstantDwordAddressBit) {
RenderTargetKey source_rt_key = source_vulkan_rt.key();
TransferAddressConstant address_constant;
address_constant.dest_pitch = dest_pitch_tiles;
address_constant.source_pitch = source_rt_key.GetPitchTiles();
address_constant.source_to_dest = int32_t(dest_rt_key.base_tiles) -
int32_t(source_rt_key.base_tiles);
if (last_address_constant != address_constant) {
last_address_constant = address_constant;
transfer_push_constants_set &=
~kTransferUsedPushConstantDwordAddressBit;
}
}
// Apply the new bindings.
// TODO(Triang3l): Merge binding updates into spans.
VkPipelineLayout transfer_pipeline_layout =
transfer_pipeline_layouts_[size_t(transfer_pipeline_layout_index)];
uint32_t transfer_descriptor_sets_unbound =
transfer_pipeline_layout_info.used_descriptor_sets &
~transfer_descriptor_sets_bound;
if (transfer_descriptor_sets_unbound &
kTransferUsedDescriptorSetHostDepthBufferBit) {
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_GRAPHICS, transfer_pipeline_layout,
xe::bit_count(transfer_pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetHostDepthBufferBit - 1)),
1, &edram_storage_buffer_descriptor_set_, 0, nullptr);
transfer_descriptor_sets_bound |=
kTransferUsedDescriptorSetHostDepthBufferBit;
}
if (transfer_descriptor_sets_unbound &
kTransferUsedDescriptorSetHostDepthStencilTexturesBit) {
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_GRAPHICS, transfer_pipeline_layout,
xe::bit_count(
transfer_pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetHostDepthStencilTexturesBit - 1)),
1, &last_descriptor_set_host_depth_stencil_textures, 0, nullptr);
transfer_descriptor_sets_bound |=
kTransferUsedDescriptorSetHostDepthStencilTexturesBit;
}
if (transfer_descriptor_sets_unbound &
kTransferUsedDescriptorSetDepthStencilTexturesBit) {
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_GRAPHICS, transfer_pipeline_layout,
xe::bit_count(
transfer_pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetDepthStencilTexturesBit - 1)),
1, &last_descriptor_set_depth_stencil_textures, 0, nullptr);
transfer_descriptor_sets_bound |=
kTransferUsedDescriptorSetDepthStencilTexturesBit;
}
if (transfer_descriptor_sets_unbound &
kTransferUsedDescriptorSetColorTextureBit) {
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_GRAPHICS, transfer_pipeline_layout,
xe::bit_count(transfer_pipeline_layout_info.used_descriptor_sets &
(kTransferUsedDescriptorSetColorTextureBit - 1)),
1, &last_descriptor_set_color_texture, 0, nullptr);
transfer_descriptor_sets_bound |=
kTransferUsedDescriptorSetColorTextureBit;
}
uint32_t transfer_push_constants_unset =
transfer_pipeline_layout_info.used_push_constant_dwords &
~transfer_push_constants_set;
if (transfer_push_constants_unset &
kTransferUsedPushConstantDwordHostDepthAddressBit) {
command_buffer.CmdVkPushConstants(
transfer_pipeline_layout, VK_SHADER_STAGE_FRAGMENT_BIT,
sizeof(uint32_t) *
xe::bit_count(
transfer_pipeline_layout_info.used_push_constant_dwords &
(kTransferUsedPushConstantDwordHostDepthAddressBit - 1)),
sizeof(uint32_t), &last_host_depth_address_constant);
transfer_push_constants_set |=
kTransferUsedPushConstantDwordHostDepthAddressBit;
}
if (transfer_push_constants_unset &
kTransferUsedPushConstantDwordAddressBit) {
command_buffer.CmdVkPushConstants(
transfer_pipeline_layout, VK_SHADER_STAGE_FRAGMENT_BIT,
sizeof(uint32_t) *
xe::bit_count(
transfer_pipeline_layout_info.used_push_constant_dwords &
(kTransferUsedPushConstantDwordAddressBit - 1)),
sizeof(uint32_t), &last_address_constant);
transfer_push_constants_set |=
kTransferUsedPushConstantDwordAddressBit;
}
for (uint32_t j = 0; j < transfer_sample_pipeline_count; ++j) {
if (j) {
command_processor_.BindExternalGraphicsPipeline(
transfer_pipelines[j]);
}
for (uint32_t k = 0; k < uint32_t(transfer_is_stencil_bit ? 8 : 1);
++k) {
if (transfer_is_stencil_bit) {
uint32_t transfer_stencil_bit = uint32_t(1) << k;
command_buffer.CmdVkPushConstants(
transfer_pipeline_layout, VK_SHADER_STAGE_FRAGMENT_BIT,
sizeof(uint32_t) *
xe::bit_count(
transfer_pipeline_layout_info
.used_push_constant_dwords &
(kTransferUsedPushConstantDwordStencilMaskBit - 1)),
sizeof(uint32_t), &transfer_stencil_bit);
command_buffer.CmdVkSetStencilWriteMask(
VK_STENCIL_FACE_FRONT_AND_BACK, transfer_stencil_bit);
}
command_buffer.CmdVkDraw(transfer_vertex_count, 1, 0, 0);
}
}
}
}
// Perform the clear.
if (resolve_clear_needed) {
command_processor_.SubmitBarriersAndEnterRenderTargetCacheRenderPass(
transfer_render_pass, transfer_framebuffer);
VkClearAttachment resolve_clear_attachment;
resolve_clear_attachment.colorAttachment = 0;
std::memset(&resolve_clear_attachment.clearValue, 0,
sizeof(resolve_clear_attachment.clearValue));
uint64_t clear_value = render_target_resolve_clear_values[i];
if (dest_rt_key.is_depth) {
resolve_clear_attachment.aspectMask =
VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT;
uint32_t depth_guest_clear_value =
(uint32_t(clear_value) >> 8) & 0xFFFFFF;
switch (dest_rt_key.GetDepthFormat()) {
case xenos::DepthRenderTargetFormat::kD24S8:
resolve_clear_attachment.clearValue.depthStencil.depth =
xenos::UNorm24To32(depth_guest_clear_value);
break;
case xenos::DepthRenderTargetFormat::kD24FS8:
// Taking [0, 2) -> [0, 1) remapping into account.
resolve_clear_attachment.clearValue.depthStencil.depth =
xenos::Float20e4To32(depth_guest_clear_value) * 0.5f;
break;
}
resolve_clear_attachment.clearValue.depthStencil.stencil =
uint32_t(clear_value) & 0xFF;
} else {
resolve_clear_attachment.aspectMask = VK_IMAGE_ASPECT_COLOR_BIT;
switch (dest_rt_key.GetColorFormat()) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA: {
for (uint32_t j = 0; j < 4; ++j) {
resolve_clear_attachment.clearValue.color.float32[j] =
((clear_value >> (j * 8)) & 0xFF) * (1.0f / 0xFF);
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10: {
for (uint32_t j = 0; j < 3; ++j) {
resolve_clear_attachment.clearValue.color.float32[j] =
((clear_value >> (j * 10)) & 0x3FF) * (1.0f / 0x3FF);
}
resolve_clear_attachment.clearValue.color.float32[3] =
((clear_value >> 30) & 0x3) * (1.0f / 0x3);
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::
k_2_10_10_10_FLOAT_AS_16_16_16_16: {
for (uint32_t j = 0; j < 3; ++j) {
resolve_clear_attachment.clearValue.color.float32[j] =
xenos::Float7e3To32((clear_value >> (j * 10)) & 0x3FF);
}
resolve_clear_attachment.clearValue.color.float32[3] =
((clear_value >> 30) & 0x3) * (1.0f / 0x3);
} break;
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT: {
// Using uint for transfers and clears of both. Disregarding the
// current -32...32 vs. -1...1 settings for consistency with color
// clear via depth aliasing.
// TODO(Triang3l): Handle cases of unsupported multisampled 16_UINT
// and completely unsupported 16_UNORM.
for (uint32_t j = 0; j < 2; ++j) {
resolve_clear_attachment.clearValue.color.uint32[j] =
uint32_t(clear_value >> (j * 16)) & 0xFFFF;
}
} break;
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT: {
// Using uint for transfers and clears of both. Disregarding the
// current -32...32 vs. -1...1 settings for consistency with color
// clear via depth aliasing.
// TODO(Triang3l): Handle cases of unsupported multisampled 16_UINT
// and completely unsupported 16_UNORM.
for (uint32_t j = 0; j < 4; ++j) {
resolve_clear_attachment.clearValue.color.uint32[j] =
uint32_t(clear_value >> (j * 16)) & 0xFFFF;
}
} break;
case xenos::ColorRenderTargetFormat::k_32_FLOAT: {
// Using uint for proper denormal and NaN handling.
resolve_clear_attachment.clearValue.color.uint32[0] =
uint32_t(clear_value);
} break;
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT: {
// Using uint for proper denormal and NaN handling.
resolve_clear_attachment.clearValue.color.uint32[0] =
uint32_t(clear_value);
resolve_clear_attachment.clearValue.color.uint32[1] =
uint32_t(clear_value >> 32);
} break;
}
}
command_buffer.CmdVkClearAttachments(1, &resolve_clear_attachment, 1,
&resolve_clear_rect);
}
}
}
VkPipeline VulkanRenderTargetCache::GetDumpPipeline(DumpPipelineKey key) {
auto pipeline_it = dump_pipelines_.find(key);
if (pipeline_it != dump_pipelines_.end()) {
return pipeline_it->second;
}
std::vector<spv::Id> id_vector_temp;
spv::Builder builder(spv::Spv_1_0,
(SpirvShaderTranslator::kSpirvMagicToolId << 16) | 1,
nullptr);
spv::Id ext_inst_glsl_std_450 = builder.import("GLSL.std.450");
builder.addCapability(spv::CapabilityShader);
builder.setMemoryModel(spv::AddressingModelLogical, spv::MemoryModelGLSL450);
builder.setSource(spv::SourceLanguageUnknown, 0);
spv::Id type_void = builder.makeVoidType();
spv::Id type_int = builder.makeIntType(32);
spv::Id type_int2 = builder.makeVectorType(type_int, 2);
spv::Id type_uint = builder.makeUintType(32);
spv::Id type_uint2 = builder.makeVectorType(type_uint, 2);
spv::Id type_uint3 = builder.makeVectorType(type_uint, 3);
spv::Id type_float = builder.makeFloatType(32);
// Bindings.
// EDRAM buffer.
bool format_is_64bpp = !key.is_depth && xenos::IsColorRenderTargetFormat64bpp(
key.GetColorFormat());
id_vector_temp.clear();
id_vector_temp.push_back(
builder.makeRuntimeArray(format_is_64bpp ? type_uint2 : type_uint));
// Storage buffers have std430 packing, no padding to 4-component vectors.
builder.addDecoration(id_vector_temp.back(), spv::DecorationArrayStride,
sizeof(uint32_t) << uint32_t(format_is_64bpp));
spv::Id type_edram = builder.makeStructType(id_vector_temp, "XeEdram");
builder.addMemberName(type_edram, 0, "edram");
builder.addMemberDecoration(type_edram, 0, spv::DecorationNonReadable);
builder.addMemberDecoration(type_edram, 0, spv::DecorationOffset, 0);
// Block since SPIR-V 1.3, but since SPIR-V 1.0 is generated, it's
// BufferBlock.
builder.addDecoration(type_edram, spv::DecorationBufferBlock);
// StorageBuffer since SPIR-V 1.3, but since SPIR-V 1.0 is generated, it's
// Uniform.
spv::Id edram_buffer = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniform, type_edram, "xe_edram");
builder.addDecoration(edram_buffer, spv::DecorationDescriptorSet,
kDumpDescriptorSetEdram);
builder.addDecoration(edram_buffer, spv::DecorationBinding, 0);
// Color or depth source.
bool source_is_multisampled = key.msaa_samples != xenos::MsaaSamples::k1X;
bool source_is_uint;
if (key.is_depth) {
source_is_uint = false;
} else {
GetColorOwnershipTransferVulkanFormat(key.GetColorFormat(),
&source_is_uint);
}
spv::Id source_component_type = source_is_uint ? type_uint : type_float;
spv::Id source_texture = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniformConstant,
builder.makeImageType(source_component_type, spv::Dim2D, false, false,
source_is_multisampled, 1, spv::ImageFormatUnknown),
"xe_edram_dump_source");
builder.addDecoration(source_texture, spv::DecorationDescriptorSet,
kDumpDescriptorSetSource);
builder.addDecoration(source_texture, spv::DecorationBinding, 0);
// Stencil source.
spv::Id source_stencil_texture = spv::NoResult;
if (key.is_depth) {
source_stencil_texture = builder.createVariable(
spv::NoPrecision, spv::StorageClassUniformConstant,
builder.makeImageType(type_uint, spv::Dim2D, false, false,
source_is_multisampled, 1,
spv::ImageFormatUnknown),
"xe_edram_dump_stencil");
builder.addDecoration(source_stencil_texture, spv::DecorationDescriptorSet,
kDumpDescriptorSetSource);
builder.addDecoration(source_stencil_texture, spv::DecorationBinding, 1);
}
// Push constants.
id_vector_temp.clear();
id_vector_temp.reserve(kDumpPushConstantCount);
for (uint32_t i = 0; i < kDumpPushConstantCount; ++i) {
id_vector_temp.push_back(type_uint);
}
spv::Id type_push_constants =
builder.makeStructType(id_vector_temp, "XeEdramDumpPushConstants");
builder.addMemberName(type_push_constants, kDumpPushConstantPitches,
"pitches");
builder.addMemberDecoration(type_push_constants, kDumpPushConstantPitches,
spv::DecorationOffset,
int(sizeof(uint32_t) * kDumpPushConstantPitches));
builder.addMemberName(type_push_constants, kDumpPushConstantOffsets,
"offsets");
builder.addMemberDecoration(type_push_constants, kDumpPushConstantOffsets,
spv::DecorationOffset,
int(sizeof(uint32_t) * kDumpPushConstantOffsets));
builder.addDecoration(type_push_constants, spv::DecorationBlock);
spv::Id push_constants = builder.createVariable(
spv::NoPrecision, spv::StorageClassPushConstant, type_push_constants,
"xe_edram_dump_push_constants");
// gl_GlobalInvocationID input.
spv::Id input_global_invocation_id =
builder.createVariable(spv::NoPrecision, spv::StorageClassInput,
type_uint3, "gl_GlobalInvocationID");
builder.addDecoration(input_global_invocation_id, spv::DecorationBuiltIn,
spv::BuiltInGlobalInvocationId);
// Begin the main function.
std::vector<spv::Id> main_param_types;
std::vector<std::vector<spv::Decoration>> main_precisions;
spv::Block* main_entry;
spv::Function* main_function =
builder.makeFunctionEntry(spv::NoPrecision, type_void, "main",
main_param_types, main_precisions, &main_entry);
// For now, as the exact addressing in 64bpp render targets relatively to
// 32bpp is unknown, treating 64bpp tiles as storing 40x16 samples rather than
// 80x16 for simplicity of addressing into the texture.
// Split the destination sample index into the 32bpp tile and the
// 32bpp-tile-relative sample index.
// Note that division by non-power-of-two constants will include a 4-cycle
// 32*32 multiplication on AMD, even though so many bits are not needed for
// the sample position - however, if an OpUnreachable path is inserted for the
// case when the position has upper bits set, for some reason, the code for it
// is not eliminated when compiling the shader for AMD via RenderDoc on
// Windows, as of June 2022.
spv::Id global_invocation_id =
builder.createLoad(input_global_invocation_id, spv::NoPrecision);
spv::Id rectangle_sample_x =
builder.createCompositeExtract(global_invocation_id, type_uint, 0);
uint32_t tile_width =
(xenos::kEdramTileWidthSamples >> uint32_t(format_is_64bpp)) *
draw_resolution_scale_x();
spv::Id const_tile_width = builder.makeUintConstant(tile_width);
spv::Id rectangle_tile_index_x = builder.createBinOp(
spv::OpUDiv, type_uint, rectangle_sample_x, const_tile_width);
spv::Id tile_sample_x = builder.createBinOp(
spv::OpUMod, type_uint, rectangle_sample_x, const_tile_width);
spv::Id rectangle_sample_y =
builder.createCompositeExtract(global_invocation_id, type_uint, 1);
uint32_t tile_height =
xenos::kEdramTileHeightSamples * draw_resolution_scale_y();
spv::Id const_tile_height = builder.makeUintConstant(tile_height);
spv::Id rectangle_tile_index_y = builder.createBinOp(
spv::OpUDiv, type_uint, rectangle_sample_y, const_tile_height);
spv::Id tile_sample_y = builder.createBinOp(
spv::OpUMod, type_uint, rectangle_sample_y, const_tile_height);
// Get the tile index in the EDRAM relative to the dump rectangle base tile.
id_vector_temp.clear();
id_vector_temp.push_back(builder.makeIntConstant(kDumpPushConstantPitches));
spv::Id pitches_constant = builder.createLoad(
builder.createAccessChain(spv::StorageClassPushConstant, push_constants,
id_vector_temp),
spv::NoPrecision);
spv::Id const_uint_0 = builder.makeUintConstant(0);
spv::Id const_edram_pitch_tiles_bits =
builder.makeUintConstant(xenos::kEdramPitchTilesBits);
spv::Id rectangle_tile_index = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.createTriOp(spv::OpBitFieldUExtract, type_uint,
pitches_constant, const_uint_0,
const_edram_pitch_tiles_bits),
rectangle_tile_index_y),
rectangle_tile_index_x);
// Add the base tile in the dispatch to the dispatch-local tile index, not
// wrapping yet so in case of a wraparound, the address relative to the base
// in the image after subtraction of the base won't be negative.
id_vector_temp.clear();
id_vector_temp.push_back(builder.makeIntConstant(kDumpPushConstantOffsets));
spv::Id offsets_constant = builder.createLoad(
builder.createAccessChain(spv::StorageClassPushConstant, push_constants,
id_vector_temp),
spv::NoPrecision);
spv::Id const_edram_base_tiles_bits_plus_1 =
builder.makeUintConstant(xenos::kEdramBaseTilesBits + 1);
spv::Id edram_tile_index_non_wrapped = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createTriOp(spv::OpBitFieldUExtract, type_uint, offsets_constant,
const_uint_0, const_edram_base_tiles_bits_plus_1),
rectangle_tile_index);
// Combine the tile sample index and the tile index, wrapping the tile
// addressing, into the EDRAM sample index.
spv::Id edram_sample_address = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(
spv::OpIMul, type_uint,
builder.makeUintConstant(tile_width * tile_height),
builder.createBinOp(
spv::OpBitwiseAnd, type_uint, edram_tile_index_non_wrapped,
builder.makeUintConstant(xenos::kEdramTileCount - 1))),
builder.createBinOp(spv::OpIAdd, type_uint,
builder.createBinOp(spv::OpIMul, type_uint,
const_tile_width, tile_sample_y),
tile_sample_x));
if (key.is_depth) {
// Swap 40-sample columns in the depth buffer in the destination address to
// get the final address of the sample in the EDRAM.
uint32_t tile_width_half = tile_width >> 1;
edram_sample_address = builder.createUnaryOp(
spv::OpBitcast, type_uint,
builder.createBinOp(
spv::OpIAdd, type_int,
builder.createUnaryOp(spv::OpBitcast, type_int,
edram_sample_address),
builder.createTriOp(
spv::OpSelect, type_int,
builder.createBinOp(spv::OpULessThan, builder.makeBoolType(),
tile_sample_x,
builder.makeUintConstant(tile_width_half)),
builder.makeIntConstant(int32_t(tile_width_half)),
builder.makeIntConstant(-int32_t(tile_width_half)))));
}
// Get the linear tile index within the source texture.
spv::Id source_tile_index = builder.createBinOp(
spv::OpISub, type_uint, edram_tile_index_non_wrapped,
builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, offsets_constant,
const_edram_base_tiles_bits_plus_1,
builder.makeUintConstant(xenos::kEdramBaseTilesBits)));
// Split the linear tile index in the source texture into X and Y in tiles.
spv::Id source_pitch_tiles = builder.createTriOp(
spv::OpBitFieldUExtract, type_uint, pitches_constant,
const_edram_pitch_tiles_bits, const_edram_pitch_tiles_bits);
spv::Id source_tile_index_y = builder.createBinOp(
spv::OpUDiv, type_uint, source_tile_index, source_pitch_tiles);
spv::Id source_tile_index_x = builder.createBinOp(
spv::OpUMod, type_uint, source_tile_index, source_pitch_tiles);
// Combine the source tile offset and the sample index within the tile.
spv::Id source_sample_x = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(spv::OpIMul, type_uint, const_tile_width,
source_tile_index_x),
tile_sample_x);
spv::Id source_sample_y = builder.createBinOp(
spv::OpIAdd, type_uint,
builder.createBinOp(spv::OpIMul, type_uint, const_tile_height,
source_tile_index_y),
tile_sample_y);
// Get the source pixel coordinate and the sample index within the pixel.
spv::Id source_pixel_x = source_sample_x, source_pixel_y = source_sample_y;
spv::Id source_sample_id = spv::NoResult;
if (source_is_multisampled) {
spv::Id const_uint_1 = builder.makeUintConstant(1);
source_pixel_y = builder.createBinOp(spv::OpShiftRightLogical, type_uint,
source_sample_y, const_uint_1);
if (key.msaa_samples >= xenos::MsaaSamples::k4X) {
source_pixel_x = builder.createBinOp(spv::OpShiftRightLogical, type_uint,
source_sample_x, const_uint_1);
// 4x MSAA source texture sample index - bit 0 for horizontal, bit 1 for
// vertical.
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(builder.createBinOp(
spv::OpBitwiseAnd, type_uint, source_sample_x, const_uint_1));
id_vector_temp.push_back(source_sample_y);
id_vector_temp.push_back(const_uint_1);
id_vector_temp.push_back(const_uint_1);
source_sample_id =
builder.createOp(spv::OpBitFieldInsert, type_uint, id_vector_temp);
} else {
// 2x MSAA source texture sample index - convert from the guest to
// the Vulkan standard sample locations.
source_sample_id = builder.createTriOp(
spv::OpSelect, type_uint,
builder.createBinOp(
spv::OpINotEqual, builder.makeBoolType(),
builder.createBinOp(spv::OpBitwiseAnd, type_uint, source_sample_y,
const_uint_1),
const_uint_0),
builder.makeUintConstant(draw_util::GetD3D10SampleIndexForGuest2xMSAA(
1, msaa_2x_attachments_supported_)),
builder.makeUintConstant(draw_util::GetD3D10SampleIndexForGuest2xMSAA(
0, msaa_2x_attachments_supported_)));
}
}
// Load the source, and pack the value into one or two 32-bit integers.
spv::Id packed[2] = {};
spv::Builder::TextureParameters source_texture_parameters = {};
source_texture_parameters.sampler =
builder.createLoad(source_texture, spv::NoPrecision);
id_vector_temp.clear();
id_vector_temp.reserve(2);
id_vector_temp.push_back(
builder.createUnaryOp(spv::OpBitcast, type_int, source_pixel_x));
id_vector_temp.push_back(
builder.createUnaryOp(spv::OpBitcast, type_int, source_pixel_y));
source_texture_parameters.coords =
builder.createCompositeConstruct(type_int2, id_vector_temp);
if (source_is_multisampled) {
source_texture_parameters.sample =
builder.createUnaryOp(spv::OpBitcast, type_int, source_sample_id);
} else {
source_texture_parameters.lod = builder.makeIntConstant(0);
}
spv::Id source_vec4 = builder.createTextureCall(
spv::NoPrecision, builder.makeVectorType(source_component_type, 4), false,
true, false, false, false, source_texture_parameters,
spv::ImageOperandsMaskNone);
if (key.is_depth) {
source_texture_parameters.sampler =
builder.createLoad(source_stencil_texture, spv::NoPrecision);
spv::Id source_stencil = builder.createCompositeExtract(
builder.createTextureCall(
spv::NoPrecision, builder.makeVectorType(type_uint, 4), false, true,
false, false, false, source_texture_parameters,
spv::ImageOperandsMaskNone),
type_uint, 0);
spv::Id source_depth32 =
builder.createCompositeExtract(source_vec4, type_float, 0);
switch (key.GetDepthFormat()) {
case xenos::DepthRenderTargetFormat::kD24S8: {
// Round to the nearest even integer. This seems to be the correct
// conversion, adding +0.5 and rounding towards zero results in red
// instead of black in the 4D5307E6 clear shader.
id_vector_temp.clear();
id_vector_temp.push_back(
builder.createBinOp(spv::OpFMul, type_float, source_depth32,
builder.makeFloatConstant(float(0xFFFFFF))));
packed[0] = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBuiltinCall(type_float, ext_inst_glsl_std_450,
GLSLstd450RoundEven, id_vector_temp));
} break;
case xenos::DepthRenderTargetFormat::kD24FS8: {
packed[0] = SpirvShaderTranslator::PreClampedDepthTo20e4(
builder, source_depth32, depth_float24_round(), true,
ext_inst_glsl_std_450);
} break;
}
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(source_stencil);
id_vector_temp.push_back(packed[0]);
id_vector_temp.push_back(builder.makeUintConstant(8));
id_vector_temp.push_back(builder.makeUintConstant(24));
packed[0] =
builder.createOp(spv::OpBitFieldInsert, type_uint, id_vector_temp);
} else {
switch (key.GetColorFormat()) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA: {
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale = builder.makeFloatConstant(255.0f);
packed[0] = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float,
builder.createCompositeExtract(source_vec4, type_float, 0),
unorm_scale),
unorm_round_offset));
spv::Id component_width = builder.makeUintConstant(8);
for (uint32_t i = 1; i < 4; ++i) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed[0]);
id_vector_temp.push_back(builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
builder.createCompositeExtract(
source_vec4, type_float, i),
unorm_scale),
unorm_round_offset)));
id_vector_temp.push_back(builder.makeUintConstant(8 * i));
id_vector_temp.push_back(component_width);
packed[0] = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10: {
spv::Id unorm_round_offset = builder.makeFloatConstant(0.5f);
spv::Id unorm_scale_rgb = builder.makeFloatConstant(1023.0f);
packed[0] = builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(
spv::OpFMul, type_float,
builder.createCompositeExtract(source_vec4, type_float, 0),
unorm_scale_rgb),
unorm_round_offset));
spv::Id width_rgb = builder.makeUintConstant(10);
spv::Id unorm_scale_a = builder.makeFloatConstant(3.0f);
spv::Id width_a = builder.makeUintConstant(2);
for (uint32_t i = 1; i < 4; ++i) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed[0]);
id_vector_temp.push_back(builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float,
builder.createCompositeExtract(
source_vec4, type_float, i),
i == 3 ? unorm_scale_a : unorm_scale_rgb),
unorm_round_offset)));
id_vector_temp.push_back(builder.makeUintConstant(10 * i));
id_vector_temp.push_back(i == 3 ? width_a : width_rgb);
packed[0] = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT_AS_16_16_16_16: {
// Float16 has a wider range for both color and alpha, also NaNs - clamp
// and convert.
packed[0] = SpirvShaderTranslator::UnclampedFloat32To7e3(
builder, builder.createCompositeExtract(source_vec4, type_float, 0),
ext_inst_glsl_std_450);
spv::Id width_rgb = builder.makeUintConstant(10);
for (uint32_t i = 1; i < 3; ++i) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed[0]);
id_vector_temp.push_back(SpirvShaderTranslator::UnclampedFloat32To7e3(
builder,
builder.createCompositeExtract(source_vec4, type_float, i),
ext_inst_glsl_std_450));
id_vector_temp.push_back(builder.makeUintConstant(10 * i));
id_vector_temp.push_back(width_rgb);
packed[0] = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
// Saturate and convert the alpha.
id_vector_temp.clear();
id_vector_temp.reserve(3);
id_vector_temp.push_back(
builder.createCompositeExtract(source_vec4, type_float, 3));
id_vector_temp.push_back(builder.makeFloatConstant(0.0f));
id_vector_temp.push_back(builder.makeFloatConstant(1.0f));
spv::Id alpha_saturated =
builder.createBuiltinCall(type_float, ext_inst_glsl_std_450,
GLSLstd450NClamp, id_vector_temp);
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(packed[0]);
id_vector_temp.push_back(builder.createUnaryOp(
spv::OpConvertFToU, type_uint,
builder.createBinOp(
spv::OpFAdd, type_float,
builder.createBinOp(spv::OpFMul, type_float, alpha_saturated,
builder.makeFloatConstant(3.0f)),
builder.makeFloatConstant(0.5f))));
id_vector_temp.push_back(builder.makeUintConstant(30));
id_vector_temp.push_back(builder.makeUintConstant(2));
packed[0] = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
} break;
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT:
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT: {
// All 64bpp formats, and all 16 bits per component formats, are
// represented as integers in ownership transfer for safe handling of
// NaN encodings and -32768 / -32767.
// TODO(Triang3l): Handle the case when that's not true (no multisampled
// sampled images, no 16-bit UNORM, no cross-packing 32bpp aliasing on a
// portability subset device or a 64bpp format where that wouldn't help
// anyway).
spv::Id component_offset_width = builder.makeUintConstant(16);
for (uint32_t i = 0; i <= uint32_t(format_is_64bpp); ++i) {
id_vector_temp.clear();
id_vector_temp.reserve(4);
id_vector_temp.push_back(
builder.createCompositeExtract(source_vec4, type_uint, 2 * i));
id_vector_temp.push_back(builder.createCompositeExtract(
source_vec4, type_uint, 2 * i + 1));
id_vector_temp.push_back(component_offset_width);
id_vector_temp.push_back(component_offset_width);
packed[i] = builder.createOp(spv::OpBitFieldInsert, type_uint,
id_vector_temp);
}
} break;
// Float32 is transferred as uint32 to preserve NaN encodings. However,
// multisampled sampled image support is optional in Vulkan.
case xenos::ColorRenderTargetFormat::k_32_FLOAT:
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT: {
for (uint32_t i = 0; i <= uint32_t(format_is_64bpp); ++i) {
spv::Id& packed_ref = packed[i];
packed_ref = builder.createCompositeExtract(source_vec4,
source_component_type, i);
if (!source_is_uint) {
packed_ref =
builder.createUnaryOp(spv::OpBitcast, type_uint, packed_ref);
}
}
} break;
}
}
// Write the packed value to the EDRAM buffer.
spv::Id store_value = packed[0];
if (format_is_64bpp) {
id_vector_temp.clear();
id_vector_temp.reserve(2);
id_vector_temp.push_back(packed[0]);
id_vector_temp.push_back(packed[1]);
store_value = builder.createCompositeConstruct(type_uint2, id_vector_temp);
}
id_vector_temp.clear();
id_vector_temp.reserve(2);
// The only SSBO structure member.
id_vector_temp.push_back(builder.makeIntConstant(0));
id_vector_temp.push_back(
builder.createUnaryOp(spv::OpBitcast, type_int, edram_sample_address));
// StorageBuffer since SPIR-V 1.3, but since SPIR-V 1.0 is generated, it's
// Uniform.
builder.createStore(store_value,
builder.createAccessChain(spv::StorageClassUniform,
edram_buffer, id_vector_temp));
// End the main function and make it the entry point.
builder.leaveFunction();
builder.addExecutionMode(main_function, spv::ExecutionModeLocalSize,
kDumpSamplesPerGroupX, kDumpSamplesPerGroupY, 1);
spv::Instruction* entry_point = builder.addEntryPoint(
spv::ExecutionModelGLCompute, main_function, "main");
// Bindings only need to be added to the entry point's interface starting with
// SPIR-V 1.4 - emitting 1.0 here, so only inputs / outputs.
entry_point->addIdOperand(input_global_invocation_id);
// Serialize the shader code.
std::vector<unsigned int> shader_code;
builder.dump(shader_code);
// Create the pipeline, and store the handle even if creation fails not to try
// to create it again later.
VkPipeline pipeline = ui::vulkan::util::CreateComputePipeline(
command_processor_.GetVulkanProvider(),
key.is_depth ? dump_pipeline_layout_depth_ : dump_pipeline_layout_color_,
reinterpret_cast<const uint32_t*>(shader_code.data()),
sizeof(uint32_t) * shader_code.size());
if (pipeline == VK_NULL_HANDLE) {
XELOGE(
"VulkanRenderTargetCache: Failed to create a render target dumping "
"pipeline for {}-sample render targets with format {}",
UINT32_C(1) << uint32_t(key.msaa_samples),
key.is_depth
? xenos::GetDepthRenderTargetFormatName(key.GetDepthFormat())
: xenos::GetColorRenderTargetFormatName(key.GetColorFormat()));
}
dump_pipelines_.emplace(key, pipeline);
return pipeline;
}
void VulkanRenderTargetCache::DumpRenderTargets(uint32_t dump_base,
uint32_t dump_row_length_used,
uint32_t dump_rows,
uint32_t dump_pitch) {
assert_true(GetPath() == Path::kHostRenderTargets);
GetResolveCopyRectanglesToDump(dump_base, dump_row_length_used, dump_rows,
dump_pitch, dump_rectangles_);
if (dump_rectangles_.empty()) {
return;
}
// Clear previously set temporary indices.
for (const ResolveCopyDumpRectangle& rectangle : dump_rectangles_) {
static_cast<VulkanRenderTarget*>(rectangle.render_target)
->SetTemporarySortIndex(UINT32_MAX);
}
// Gather all needed barriers and info needed to sort the invocations.
UseEdramBuffer(EdramBufferUsage::kComputeWrite);
dump_invocations_.clear();
dump_invocations_.reserve(dump_rectangles_.size());
constexpr VkPipelineStageFlags kRenderTargetDstStageMask =
VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT;
constexpr VkAccessFlags kRenderTargetDstAccessMask =
VK_ACCESS_SHADER_READ_BIT;
constexpr VkImageLayout kRenderTargetNewLayout =
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL;
uint32_t rt_sort_index = 0;
for (const ResolveCopyDumpRectangle& rectangle : dump_rectangles_) {
auto& vulkan_rt =
*static_cast<VulkanRenderTarget*>(rectangle.render_target);
RenderTargetKey rt_key = vulkan_rt.key();
command_processor_.PushImageMemoryBarrier(
vulkan_rt.image(),
ui::vulkan::util::InitializeSubresourceRange(
rt_key.is_depth
? (VK_IMAGE_ASPECT_DEPTH_BIT | VK_IMAGE_ASPECT_STENCIL_BIT)
: VK_IMAGE_ASPECT_COLOR_BIT),
vulkan_rt.current_stage_mask(), VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
vulkan_rt.current_access_mask(), VK_ACCESS_SHADER_READ_BIT,
vulkan_rt.current_layout(), VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL);
vulkan_rt.SetUsage(VK_PIPELINE_STAGE_COMPUTE_SHADER_BIT,
VK_ACCESS_SHADER_READ_BIT,
VK_IMAGE_LAYOUT_SHADER_READ_ONLY_OPTIMAL);
if (vulkan_rt.temporary_sort_index() == UINT32_MAX) {
vulkan_rt.SetTemporarySortIndex(rt_sort_index++);
}
DumpPipelineKey pipeline_key;
pipeline_key.msaa_samples = rt_key.msaa_samples;
pipeline_key.resource_format = rt_key.resource_format;
pipeline_key.is_depth = rt_key.is_depth;
dump_invocations_.emplace_back(rectangle, pipeline_key);
}
// Sort the invocations to reduce context and binding switches.
std::sort(dump_invocations_.begin(), dump_invocations_.end());
// Dump the render targets.
DeferredCommandBuffer& command_buffer =
command_processor_.deferred_command_buffer();
bool edram_buffer_bound = false;
VkDescriptorSet last_source_descriptor_set = VK_NULL_HANDLE;
DumpPitches last_pitches;
DumpOffsets last_offsets;
bool pitches_bound = false, offsets_bound = false;
for (const DumpInvocation& invocation : dump_invocations_) {
const ResolveCopyDumpRectangle& rectangle = invocation.rectangle;
auto& vulkan_rt =
*static_cast<VulkanRenderTarget*>(rectangle.render_target);
RenderTargetKey rt_key = vulkan_rt.key();
DumpPipelineKey pipeline_key = invocation.pipeline_key;
VkPipeline pipeline = GetDumpPipeline(pipeline_key);
if (!pipeline) {
continue;
}
command_processor_.BindExternalComputePipeline(pipeline);
VkPipelineLayout pipeline_layout = rt_key.is_depth
? dump_pipeline_layout_depth_
: dump_pipeline_layout_color_;
// Only need to bind the EDRAM buffer once (relying on pipeline layout
// compatibility).
if (!edram_buffer_bound) {
edram_buffer_bound = true;
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_COMPUTE, pipeline_layout,
kDumpDescriptorSetEdram, 1, &edram_storage_buffer_descriptor_set_, 0,
nullptr);
}
VkDescriptorSet source_descriptor_set =
vulkan_rt.GetDescriptorSetTransferSource();
if (last_source_descriptor_set != source_descriptor_set) {
last_source_descriptor_set = source_descriptor_set;
command_buffer.CmdVkBindDescriptorSets(
VK_PIPELINE_BIND_POINT_COMPUTE, pipeline_layout,
kDumpDescriptorSetSource, 1, &source_descriptor_set, 0, nullptr);
}
DumpPitches pitches;
pitches.dest_pitch = dump_pitch;
pitches.source_pitch = rt_key.GetPitchTiles();
if (last_pitches != pitches) {
last_pitches = pitches;
pitches_bound = false;
}
if (!pitches_bound) {
pitches_bound = true;
command_buffer.CmdVkPushConstants(
pipeline_layout, VK_SHADER_STAGE_COMPUTE_BIT,
sizeof(uint32_t) * kDumpPushConstantPitches, sizeof(last_pitches),
&last_pitches);
}
DumpOffsets offsets;
offsets.source_base_tiles = rt_key.base_tiles;
ResolveCopyDumpRectangle::Dispatch
dispatches[ResolveCopyDumpRectangle::kMaxDispatches];
uint32_t dispatch_count =
rectangle.GetDispatches(dump_pitch, dump_row_length_used, dispatches);
for (uint32_t i = 0; i < dispatch_count; ++i) {
const ResolveCopyDumpRectangle::Dispatch& dispatch = dispatches[i];
offsets.dispatch_first_tile = dump_base + dispatch.offset;
if (last_offsets != offsets) {
last_offsets = offsets;
offsets_bound = false;
}
if (!offsets_bound) {
offsets_bound = true;
command_buffer.CmdVkPushConstants(
pipeline_layout, VK_SHADER_STAGE_COMPUTE_BIT,
sizeof(uint32_t) * kDumpPushConstantOffsets, sizeof(last_offsets),
&last_offsets);
}
command_processor_.SubmitBarriers(true);
command_buffer.CmdVkDispatch(
(draw_resolution_scale_x() *
(xenos::kEdramTileWidthSamples >> uint32_t(rt_key.Is64bpp())) *
dispatch.width_tiles +
(kDumpSamplesPerGroupX - 1)) /
kDumpSamplesPerGroupX,
(draw_resolution_scale_y() * xenos::kEdramTileHeightSamples *
dispatch.height_tiles +
(kDumpSamplesPerGroupY - 1)) /
kDumpSamplesPerGroupY,
1);
}
MarkEdramBufferModified();
}
}
} // namespace vulkan
} // namespace gpu
} // namespace xe