diff --git a/src/xenia/gpu/spirv_shader_translator.cc b/src/xenia/gpu/spirv_shader_translator.cc index ca8eaac7f..0de5d9a98 100644 --- a/src/xenia/gpu/spirv_shader_translator.cc +++ b/src/xenia/gpu/spirv_shader_translator.cc @@ -119,7 +119,7 @@ void SpirvShaderTranslator::Reset() { // Vertex shader inputs. input_vertex_index_ = spv::NoResult; // Tessellation evaluation shader inputs. - input_primitive_id_ = spv::NoResult; + input_control_point_index_ = spv::NoResult; input_tess_coord_ = spv::NoResult; // Pixel shader inputs. input_point_coordinates_ = spv::NoResult; @@ -795,16 +795,22 @@ std::vector SpirvShaderTranslator::CompleteTranslation() { assert_unhandled_case(host_type); break; } - // Tessellation spacing - fractional_even for continuous mode, equal - // (integer) for discrete mode. The actual mode is determined by the TCS - // (hull shader), but we use fractional_even here as the default since it - // provides smooth results. The TCS sets the actual tessellation levels. - // For now, use fractional_even as it's more compatible. + // Tessellation spacing. In SPIR-V the spacing is part of the domain + // shader rather than the hull shader. Match the Direct3D 12 hull shader + // partitioning. Integer (equal) for discrete, fractional even for + // continuous and adaptive. builder_->addExecutionMode(function_main_, - spv::ExecutionModeSpacingFractionalEven); - // Vertex ordering - counter-clockwise (Vulkan default for front face). + shader_modification.vertex.tessellation_mode == + xenos::TessellationMode::kDiscrete + ? spv::ExecutionModeSpacingEqual + : spv::ExecutionModeSpacingFractionalEven); + // Vertex ordering. Xenia does not flip the clip space Y on Vulkan + // (origin_bottom_left is false, so ndc_scale.y keeps the guest sign), so + // the tessellator must wind the same way as the Direct3D 12 hull shaders, + // which use triangle_cw. Counter-clockwise here inverts the facing and + // the guest backface culling removes the whole surface. builder_->addExecutionMode(function_main_, - spv::ExecutionModeVertexOrderCcw); + spv::ExecutionModeVertexOrderCw); } else { execution_model = spv::ExecutionModelVertex; } @@ -1273,11 +1279,33 @@ void SpirvShaderTranslator::EnsureBuildPointAvailable() { void SpirvShaderTranslator::StartVertexOrTessEvalShaderBeforeMain() { // Create the inputs. if (IsSpirvTessEvalShader()) { - input_primitive_id_ = builder_->createVariable( - spv::NoPrecision, spv::StorageClassInput, type_int_, "gl_PrimitiveID"); - builder_->addDecoration(input_primitive_id_, spv::DecorationBuiltIn, - static_cast(spv::BuiltIn::PrimitiveId)); - main_interface_.push_back(input_primitive_id_); + // Per-control-point index input from the hull shader, mirroring the control + // point input read by the Direct3D 12 domain shader. The hull shader has + // already applied the endian swap, the vertex index offset, the low 24-bit + // wrap and the min/max clamp, so this is the index the guest expects rather + // than the raw gl_PrimitiveID. The array size matches the hull shader's + // output control point count for the domain type. + uint32_t control_point_count = 1; + switch (GetSpirvShaderModification().vertex.host_vertex_shader_type) { + case Shader::HostVertexShaderType::kTriangleDomainCPIndexed: + control_point_count = 3; + break; + case Shader::HostVertexShaderType::kQuadDomainCPIndexed: + control_point_count = 4; + break; + default: + // Patch-indexed (and line) domains output a single control point. + control_point_count = 1; + break; + } + input_control_point_index_ = builder_->createVariable( + spv::NoPrecision, spv::StorageClassInput, + builder_->makeArrayType( + type_float_, builder_->makeUintConstant(control_point_count), 0), + "xe_in_control_point_index"); + builder_->addDecoration(input_control_point_index_, spv::DecorationLocation, + 0); + main_interface_.push_back(input_control_point_index_); // Tessellation coordinates (barycentric coordinates for the tessellated // vertex within the patch). @@ -1562,32 +1590,56 @@ void SpirvShaderTranslator::StartVertexOrTessEvalShaderInMain() { break; } case Shader::HostVertexShaderType::kQuadDomainCPIndexed: { - // Quad domain requires at least 2 registers (r0 for domain location, - // r1 for control point indices). + // Quad domain requires at least 2 registers (r0 for the domain + // location and the first control point index, r1 for the other + // three). assert_true(register_count() >= 2); - // Quad domain CP-indexed: gl_TessCoord.xy -> r0.yz, r0.x = 0, r0.w = - // 1 XY swizzle according to the ground shader in 4D5307F2. - uint_vector_temp_.clear(); - uint_vector_temp_.push_back(0); // x -> r0.y - uint_vector_temp_.push_back(1); // y -> r0.z - spv::Id tess_coord_xy = builder_->createRvalueSwizzle( - spv::NoPrecision, type_float2_, tess_coord, uint_vector_temp_); - // Store to r0 + // Quad domain CP-indexed, matching the Direct3D 12 domain shader: + // r0.xy = domain location, r0.z = control point index 0, + // r1.xyz = control point indices 1, 2, 3 (already endian swapped and + // converted to float by the host vertex and hull shaders). + spv::Id tess_coord_x = + builder_->createCompositeExtract(tess_coord, type_float_, 0); + spv::Id tess_coord_y = + builder_->createCompositeExtract(tess_coord, type_float_, 1); + // Load control point index 0. + id_vector_temp_.clear(); + id_vector_temp_.push_back(const_int_0_); + spv::Id control_point_index_0 = builder_->createLoad( + builder_->createAccessChain(spv::StorageClassInput, + input_control_point_index_, + id_vector_temp_), + spv::NoPrecision); + // Store r0 = (domain.x, domain.y, control point index 0, 0). id_vector_temp_.clear(); id_vector_temp_.push_back(const_int_0_); spv::Id r0_ptr = builder_->createAccessChain( spv::StorageClassFunction, var_main_registers_, id_vector_temp_); - // Build float4 with x=0, yz from tess coord, w=1 id_vector_temp_.clear(); + id_vector_temp_.push_back(tess_coord_x); + id_vector_temp_.push_back(tess_coord_y); + id_vector_temp_.push_back(control_point_index_0); id_vector_temp_.push_back(const_float_0_); - id_vector_temp_.push_back( - builder_->createCompositeExtract(tess_coord_xy, type_float_, 0)); - id_vector_temp_.push_back( - builder_->createCompositeExtract(tess_coord_xy, type_float_, 1)); - id_vector_temp_.push_back(const_float_1_); builder_->createStore( builder_->createCompositeConstruct(type_float4_, id_vector_temp_), r0_ptr); + // Store r1.xyz = control point indices 1, 2, 3. + for (uint32_t i = 1; i <= 3; ++i) { + id_vector_temp_.clear(); + id_vector_temp_.push_back(builder_->makeIntConstant(int(i))); + spv::Id control_point_index = builder_->createLoad( + builder_->createAccessChain(spv::StorageClassInput, + input_control_point_index_, + id_vector_temp_), + spv::NoPrecision); + id_vector_temp_.clear(); + id_vector_temp_.push_back(builder_->makeIntConstant(1)); + id_vector_temp_.push_back(builder_->makeIntConstant(int(i - 1))); + builder_->createStore(control_point_index, + builder_->createAccessChain( + spv::StorageClassFunction, + var_main_registers_, id_vector_temp_)); + } break; } case Shader::HostVertexShaderType::kQuadDomainPatchIndexed: { @@ -1602,11 +1654,17 @@ void SpirvShaderTranslator::StartVertexOrTessEvalShaderInMain() { uint_vector_temp_.push_back(1); // y -> r0.z spv::Id tess_coord_xy = builder_->createRvalueSwizzle( spv::NoPrecision, type_float2_, tess_coord, uint_vector_temp_); - // Load primitive ID (patch index) and convert to float. - spv::Id primitive_id = - builder_->createLoad(input_primitive_id_, spv::NoPrecision); - spv::Id patch_index_float = builder_->createUnaryOp( - spv::OpConvertSToF, type_float_, primitive_id); + // Read the patch index from control point 0 (already endian swapped, + // offset, wrapped and clamped by the host vertex and hull shaders), + // matching the Direct3D 12 domain shader, rather than using the raw + // gl_PrimitiveID. + id_vector_temp_.clear(); + id_vector_temp_.push_back(const_int_0_); + spv::Id patch_index_float = builder_->createLoad( + builder_->createAccessChain(spv::StorageClassInput, + input_control_point_index_, + id_vector_temp_), + spv::NoPrecision); // Store to r0: x = patch index, yz = tess coord, w = 1 id_vector_temp_.clear(); id_vector_temp_.push_back(const_int_0_); @@ -1652,11 +1710,17 @@ void SpirvShaderTranslator::StartVertexOrTessEvalShaderInMain() { if (register_count() >= 2) { if (host_type == Shader::HostVertexShaderType::kTriangleDomainPatchIndexed) { - // Load primitive ID (patch index) and convert to float. - spv::Id primitive_id = - builder_->createLoad(input_primitive_id_, spv::NoPrecision); - spv::Id patch_index_float = builder_->createUnaryOp( - spv::OpConvertSToF, type_float_, primitive_id); + // Read the patch index from control point 0 (already endian swapped, + // offset, wrapped and clamped by the host vertex and hull shaders), + // matching the Direct3D 12 domain shader, rather than using the raw + // gl_PrimitiveID. + id_vector_temp_.clear(); + id_vector_temp_.push_back(const_int_0_); + spv::Id patch_index_float = builder_->createLoad( + builder_->createAccessChain(spv::StorageClassInput, + input_control_point_index_, + id_vector_temp_), + spv::NoPrecision); // Store patch index to r1.x id_vector_temp_.clear(); id_vector_temp_.push_back(builder_->makeIntConstant(1)); @@ -1675,6 +1739,27 @@ void SpirvShaderTranslator::StartVertexOrTessEvalShaderInMain() { builder_->createAccessChain( spv::StorageClassFunction, var_main_registers_, id_vector_temp_)); + } else if (host_type == + Shader::HostVertexShaderType::kTriangleDomainCPIndexed) { + // Store the three control point indices (already endian swapped and + // converted to float by the host vertex and hull shaders) to r1.xyz, + // matching the Direct3D 12 domain shader. + for (uint32_t i = 0; i < 3; ++i) { + id_vector_temp_.clear(); + id_vector_temp_.push_back(builder_->makeIntConstant(int(i))); + spv::Id control_point_index = builder_->createLoad( + builder_->createAccessChain(spv::StorageClassInput, + input_control_point_index_, + id_vector_temp_), + spv::NoPrecision); + id_vector_temp_.clear(); + id_vector_temp_.push_back(builder_->makeIntConstant(1)); + id_vector_temp_.push_back(builder_->makeIntConstant(int(i))); + builder_->createStore(control_point_index, + builder_->createAccessChain( + spv::StorageClassFunction, + var_main_registers_, id_vector_temp_)); + } } } } else if (IsSpirvVertexShader()) { diff --git a/src/xenia/gpu/spirv_shader_translator.h b/src/xenia/gpu/spirv_shader_translator.h index f1bd00060..ca4d7d5d0 100644 --- a/src/xenia/gpu/spirv_shader_translator.h +++ b/src/xenia/gpu/spirv_shader_translator.h @@ -34,7 +34,7 @@ class SpirvShaderTranslator : public ShaderTranslator { // TODO(Triang3l): Change to 0xYYYYMMDD once it's out of the rapid // prototyping stage (easier to do small granular updates with an // incremental counter). - static constexpr uint32_t kVersion = 17; + static constexpr uint32_t kVersion = 18; enum class DepthStencilMode : uint32_t { kNoModifiers, @@ -86,6 +86,11 @@ class SpirvShaderTranslator : public ShaderTranslator { // cull distance written after the user clip plane cull distances. The // "or" operator sets the position to NaN instead and needs no bit here. uint32_t vertex_kill_and : 1; + // The tessellation mode for domain shaders, selecting the tessellation + // evaluation shader spacing (Direct3D 12 sets it in the hull shaders, but + // in SPIR-V the spacing lives in the domain shader). Discrete uses equal + // spacing, continuous and adaptive use fractional even. + xenos::TessellationMode tessellation_mode : 2; } vertex; struct PixelShaderModification { // uint32_t 0. @@ -1024,8 +1029,9 @@ class SpirvShaderTranslator : public ShaderTranslator { // VS as VS only - int. spv::Id input_vertex_index_; - // VS as TES only - int. - spv::Id input_primitive_id_; + // VS as TES only - per-control-point float array carrying the patch/control + // point index computed by the host vertex and hull shaders. + spv::Id input_control_point_index_; // VS as TES only - float3 (barycentric coordinates). spv::Id input_tess_coord_; // PS, only when needed - float2. diff --git a/src/xenia/gpu/vulkan/vulkan_command_processor.cc b/src/xenia/gpu/vulkan/vulkan_command_processor.cc index 587b2295c..585b323a2 100644 --- a/src/xenia/gpu/vulkan/vulkan_command_processor.cc +++ b/src/xenia/gpu/vulkan/vulkan_command_processor.cc @@ -1402,6 +1402,17 @@ void VulkanCommandProcessor::WriteRegister(uint32_t index, uint32_t value) { texture_cache_->TextureFetchConstantWritten( (index - XE_GPU_REG_SHADER_CONSTANT_FETCH_00_0) / 6); } + } else if (index == XE_GPU_REG_VGT_MAX_VTX_INDX || + index == XE_GPU_REG_VGT_MIN_VTX_INDX || + index == XE_GPU_REG_VGT_INDX_OFFSET || + index == XE_GPU_REG_VGT_DMA_SIZE || + index == XE_GPU_REG_VGT_HOS_MAX_TESS_LEVEL || + index == XE_GPU_REG_VGT_HOS_MIN_TESS_LEVEL) { + // Source registers for the tessellation constant buffer. Invalidate it so + // the factor range and index parameters are refreshed per draw instead of + // staying stale from the first draw of the submission. + current_constant_buffers_up_to_date_ &= + ~(UINT32_C(1) << SpirvShaderTranslator::kConstantBufferTessellation); } } void VulkanCommandProcessor::WriteRegistersFromMem(uint32_t start_index, @@ -2749,6 +2760,18 @@ bool VulkanCommandProcessor::IssueDraw(xenos::PrimitiveType prim_type, primitive_processing_result.host_vertex_shader_type == Shader::HostVertexShaderType::kVertex; + // The tessellation vertex index endian is a per-draw value from the + // primitive processor rather than a register, kNone for auto draws whose + // factors were already converted on the host. A change between draws must + // invalidate the tessellation constant buffer explicitly. + if (current_tessellation_index_endian_ != + primitive_processing_result.host_shader_index_endian) { + current_tessellation_index_endian_ = + primitive_processing_result.host_shader_index_endian; + current_constant_buffers_up_to_date_ &= + ~(UINT32_C(1) << SpirvShaderTranslator::kConstantBufferTessellation); + } + // Update system constants before uploading them. UpdateSystemConstantValues( primitive_polygonal, primitive_processing_result, shader_32bit_index_dma, @@ -5490,10 +5513,13 @@ bool VulkanCommandProcessor::UpdateBindings(const VulkanShader* vertex_shader, tessellation_constants.tessellation_factor_range[1] = tess_factor_max; tessellation_constants.padding0[0] = 0.0f; tessellation_constants.padding0[1] = 0.0f; - // Vertex index processing parameters for tessellation shaders. - auto vgt_dma_size = regs.Get(); + // Vertex index processing parameters for tessellation shaders. The + // endian comes from the primitive processor, not raw + // VGT_DMA_SIZE.swap_mode. Auto draws read factors the host has already + // byte swapped, so swapping them again in the hull shader corrupted every + // adaptive factor. tessellation_constants.vertex_index_endian = - static_cast(vgt_dma_size.swap_mode); + static_cast(current_tessellation_index_endian_); tessellation_constants.vertex_index_offset = regs[XE_GPU_REG_VGT_INDX_OFFSET]; tessellation_constants.vertex_index_min_max[0] = diff --git a/src/xenia/gpu/vulkan/vulkan_command_processor.h b/src/xenia/gpu/vulkan/vulkan_command_processor.h index 6ac3c0c05..cef3cf33f 100644 --- a/src/xenia/gpu/vulkan/vulkan_command_processor.h +++ b/src/xenia/gpu/vulkan/vulkan_command_processor.h @@ -788,6 +788,10 @@ class VulkanCommandProcessor final : public CommandProcessor { // Whether up-to-date data has been written to constant (uniform) buffers, and // the buffer infos in current_constant_buffer_infos_ point to them. uint32_t current_constant_buffers_up_to_date_; + // The index endian the tessellation constant buffer was filled with, from the + // primitive processor for the current draw. Not a register, so changes + // between draws invalidate the buffer separately from WriteRegister. + xenos::Endian current_tessellation_index_endian_ = xenos::Endian::kNone; VkDescriptorSet current_graphics_descriptor_sets_ [SpirvShaderTranslator::kDescriptorSetCount]; // Whether descriptor sets in current_graphics_descriptor_sets_ point to diff --git a/src/xenia/gpu/vulkan/vulkan_pipeline_cache.cc b/src/xenia/gpu/vulkan/vulkan_pipeline_cache.cc index bb22930ce..d978271e0 100644 --- a/src/xenia/gpu/vulkan/vulkan_pipeline_cache.cc +++ b/src/xenia/gpu/vulkan/vulkan_pipeline_cache.cc @@ -409,6 +409,12 @@ VulkanPipelineCache::GetCurrentVertexShaderModification( modification.vertex.interpolator_mask = interpolator_mask; + // Tessellation mode selects the domain shader spacing. + if (Shader::IsHostVertexShaderTypeDomain(host_vertex_shader_type)) { + modification.vertex.tessellation_mode = + regs.Get().tess_mode; + } + // User clip planes. auto pa_cl_clip_cntl = regs.Get(); uint32_t user_clip_planes =