[Vulkan] Implement alpha to coverage (FBO)

Fixes the foliage rendering in places like Red Dead Redemption menu
This commit is contained in:
Herman S.
2025-11-19 10:55:06 +09:00
parent 75589de0a4
commit cba315cf38
4 changed files with 366 additions and 3 deletions

View File

@@ -272,6 +272,7 @@ void SpirvShaderTranslator::StartTranslation() {
type_uint_},
{"alpha_test_reference", offsetof(SystemConstants, alpha_test_reference),
type_float_},
{"alpha_to_mask", offsetof(SystemConstants, alpha_to_mask), type_uint_},
{"edram_32bpp_tile_pitch_dwords_scaled",
offsetof(SystemConstants, edram_32bpp_tile_pitch_dwords_scaled),
type_uint_},
@@ -1982,7 +1983,12 @@ void SpirvShaderTranslator::StartFragmentShaderBeforeMain() {
// TODO(Triang3l): More conditions - alpha to coverage (if RT 0 is written,
// and there's no early depth / stencil), depth writing in the fragment shader
// (per-sample if supported).
if (edram_fragment_shader_interlock_ || param_gen_needed) {
// Also needed for alpha-to-mask dithering in FBO mode.
bool need_frag_coord =
edram_fragment_shader_interlock_ || param_gen_needed ||
(!edram_fragment_shader_interlock_ && !is_depth_only_fragment_shader_ &&
current_shader().writes_color_target(0));
if (need_frag_coord) {
input_fragment_coordinates_ = builder_->createVariable(
spv::NoPrecision, spv::StorageClassInput, type_float4_, "gl_FragCoord");
builder_->addDecoration(input_fragment_coordinates_, spv::DecorationBuiltIn,
@@ -2048,6 +2054,22 @@ void SpirvShaderTranslator::StartFragmentShaderBeforeMain() {
}
}
}
// Sample mask output for alpha-to-coverage.
// Only needed for non-FSI mode. FSI uses main_fsi_sample_mask_ instead.
output_fragment_sample_mask_ = spv::NoResult;
if (!edram_fragment_shader_interlock_ && !is_depth_only_fragment_shader_) {
// gl_SampleMask is an array of int in SPIR-V.
spv::Id type_sample_mask_array =
builder_->makeArrayType(type_int_, builder_->makeUintConstant(1), 0);
output_fragment_sample_mask_ =
builder_->createVariable(spv::NoPrecision, spv::StorageClassOutput,
type_sample_mask_array, "gl_SampleMask");
builder_->addDecoration(output_fragment_sample_mask_,
spv::DecorationBuiltIn,
static_cast<int>(spv::BuiltIn::SampleMask));
main_interface_.push_back(output_fragment_sample_mask_);
}
}
void SpirvShaderTranslator::StartFragmentShaderInMain() {

View File

@@ -214,7 +214,10 @@ class SpirvShaderTranslator : public ShaderTranslator {
float alpha_test_reference;
uint32_t edram_32bpp_tile_pitch_dwords_scaled;
uint32_t edram_depth_base_dwords_scaled;
float padding_edram_depth_base_dwords_scaled;
// If alpha to mask is disabled, the entire alpha_to_mask value must be 0.
// If alpha to mask is enabled, bits 0:7 are sample offsets, and bit 8 must
// be 1.
uint32_t alpha_to_mask;
float color_exp_bias[4];
@@ -696,6 +699,16 @@ class SpirvShaderTranslator : public ShaderTranslator {
// flow because of taking derivatives of the fragment depth.
void FSI_DepthStencilTest(spv::Id msaa_samples,
bool sample_mask_potentially_narrowed_previouly);
// Alpha to coverage helper - tests one sample.
// coverage_out is modified to include this sample if it passes.
void FSI_AlphaToMaskSample(bool initialize, uint32_t sample_index,
float threshold_base, spv::Id threshold_offset,
float threshold_offset_scale,
spv::Id& coverage_out);
// Alpha to coverage main function.
void FSI_AlphaToMask();
// Returns the first and the second 32 bits as two uints.
std::array<spv::Id, 2> FSI_ClampAndPackColor(spv::Id color_float4,
spv::Id format_with_flags);
@@ -838,6 +851,7 @@ class SpirvShaderTranslator : public ShaderTranslator {
kSystemConstantTextureSwizzles,
kSystemConstantTexturesResolved,
kSystemConstantAlphaTestReference,
kSystemConstantAlphaToMask,
kSystemConstantEdram32bppTilePitchDwordsScaled,
kSystemConstantEdramDepthBaseDwordsScaled,
kSystemConstantColorExpBias,
@@ -915,6 +929,11 @@ class SpirvShaderTranslator : public ShaderTranslator {
std::array<spv::Id, xenos::kMaxColorRenderTargets>
output_or_var_fragment_data_;
// Fragment shader sample mask output (gl_SampleMask).
// Only used for alpha-to-coverage in non-FSI mode.
// For FSI mode, sample mask is handled via main_fsi_sample_mask_.
spv::Id output_fragment_sample_mask_;
std::vector<spv::Id> main_interface_;
spv::Function* function_main_;
spv::Id main_system_constant_flags_;

View File

@@ -624,7 +624,8 @@ void SpirvShaderTranslator::CompleteFragmentShaderInMain() {
}
if_alpha_test_function_is_non_always.makeEndIf();
// TODO(Triang3l): Alpha to coverage.
// Alpha to coverage.
FSI_AlphaToMask();
if (edram_fragment_shader_interlock_) {
// Close the render target 0 written check.
@@ -3462,5 +3463,319 @@ spv::Id SpirvShaderTranslator::FSI_BlendColorOrAlphaWithUnclampedResult(
return builder_->createOp(spv::OpPhi, value_type, id_vector_temp_);
}
void SpirvShaderTranslator::FSI_AlphaToMaskSample(bool initialize,
uint32_t sample_index,
float threshold_base,
spv::Id threshold_offset,
float threshold_offset_scale,
spv::Id& coverage_out) {
// Based on D3D12's CompletePixelShader_AlphaToMaskSample.
// Calculates threshold and tests alpha against it.
// threshold = threshold_base + threshold_offset * (-threshold_offset_scale)
spv::Id const_threshold_offset_scale =
builder_->makeFloatConstant(-threshold_offset_scale);
spv::Id threshold = builder_->createNoContractionBinOp(
spv::OpFMul, type_float_, threshold_offset, const_threshold_offset_scale);
spv::Id const_threshold_base = builder_->makeFloatConstant(threshold_base);
threshold = builder_->createNoContractionBinOp(
spv::OpFAdd, type_float_, const_threshold_base, threshold);
// Load alpha from RT0.w (oC0.w)
// Use access chain to get pointer to alpha component, then load
assert_true(output_or_var_fragment_data_[0] != spv::NoResult);
id_vector_temp_.clear();
id_vector_temp_.push_back(builder_->makeIntConstant(3)); // W component
spv::Id alpha = builder_->createLoad(
builder_->createAccessChain(
edram_fragment_shader_interlock_ ? spv::StorageClassFunction
: spv::StorageClassOutput,
output_or_var_fragment_data_[0], id_vector_temp_),
spv::NoPrecision);
// Test: alpha >= threshold
// Using OpFOrdGreaterThanEqual for proper NaN handling (NaN results in false)
spv::Id sample_passes = builder_->createBinOp(spv::OpFOrdGreaterThanEqual,
type_bool_, alpha, threshold);
if (edram_fragment_shader_interlock_) {
// FSI mode: Start with full coverage, clear bits for failed samples.
if (initialize) {
coverage_out = builder_->makeUintConstant(0xFFFFFFFFu);
}
// If sample fails, clear its bit from coverage mask.
// Create a mask with the sample bit cleared: ~(1 << sample_index)
spv::Id sample_bit = builder_->makeUintConstant(1u << sample_index);
spv::Id clear_mask =
builder_->createUnaryOp(spv::OpNot, type_uint_, sample_bit);
// coverage = select(sample_passes, coverage, coverage & clear_mask)
spv::Id coverage_cleared = builder_->createBinOp(
spv::OpBitwiseAnd, type_uint_, coverage_out, clear_mask);
coverage_out =
builder_->createTriOp(spv::OpSelect, type_uint_, sample_passes,
coverage_out, coverage_cleared);
} else {
// Non-FSI mode: Start with zero coverage, set bits for passed samples.
if (initialize) {
coverage_out = builder_->makeUintConstant(0u);
}
// If sample passes, set its bit in coverage mask.
// Create a mask with the sample bit set: (1 << sample_index)
spv::Id sample_bit = builder_->makeUintConstant(1u << sample_index);
// coverage = select(sample_passes, coverage | sample_bit, coverage)
spv::Id coverage_set = builder_->createBinOp(spv::OpBitwiseOr, type_uint_,
coverage_out, sample_bit);
coverage_out = builder_->createTriOp(
spv::OpSelect, type_uint_, sample_passes, coverage_set, coverage_out);
}
}
void SpirvShaderTranslator::FSI_AlphaToMask() {
// Based on D3D12's CompletePixelShader_AlphaToMask.
// TODO(has207): Implement alpha-to-mask for FSI mode.
// FSI mode requires proper Phi nodes to merge coverage values from different
// MSAA branches, which needs more careful SSA handling.
if (edram_fragment_shader_interlock_) {
return;
}
// Check if alpha to coverage can be done at all in this shader.
if (!current_shader().writes_color_target(0)) {
return;
}
// For FBO mode, ensure gl_SampleMask output was created
// For depth-only shaders, we don't create gl_SampleMask, so skip
if (output_fragment_sample_mask_ == spv::NoResult) {
return;
}
// Initialize OMask to full coverage (in case alpha to mask is disabled).
// gl_SampleMask is an array, so we need to access element [0].
id_vector_temp_.clear();
id_vector_temp_.push_back(builder_->makeIntConstant(0));
spv::Id sample_mask_element = builder_->createAccessChain(
spv::StorageClassOutput, output_fragment_sample_mask_, id_vector_temp_);
spv::Id full_coverage = builder_->makeIntConstant(-1); // All bits set
builder_->createStore(full_coverage, sample_mask_element);
// Load alpha_to_mask constant and check if enabled.
id_vector_temp_.clear();
id_vector_temp_.push_back(
builder_->makeIntConstant(kSystemConstantAlphaToMask));
spv::Id alpha_to_mask_constant = builder_->createLoad(
builder_->createAccessChain(spv::StorageClassUniform,
uniform_system_constants_, id_vector_temp_),
spv::NoPrecision);
spv::Id alpha_to_mask_enabled = builder_->createBinOp(
spv::OpINotEqual, type_bool_, alpha_to_mask_constant,
builder_->makeUintConstant(0));
spv::Block& block_alpha_to_mask_enabled = builder_->makeNewBlock();
spv::Block& block_alpha_to_mask_merge = builder_->makeNewBlock();
builder_->createSelectionMerge(&block_alpha_to_mask_merge,
spv::SelectionControlDontFlattenMask);
builder_->createConditionalBranch(alpha_to_mask_enabled,
&block_alpha_to_mask_enabled,
&block_alpha_to_mask_merge);
builder_->setBuildPoint(&block_alpha_to_mask_enabled);
// Extract dithering threshold offset from fragment position.
// offset_index = (Y & 1) | ((X & 1) << 1)
// offset = (alpha_to_mask >> (offset_index * 2)) & 0b11
spv::Id frag_coord =
builder_->createLoad(input_fragment_coordinates_, spv::NoPrecision);
// Extract X and Y as floats first, then convert to uint
spv::Id frag_x_float =
builder_->createCompositeExtract(frag_coord, type_float_, 0);
spv::Id frag_y_float =
builder_->createCompositeExtract(frag_coord, type_float_, 1);
spv::Id frag_x =
builder_->createUnaryOp(spv::OpConvertFToU, type_uint_, frag_x_float);
spv::Id frag_y =
builder_->createUnaryOp(spv::OpConvertFToU, type_uint_, frag_y_float);
// Y & 1
spv::Id y_bit = builder_->createBinOp(spv::OpBitwiseAnd, type_uint_, frag_y,
builder_->makeUintConstant(1));
// (X & 1) << 1
spv::Id x_bit = builder_->createBinOp(spv::OpBitwiseAnd, type_uint_, frag_x,
builder_->makeUintConstant(1));
spv::Id x_bit_shifted =
builder_->createBinOp(spv::OpShiftLeftLogical, type_uint_, x_bit,
builder_->makeUintConstant(1));
// offset_index = y_bit | x_bit_shifted
spv::Id offset_index =
builder_->createBinOp(spv::OpBitwiseOr, type_uint_, y_bit, x_bit_shifted);
// bit_position = offset_index * 2
spv::Id bit_position =
builder_->createBinOp(spv::OpShiftLeftLogical, type_uint_, offset_index,
builder_->makeUintConstant(1));
// Extract 2-bit offset: (alpha_to_mask >> bit_position) & 0b11
spv::Id offset_shifted =
builder_->createBinOp(spv::OpShiftRightLogical, type_uint_,
alpha_to_mask_constant, bit_position);
spv::Id threshold_offset_uint =
builder_->createBinOp(spv::OpBitwiseAnd, type_uint_, offset_shifted,
builder_->makeUintConstant(0b11));
// Convert offset to float (0.0, 1.0, 2.0, 3.0)
spv::Id threshold_offset = builder_->createUnaryOp(
spv::OpConvertUToF, type_float_, threshold_offset_uint);
// Load MSAA sample count to determine which mode to use.
// 0 = 1x, 1 = 2x, 2 = 4x
spv::Id msaa_samples = LoadMsaaSamplesFromFlags();
// Check if MSAA is enabled (msaa_samples != 0)
spv::Id msaa_enabled =
builder_->createBinOp(spv::OpINotEqual, type_bool_, msaa_samples,
builder_->makeUintConstant(0));
spv::Block& block_msaa_head = builder_->makeNewBlock();
spv::Block& block_msaa_enabled = builder_->makeNewBlock();
spv::Block& block_msaa_disabled = builder_->makeNewBlock();
spv::Block& block_msaa_merge = builder_->makeNewBlock();
builder_->createSelectionMerge(&block_msaa_merge,
spv::SelectionControlDontFlattenMask);
builder_->createConditionalBranch(msaa_enabled, &block_msaa_enabled,
&block_msaa_disabled);
// MSAA enabled - check if 4x or 2x
builder_->setBuildPoint(&block_msaa_enabled);
// msaa_samples: 1 = 2x, 2 = 4x
spv::Id is_4x_msaa = builder_->createBinOp(
spv::OpIEqual, type_bool_, msaa_samples, builder_->makeUintConstant(2));
spv::Block& block_4x_msaa = builder_->makeNewBlock();
spv::Block& block_2x_msaa = builder_->makeNewBlock();
spv::Block& block_msaa_mode_merge = builder_->makeNewBlock();
builder_->createSelectionMerge(&block_msaa_mode_merge,
spv::SelectionControlDontFlattenMask);
builder_->createConditionalBranch(is_4x_msaa, &block_4x_msaa, &block_2x_msaa);
// 4x MSAA
builder_->setBuildPoint(&block_4x_msaa);
spv::Id coverage_4x = spv::NoResult;
{
FSI_AlphaToMaskSample(true, 0, 0.75f, threshold_offset, 1.0f / 16.0f,
coverage_4x);
FSI_AlphaToMaskSample(false, 1, 0.25f, threshold_offset, 1.0f / 16.0f,
coverage_4x);
FSI_AlphaToMaskSample(false, 2, 0.5f, threshold_offset, 1.0f / 16.0f,
coverage_4x);
FSI_AlphaToMaskSample(false, 3, 1.0f, threshold_offset, 1.0f / 16.0f,
coverage_4x);
if (!edram_fragment_shader_interlock_) {
// Write to gl_SampleMask[0] and discard if zero coverage
id_vector_temp_.clear();
id_vector_temp_.push_back(builder_->makeIntConstant(0));
spv::Id sample_mask_element = builder_->createAccessChain(
spv::StorageClassOutput, output_fragment_sample_mask_,
id_vector_temp_);
spv::Id coverage_int =
builder_->createUnaryOp(spv::OpBitcast, type_int_, coverage_4x);
builder_->createStore(coverage_int, sample_mask_element);
// Discard if coverage is zero
spv::Id coverage_zero =
builder_->createBinOp(spv::OpIEqual, type_bool_, coverage_4x,
builder_->makeUintConstant(0));
SpirvBuilder::IfBuilder coverage_discard_if(
coverage_zero, spv::SelectionControlDontFlattenMask, *builder_);
builder_->createNoResultOp(spv::OpKill);
// OpKill terminates the block, so makeEndIf with false to not branch
coverage_discard_if.makeEndIf(false);
}
}
builder_->createBranch(&block_msaa_mode_merge);
// 2x MSAA
builder_->setBuildPoint(&block_2x_msaa);
spv::Id coverage_2x = spv::NoResult;
{
// Using guest sample indices (0 and 1) for both FSI and non-FSI.
FSI_AlphaToMaskSample(true, 0, 0.5f, threshold_offset, 1.0f / 8.0f,
coverage_2x);
FSI_AlphaToMaskSample(false, 1, 1.0f, threshold_offset, 1.0f / 8.0f,
coverage_2x);
if (!edram_fragment_shader_interlock_) {
id_vector_temp_.clear();
id_vector_temp_.push_back(builder_->makeIntConstant(0));
spv::Id sample_mask_element = builder_->createAccessChain(
spv::StorageClassOutput, output_fragment_sample_mask_,
id_vector_temp_);
spv::Id coverage_int =
builder_->createUnaryOp(spv::OpBitcast, type_int_, coverage_2x);
builder_->createStore(coverage_int, sample_mask_element);
spv::Id coverage_zero =
builder_->createBinOp(spv::OpIEqual, type_bool_, coverage_2x,
builder_->makeUintConstant(0));
SpirvBuilder::IfBuilder coverage_discard_if(
coverage_zero, spv::SelectionControlDontFlattenMask, *builder_);
builder_->createNoResultOp(spv::OpKill);
coverage_discard_if.makeEndIf(false);
}
}
builder_->createBranch(&block_msaa_mode_merge);
builder_->setBuildPoint(&block_msaa_mode_merge);
builder_->createBranch(&block_msaa_merge);
// MSAA disabled - single sample
builder_->setBuildPoint(&block_msaa_disabled);
spv::Id coverage_1x = spv::NoResult;
{
FSI_AlphaToMaskSample(true, 0, 1.0f, threshold_offset, 1.0f / 4.0f,
coverage_1x);
if (!edram_fragment_shader_interlock_) {
id_vector_temp_.clear();
id_vector_temp_.push_back(builder_->makeIntConstant(0));
spv::Id sample_mask_element = builder_->createAccessChain(
spv::StorageClassOutput, output_fragment_sample_mask_,
id_vector_temp_);
spv::Id coverage_int =
builder_->createUnaryOp(spv::OpBitcast, type_int_, coverage_1x);
builder_->createStore(coverage_int, sample_mask_element);
spv::Id coverage_zero =
builder_->createBinOp(spv::OpIEqual, type_bool_, coverage_1x,
builder_->makeUintConstant(0));
SpirvBuilder::IfBuilder coverage_discard_if(
coverage_zero, spv::SelectionControlDontFlattenMask, *builder_);
builder_->createNoResultOp(spv::OpKill);
coverage_discard_if.makeEndIf(false);
}
}
builder_->createBranch(&block_msaa_merge);
builder_->setBuildPoint(&block_msaa_merge);
builder_->createBranch(&block_alpha_to_mask_merge);
builder_->setBuildPoint(&block_alpha_to_mask_merge);
}
} // namespace gpu
} // namespace xe

View File

@@ -4198,6 +4198,13 @@ void VulkanCommandProcessor::UpdateSystemConstantValues(
dirty |= system_constants_.alpha_test_reference != rb_alpha_ref;
system_constants_.alpha_test_reference = rb_alpha_ref;
// Alpha to coverage.
uint32_t alpha_to_mask = rb_colorcontrol.alpha_to_mask_enable
? (rb_colorcontrol.value >> 24) | (1 << 8)
: 0;
dirty |= system_constants_.alpha_to_mask != alpha_to_mask;
system_constants_.alpha_to_mask = alpha_to_mask;
uint32_t edram_tile_dwords_scaled =
xenos::kEdramTileWidthSamples * xenos::kEdramTileHeightSamples *
(draw_resolution_scale_x * draw_resolution_scale_y);