Files
Xenia-Canary/src/xenia/gpu/dxbc_shader_translator_om.cc
2022-06-07 21:26:34 +03:00

3471 lines
158 KiB
C++

/**
******************************************************************************
* Xenia : Xbox 360 Emulator Research Project *
******************************************************************************
* Copyright 2021 Ben Vanik. All rights reserved. *
* Released under the BSD license - see LICENSE in the root for more details. *
******************************************************************************
*/
#include "xenia/gpu/dxbc_shader_translator.h"
#include <cstdint>
#include "xenia/base/assert.h"
#include "xenia/base/math.h"
#include "xenia/gpu/draw_util.h"
#include "xenia/gpu/texture_cache.h"
namespace xe {
namespace gpu {
using namespace ucode;
void DxbcShaderTranslator::ROV_GetColorFormatSystemConstants(
xenos::ColorRenderTargetFormat format, uint32_t write_mask,
float& clamp_rgb_low, float& clamp_alpha_low, float& clamp_rgb_high,
float& clamp_alpha_high, uint32_t& keep_mask_low,
uint32_t& keep_mask_high) {
keep_mask_low = keep_mask_high = 0;
switch (format) {
case xenos::ColorRenderTargetFormat::k_8_8_8_8:
case xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA: {
clamp_rgb_low = clamp_alpha_low = 0.0f;
clamp_rgb_high = clamp_alpha_high = 1.0f;
for (uint32_t i = 0; i < 4; ++i) {
if (!(write_mask & (1 << i))) {
keep_mask_low |= uint32_t(0xFF) << (i * 8);
}
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10: {
clamp_rgb_low = clamp_alpha_low = 0.0f;
clamp_rgb_high = clamp_alpha_high = 1.0f;
for (uint32_t i = 0; i < 3; ++i) {
if (!(write_mask & (1 << i))) {
keep_mask_low |= uint32_t(0x3FF) << (i * 10);
}
}
if (!(write_mask & 0b1000)) {
keep_mask_low |= uint32_t(3) << 30;
}
} break;
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT:
case xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT_AS_16_16_16_16: {
clamp_rgb_low = clamp_alpha_low = 0.0f;
clamp_rgb_high = 31.875f;
clamp_alpha_high = 1.0f;
for (uint32_t i = 0; i < 3; ++i) {
if (!(write_mask & (1 << i))) {
keep_mask_low |= uint32_t(0x3FF) << (i * 10);
}
}
if (!(write_mask & 0b1000)) {
keep_mask_low |= uint32_t(3) << 30;
}
} break;
case xenos::ColorRenderTargetFormat::k_16_16:
case xenos::ColorRenderTargetFormat::k_16_16_16_16:
// Alpha clamping affects blending source, so it's non-zero for alpha for
// k_16_16 (the render target is fixed-point). There's one deviation from
// how Direct3D 11.3 functional specification defines SNorm conversion
// (NaN should be 0, not the lowest negative number), but NaN handling in
// output shouldn't be very important.
clamp_rgb_low = clamp_alpha_low = -32.0f;
clamp_rgb_high = clamp_alpha_high = 32.0f;
if (!(write_mask & 0b0001)) {
keep_mask_low |= 0xFFFFu;
}
if (!(write_mask & 0b0010)) {
keep_mask_low |= 0xFFFF0000u;
}
if (format == xenos::ColorRenderTargetFormat::k_16_16_16_16) {
if (!(write_mask & 0b0100)) {
keep_mask_high |= 0xFFFFu;
}
if (!(write_mask & 0b1000)) {
keep_mask_high |= 0xFFFF0000u;
}
} else {
write_mask &= 0b0011;
}
break;
case xenos::ColorRenderTargetFormat::k_16_16_FLOAT:
case xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT:
// No NaNs on the Xbox 360 GPU, though can't use the extended range with
// f32tof16.
clamp_rgb_low = clamp_alpha_low = -65504.0f;
clamp_rgb_high = clamp_alpha_high = 65504.0f;
if (!(write_mask & 0b0001)) {
keep_mask_low |= 0xFFFFu;
}
if (!(write_mask & 0b0010)) {
keep_mask_low |= 0xFFFF0000u;
}
if (format == xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT) {
if (!(write_mask & 0b0100)) {
keep_mask_high |= 0xFFFFu;
}
if (!(write_mask & 0b1000)) {
keep_mask_high |= 0xFFFF0000u;
}
} else {
write_mask &= 0b0011;
}
break;
case xenos::ColorRenderTargetFormat::k_32_FLOAT:
// No clamping - let min/max always pick the original value.
clamp_rgb_low = clamp_alpha_low = clamp_rgb_high = clamp_alpha_high =
std::nanf("");
write_mask &= 0b0001;
if (!(write_mask & 0b0001)) {
keep_mask_low = ~uint32_t(0);
}
break;
case xenos::ColorRenderTargetFormat::k_32_32_FLOAT:
// No clamping - let min/max always pick the original value.
clamp_rgb_low = clamp_alpha_low = clamp_rgb_high = clamp_alpha_high =
std::nanf("");
write_mask &= 0b0011;
if (!(write_mask & 0b0001)) {
keep_mask_low = ~uint32_t(0);
}
if (!(write_mask & 0b0010)) {
keep_mask_high = ~uint32_t(0);
}
break;
default:
assert_unhandled_case(format);
// Disable invalid render targets.
write_mask = 0;
break;
}
// Special case handled in the shaders for empty write mask to completely skip
// a disabled render target: all keep bits are set.
if (!write_mask) {
keep_mask_low = keep_mask_high = ~uint32_t(0);
}
}
void DxbcShaderTranslator::StartPixelShader_LoadROVParameters() {
bool any_color_targets_written = current_shader().writes_color_targets() != 0;
// ***************************************************************************
// Get EDRAM offsets for the pixel:
// system_temp_rov_params_.y - for depth (absolute).
// system_temp_rov_params_.z - for 32bpp color (base-relative).
// system_temp_rov_params_.w - for 64bpp color (base-relative).
// ***************************************************************************
// For now, while we don't know the encoding of 64bpp render targets when
// interpreted as 32bpp (no game has been seen reinterpreting between the two
// yet), for consistency with the conventional render target logic and to have
// the same resolve logic for both, storing 64bpp color as 40x16 samples
// (multiplied by the resolution scale) per 1280-byte tile. It's also
// convenient to use 40x16 granularity in the calculations here because depth
// render targets have 40-sample halves swapped as opposed to color in each
// tile, and reinterpretation between depth and color is common for depth /
// stencil reloading into the EDRAM (such as in the background of the main
// menu of 4D5307E6).
// Convert the host pixel position to integer to system_temp_rov_params_.xy.
// system_temp_rov_params_.x = X host pixel position as uint
// system_temp_rov_params_.y = Y host pixel position as uint
in_position_used_ |= 0b0011;
a_.OpFToU(dxbc::Dest::R(system_temp_rov_params_, 0b0011),
dxbc::Src::V1D(uint32_t(InOutRegister::kPSInPosition)));
// Convert the position from pixels to samples.
// system_temp_rov_params_.x = X sample 0 position
// system_temp_rov_params_.y = Y sample 0 position
a_.OpIShL(
dxbc::Dest::R(system_temp_rov_params_, 0b0011),
dxbc::Src::R(system_temp_rov_params_),
LoadSystemConstant(SystemConstants::Index::kSampleCountLog2,
offsetof(SystemConstants, sample_count_log2), 0b0100));
// For cases of both color and depth:
// Get 40 x 16 x resolution scale 32bpp half-tile or 40x16 64bpp tile index
// to system_temp_rov_params_.zw, and put the sample index within such a
// region in system_temp_rov_params_.xy.
// Working with 40x16-sample portions for 64bpp and for swapping for depth -
// dividing by 40, not by 80.
// For depth-only:
// Same, but for full 80x16 tiles, not 40x16 half-tiles.
uint32_t tile_or_half_tile_width = 80 * draw_resolution_scale_x_;
uint32_t tile_or_half_tile_width_divide_scale;
uint32_t tile_or_half_tile_width_divide_upper_shift;
draw_util::GetEdramTileWidthDivideScaleAndUpperShift(
draw_resolution_scale_x_, tile_or_half_tile_width_divide_scale,
tile_or_half_tile_width_divide_upper_shift);
if (any_color_targets_written) {
tile_or_half_tile_width >>= 1;
assert_not_zero(tile_or_half_tile_width_divide_upper_shift);
--tile_or_half_tile_width_divide_upper_shift;
}
static_assert(
TextureCache::kMaxDrawResolutionScaleAlongAxis <= 3,
"DxbcShaderTranslator ROV sample address calculation supports Y draw "
"resolution scaling factors of only up to 3");
if (draw_resolution_scale_y_ == 3) {
// Multiplication part of the division by 40|80 x 16 x scale (specifically
// 40|80 * scale width here, and 48 height, or 16 * 3 height).
// system_temp_rov_params_.x = X sample 0 position
// system_temp_rov_params_.y = Y sample 0 position
// system_temp_rov_params_.z = (X * tile_or_half_tile_width_divide_scale) >>
// 32
// system_temp_rov_params_.w = (Y * kDivideScale3) >> 32
a_.OpUMul(dxbc::Dest::R(system_temp_rov_params_, 0b1100),
dxbc::Dest::Null(),
dxbc::Src::R(system_temp_rov_params_, 0b0100 << 4),
dxbc::Src::LU(0, 0, tile_or_half_tile_width_divide_scale,
draw_util::kDivideScale3));
// Shift part of the division by 40|80 x 16 x scale.
// system_temp_rov_params_.x = X sample 0 position
// system_temp_rov_params_.y = Y sample 0 position
// system_temp_rov_params_.z = X half-tile or tile position
// system_temp_rov_params_.w = Y tile position
a_.OpUShR(dxbc::Dest::R(system_temp_rov_params_, 0b1100),
dxbc::Src::R(system_temp_rov_params_),
dxbc::Src::LU(0, 0, tile_or_half_tile_width_divide_upper_shift,
draw_util::kDivideUpperShift3 + 4));
// Take the remainder of the performed division to
// system_temp_rov_params_.xy.
// system_temp_rov_params_.x = X sample 0 position within the half-tile
// system_temp_rov_params_.y = Y sample 0 position within the (half-)tile
// system_temp_rov_params_.z = X half-tile or tile position
// system_temp_rov_params_.w = Y tile position
a_.OpIMAd(dxbc::Dest::R(system_temp_rov_params_, 0b0011),
dxbc::Src::R(system_temp_rov_params_, 0b1110),
dxbc::Src::LI(-int32_t(tile_or_half_tile_width),
-16 * draw_resolution_scale_y_, 0, 0),
dxbc::Src::R(system_temp_rov_params_));
} else {
assert_true(draw_resolution_scale_y_ <= 2);
// Multiplication part of the division of X by 40|80 * scale.
// system_temp_rov_params_.x = X sample 0 position
// system_temp_rov_params_.y = Y sample 0 position
// system_temp_rov_params_.z = (X * tile_or_half_tile_width_divide_scale) >>
// 32
a_.OpUMul(dxbc::Dest::R(system_temp_rov_params_, 0b0100),
dxbc::Dest::Null(),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(tile_or_half_tile_width_divide_scale));
// Shift part of the division of X by 40 * scale, division of Y by
// 16 * scale as it's power of two in this case.
// system_temp_rov_params_.x = X sample 0 position
// system_temp_rov_params_.y = Y sample 0 position
// system_temp_rov_params_.z = X half-tile or tile position
// system_temp_rov_params_.w = Y tile position
a_.OpUShR(dxbc::Dest::R(system_temp_rov_params_, 0b1100),
dxbc::Src::R(system_temp_rov_params_, 0b0110 << 4),
dxbc::Src::LU(0, 0, tile_or_half_tile_width_divide_upper_shift,
draw_resolution_scale_y_ == 2 ? 5 : 4));
// Take the remainder of the performed division (via multiply-subtract for
// X, via AND for Y which is power-of-two here) to
// system_temp_rov_params_.xy.
// system_temp_rov_params_.x = X sample 0 position within the half-tile or
// tile
// system_temp_rov_params_.y = Y sample 0 position within the (half-)tile
// system_temp_rov_params_.z = X half-tile or tile position
// system_temp_rov_params_.w = Y tile position
a_.OpIMAd(dxbc::Dest::R(system_temp_rov_params_, 0b0001),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kZZZZ),
dxbc::Src::LI(-int32_t(tile_or_half_tile_width)),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX));
a_.OpAnd(dxbc::Dest::R(system_temp_rov_params_, 0b0010),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY),
dxbc::Src::LU((16 * draw_resolution_scale_y_) - 1));
}
// Convert the Y sample 0 position within the half-tile or tile to the dword
// offset of the row within a 80x16 32bpp tile or a 40x16 64bpp half-tile to
// system_temp_rov_params_.y.
// system_temp_rov_params_.x = X sample 0 position within the half-tile or
// tile
// system_temp_rov_params_.y = Y sample 0 row dword offset within the
// 80x16-dword tile
// system_temp_rov_params_.z = X half-tile position
// system_temp_rov_params_.w = Y tile position
a_.OpUMul(dxbc::Dest::Null(), dxbc::Dest::R(system_temp_rov_params_, 0b0010),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY),
dxbc::Src::LU(80 * draw_resolution_scale_x_));
if (any_color_targets_written) {
// Depth, 32bpp color, 64bpp color are all needed.
// X sample 0 position within in the half-tile in system_temp_rov_params_.x,
// for 64bpp, will be used directly as sample X the within the 80x16-dword
// region, but for 32bpp color and depth, 40x16 half-tile index within the
// 80x16 tile - system_temp_rov_params_.z & 1 - will also be taken into
// account when calculating the X (directly for color, flipped for depth).
uint32_t rov_address_temp = PushSystemTemp();
// Multiply the Y tile position by the surface tile pitch in dwords to get
// the address of the origin of the row of tiles within a 32bpp surface in
// dwords (later it needs to be multiplied by 2 for 64bpp).
// system_temp_rov_params_.x = X sample 0 position within the half-tile
// system_temp_rov_params_.y = Y sample 0 row dword offset within the
// 80x16-dword tile
// system_temp_rov_params_.z = X half-tile position
// system_temp_rov_params_.w = Y tile row dword origin in a 32bpp surface
a_.OpUMul(
dxbc::Dest::Null(), dxbc::Dest::R(system_temp_rov_params_, 0b1000),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kWWWW),
LoadSystemConstant(
SystemConstants::Index::kEdram32bppTilePitchDwordsScaled,
offsetof(SystemConstants, edram_32bpp_tile_pitch_dwords_scaled),
dxbc::Src::kXXXX));
// Get the 32bpp tile X position within the row of tiles to
// rov_address_temp.x.
// system_temp_rov_params_.x = X sample 0 position within the half-tile
// system_temp_rov_params_.y = Y sample 0 row dword offset within the
// 80x16-dword tile
// system_temp_rov_params_.z = X half-tile position
// system_temp_rov_params_.w = Y tile row dword origin in a 32bpp surface
// rov_address_temp.x = X 32bpp tile position
a_.OpUShR(dxbc::Dest::R(rov_address_temp, 0b0001),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kZZZZ),
dxbc::Src::LU(1));
// Get the dword offset of the beginning of the row of samples within a row
// of 32bpp 80x16 tiles to rov_address_temp.x.
// system_temp_rov_params_.x = X sample 0 position within the half-tile
// system_temp_rov_params_.y = Y sample 0 row dword offset within the
// 80x16-dword tile
// system_temp_rov_params_.z = X half-tile position
// system_temp_rov_params_.w = Y tile row dword origin in a 32bpp surface
// rov_address_temp.x = dword offset of the beginning of the row of samples
// within a row of 32bpp tiles
a_.OpUMAd(
dxbc::Dest::R(rov_address_temp, 0b0001),
dxbc::Src::R(rov_address_temp, dxbc::Src::kXXXX),
dxbc::Src::LU(80 * 16 *
(draw_resolution_scale_x_ * draw_resolution_scale_y_)),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY));
// Get the dword offset of the beginning of the row of samples within a
// 32bpp surface to rov_address_temp.x.
// system_temp_rov_params_.x = X sample 0 position within the half-tile
// system_temp_rov_params_.y = Y sample 0 row dword offset within the
// 80x16-dword tile
// system_temp_rov_params_.z = X half-tile position
// system_temp_rov_params_.w = Y tile row dword origin in a 32bpp surface
// rov_address_temp.x = dword offset of the beginning of the row of samples
// within a 32bpp surface
a_.OpIAdd(dxbc::Dest::R(rov_address_temp, 0b0001),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kWWWW),
dxbc::Src::R(rov_address_temp, dxbc::Src::kXXXX));
// Get the dword offset of the beginning of the row of samples within a row
// of 64bpp 80x16 tiles to system_temp_rov_params_.y (last time the
// tile-local Y offset is needed).
// system_temp_rov_params_.x = X sample 0 position within the half-tile
// system_temp_rov_params_.y = dword offset of the beginning of the row of
// samples within a row of 64bpp tiles
// system_temp_rov_params_.z = X half-tile position
// system_temp_rov_params_.w = Y tile row dword origin in a 32bpp surface
// rov_address_temp.x = dword offset of the beginning of the row of samples
// within a 32bpp surface
a_.OpUMAd(
dxbc::Dest::R(system_temp_rov_params_, 0b0010),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kZZZZ),
dxbc::Src::LU(80 * 16 *
(draw_resolution_scale_x_ * draw_resolution_scale_y_)),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY));
// Get the dword offset of the beginning of the row of samples within a
// 64bpp surface to system_temp_rov_params_.w (last time the Y tile row
// offset is needed).
// system_temp_rov_params_.x = X sample 0 position within the half-tile
// system_temp_rov_params_.y = free
// system_temp_rov_params_.z = X half-tile position
// system_temp_rov_params_.w = dword offset of the beginning of the row of
// samples within a 64bpp surface
// rov_address_temp.x = dword offset of the beginning of the row of samples
// within a 32bpp surface
a_.OpUMAd(dxbc::Dest::R(system_temp_rov_params_, 0b1000),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kWWWW),
dxbc::Src::LU(2),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY));
// Get the final offset of the sample 0 within a 64bpp surface to
// system_temp_rov_params_.w.
// system_temp_rov_params_.x = X sample 0 position within the half-tile
// system_temp_rov_params_.z = X half-tile position
// system_temp_rov_params_.w = dword sample 0 offset within a 64bpp surface
// rov_address_temp.x = dword offset of the beginning of the row of samples
// within a 32bpp surface
a_.OpUMAd(dxbc::Dest::R(system_temp_rov_params_, 0b1000),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(2),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kWWWW));
// Get the half-tile index within the tile to system_temp_rov_params_.y
// (last time the X half-tile position is needed).
// system_temp_rov_params_.x = X sample 0 position within the half-tile
// system_temp_rov_params_.y = half-tile index within the tile
// system_temp_rov_params_.z = free
// system_temp_rov_params_.w = dword sample 0 offset within a 64bpp surface
// rov_address_temp.x = dword offset of the beginning of the row of samples
// within a 32bpp surface
a_.OpAnd(dxbc::Dest::R(system_temp_rov_params_, 0b0010),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kZZZZ),
dxbc::Src::LU(1));
// Get the X position within the 32bpp tile to system_temp_rov_params_.z
// (last time the X position within the half-tile is needed).
// system_temp_rov_params_.x = free
// system_temp_rov_params_.y = half-tile index within the tile
// system_temp_rov_params_.z = X sample 0 position within the tile
// system_temp_rov_params_.w = dword sample 0 offset within a 64bpp surface
// rov_address_temp.x = dword offset of the beginning of the row of samples
// within a 32bpp surface
a_.OpUMAd(dxbc::Dest::R(system_temp_rov_params_, 0b0100),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY),
dxbc::Src::LU(40 * draw_resolution_scale_x_),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX));
// Get the final offset of the sample 0 within a 32bpp color surface to
// system_temp_rov_params_.z (last time the 32bpp row offset is needed).
// system_temp_rov_params_.y = half-tile index within the tile
// system_temp_rov_params_.z = dword sample 0 offset within a 32bpp surface
// system_temp_rov_params_.w = dword sample 0 offset within a 64bpp surface
// rov_address_temp.x = free
a_.OpIAdd(dxbc::Dest::R(system_temp_rov_params_, 0b0100),
dxbc::Src::R(rov_address_temp, dxbc::Src::kXXXX),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kZZZZ));
// Flip the 40x16 half-tiles for depth / stencil as opposed to 32bpp color -
// get the dword offset to add for flipping to system_temp_rov_params_.y.
// system_temp_rov_params_.y = depth half-tile flipping offset
// system_temp_rov_params_.z = dword sample 0 offset within a 32bpp surface
// system_temp_rov_params_.w = dword sample 0 offset within a 64bpp surface
a_.OpMovC(dxbc::Dest::R(system_temp_rov_params_, 0b0010),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY),
dxbc::Src::LI(-40 * draw_resolution_scale_x_),
dxbc::Src::LI(40 * draw_resolution_scale_x_));
// Flip the 40x16 half-tiles for depth / stencil as opposed to 32bpp color -
// get the final offset of the sample 0 within a 32bpp depth / stencil
// surface to system_temp_rov_params_.y.
// system_temp_rov_params_.y = dword sample 0 offset within depth / stencil
// system_temp_rov_params_.z = dword sample 0 offset within a 32bpp surface
// system_temp_rov_params_.w = dword sample 0 offset within a 64bpp surface
a_.OpIAdd(dxbc::Dest::R(system_temp_rov_params_, 0b0010),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kZZZZ),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY));
// Release rov_address_temp.
PopSystemTemp();
} else {
// Simpler logic for depth-only, not involving half-tile indices (flipping
// half-tiles via comparison).
// Get the dword offset of the beginning of the row of samples within a row
// of 32bpp 80x16 tiles to system_temp_rov_params_.z (last time the X tile
// position is needed).
// system_temp_rov_params_.x = X sample 0 position within the tile
// system_temp_rov_params_.y = Y sample 0 row dword offset within the
// 80x16-dword tile
// system_temp_rov_params_.z = dword offset of the beginning of the row of
// samples within a row of 32bpp tiles
// system_temp_rov_params_.w = Y tile position
a_.OpUMAd(
dxbc::Dest::R(system_temp_rov_params_, 0b0100),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kZZZZ),
dxbc::Src::LU(80 * 16 *
(draw_resolution_scale_x_ * draw_resolution_scale_y_)),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY));
// Get the dword offset of the beginning of the row of samples within a
// 32bpp surface to system_temp_rov_params_.y (last time anything Y-related
// is needed, as well as the sample row offset within the tile row).
// system_temp_rov_params_.x = X sample 0 position within the tile
// system_temp_rov_params_.y = dword offset of the beginning of the row of
// samples within a 32bpp surface
// system_temp_rov_params_.z = free
// system_temp_rov_params_.w = free
a_.OpUMAd(
dxbc::Dest::R(system_temp_rov_params_, 0b0010),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kWWWW),
LoadSystemConstant(
SystemConstants::Index::kEdram32bppTilePitchDwordsScaled,
offsetof(SystemConstants, edram_32bpp_tile_pitch_dwords_scaled),
dxbc::Src::kXXXX),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kZZZZ));
// Add the tile-local X to the depth offset in system_temp_rov_params_.y.
// system_temp_rov_params_.x = X sample 0 position within the tile
// system_temp_rov_params_.y = dword sample 0 offset within a 32bpp surface
a_.OpIAdd(dxbc::Dest::R(system_temp_rov_params_, 0b0010),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX));
// Flip the 40x16 half-tiles for depth / stencil as opposed to 32bpp color -
// check in which half-tile the pixel is in to system_temp_rov_params_.x.
// system_temp_rov_params_.x = free
// system_temp_rov_params_.y = dword sample 0 offset within a 32bpp surface
// system_temp_rov_params_.z = 0xFFFFFFFF if in the right half-tile, 0
// otherwise
a_.OpUGE(dxbc::Dest::R(system_temp_rov_params_, 0b0001),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(40 * draw_resolution_scale_x_));
// Flip the 40x16 half-tiles for depth / stencil as opposed to 32bpp color -
// get the dword offset to add for flipping to system_temp_rov_params_.x.
// system_temp_rov_params_.x = depth half-tile flipping offset
// system_temp_rov_params_.y = dword sample 0 offset within a 32bpp surface
a_.OpMovC(dxbc::Dest::R(system_temp_rov_params_, 0b0001),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LI(-40 * draw_resolution_scale_x_),
dxbc::Src::LI(40 * draw_resolution_scale_x_));
// Flip the 40x16 half-tiles for depth / stencil as opposed to 32bpp color -
// get the final offset of the sample 0 within a 32bpp depth / stencil
// surface to system_temp_rov_params_.y.
// system_temp_rov_params_.x = free
// system_temp_rov_params_.y = dword sample 0 offset within depth / stencil
a_.OpIAdd(dxbc::Dest::R(system_temp_rov_params_, 0b0010),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX));
}
// Add the EDRAM base for depth/stencil.
// system_temp_rov_params_.y = EDRAM depth / stencil address
// system_temp_rov_params_.z = dword sample 0 offset within a 32bpp surface if
// needed
// system_temp_rov_params_.w = dword sample 0 offset within a 64bpp surface if
// needed
a_.OpIAdd(dxbc::Dest::R(system_temp_rov_params_, 0b0010),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY),
LoadSystemConstant(
SystemConstants::Index::kEdramDepthBaseDwordsScaled,
offsetof(SystemConstants, edram_depth_base_dwords_scaled),
dxbc::Src::kXXXX));
// ***************************************************************************
// Sample coverage to system_temp_rov_params_.x.
// ***************************************************************************
// Using ForcedSampleCount of 4 (2 is not supported on Nvidia), so for 2x
// MSAA, handling samples 0 and 3 (upper-left and lower-right) as 0 and 1.
// Check if 4x MSAA is enabled.
a_.OpIf(true, LoadSystemConstant(SystemConstants::Index::kSampleCountLog2,
offsetof(SystemConstants, sample_count_log2),
dxbc::Src::kXXXX));
{
// Copy the 4x AA coverage to system_temp_rov_params_.x, making top-right
// the sample [2] and bottom-left the sample [1] (the opposite of Direct3D
// 12), because on the Xbox 360, 2x MSAA doubles the storage width, 4x MSAA
// doubles the storage height.
// Flip samples in bits 0:1 to bits 29:30.
a_.OpBFRev(dxbc::Dest::R(system_temp_rov_params_, 0b0001),
dxbc::Src::VCoverage());
a_.OpUShR(dxbc::Dest::R(system_temp_rov_params_, 0b0001),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(29));
a_.OpBFI(dxbc::Dest::R(system_temp_rov_params_, 0b0001), dxbc::Src::LU(2),
dxbc::Src::LU(1),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::VCoverage());
}
// Handle 1 or 2 samples.
a_.OpElse();
{
// Extract sample 3 coverage, which will be used as sample 1.
a_.OpUBFE(dxbc::Dest::R(system_temp_rov_params_, 0b0001), dxbc::Src::LU(1),
dxbc::Src::LU(3), dxbc::Src::VCoverage());
// Combine coverage of samples 0 (in bit 0 of vCoverage) and 3 (in bit 0 of
// system_temp_rov_params_.x).
a_.OpBFI(dxbc::Dest::R(system_temp_rov_params_, 0b0001), dxbc::Src::LU(31),
dxbc::Src::LU(1),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::VCoverage());
}
// Close the 4x MSAA conditional.
a_.OpEndIf();
}
void DxbcShaderTranslator::ROV_DepthStencilTest() {
uint32_t temp = PushSystemTemp();
dxbc::Dest temp_x_dest(dxbc::Dest::R(temp, 0b0001));
dxbc::Src temp_x_src(dxbc::Src::R(temp, dxbc::Src::kXXXX));
dxbc::Dest temp_y_dest(dxbc::Dest::R(temp, 0b0010));
dxbc::Src temp_y_src(dxbc::Src::R(temp, dxbc::Src::kYYYY));
dxbc::Dest temp_z_dest(dxbc::Dest::R(temp, 0b0100));
dxbc::Src temp_z_src(dxbc::Src::R(temp, dxbc::Src::kZZZZ));
dxbc::Dest temp_w_dest(dxbc::Dest::R(temp, 0b1000));
dxbc::Src temp_w_src(dxbc::Src::R(temp, dxbc::Src::kWWWW));
// Check whether depth/stencil is enabled.
// temp.x = kSysFlag_ROVDepthStencil
a_.OpAnd(temp_x_dest, LoadFlagsSystemConstant(),
dxbc::Src::LU(kSysFlag_ROVDepthStencil));
// Open the depth/stencil enabled conditional.
// temp.x = free
a_.OpIf(true, temp_x_src);
bool shader_writes_depth = current_shader().writes_depth();
bool depth_stencil_early = ROV_IsDepthStencilEarly();
dxbc::Src z_ddx_src(dxbc::Src::LF(0.0f)), z_ddy_src(dxbc::Src::LF(0.0f));
if (shader_writes_depth) {
// Convert the shader-generated depth to 24-bit, using temp.x as
// temporary. oDepth is already written by StoreResult with saturation,
// no need to clamp here. Adreno 200 doesn't have PA_SC_VPORT_ZMIN/ZMAX,
// so likely there's no need to clamp to the viewport depth bounds.
ROV_DepthTo24Bit(system_temp_depth_stencil_, 0, system_temp_depth_stencil_,
0, temp, 0);
} else {
dxbc::Src in_position_z(dxbc::Src::V1D(
uint32_t(InOutRegister::kPSInPosition), dxbc::Src::kZZZZ));
// Get the derivatives of the screen-space (but not clamped to the viewport
// depth bounds yet - this happens after the pixel shader in Direct3D 11+;
// also linear within the triangle - thus constant derivatives along the
// triangle) Z for calculating per-sample depth values and the slope-scaled
// polygon offset.
// We're using derivatives instead of eval_sample_index for various reasons:
// - eval_sample_index doesn't work with SV_Position - need to use an
// additional interpolant.
// - On AMD, eval_sample_index is actually implemented via calculation and
// scaling of derivatives of barycentric coordinates, therefore there's no
// advantage of using it there.
// - eval_sample_index is (inconsistently, but often) one of the sources of
// the infamous AMD shader compiler crashes when ROV is used in Xenia, in
// addition to shader compiler crashes on WARP.
if (depth_stencil_early) {
z_ddx_src = dxbc::Src::R(temp, dxbc::Src::kXXXX);
z_ddy_src = dxbc::Src::R(temp, dxbc::Src::kYYYY);
// temp.x = ddx(z)
// temp.y = ddy(z)
in_position_used_ |= 0b0100;
a_.OpDerivRTXCoarse(temp_x_dest, in_position_z);
a_.OpDerivRTYCoarse(temp_y_dest, in_position_z);
} else {
// For late depth / stencil testing, derivatives are calculated in the
// beginning of the shader before any return statement is possibly
// reached, and written to system_temp_depth_stencil_.xy.
assert_true(system_temp_depth_stencil_ != UINT32_MAX);
z_ddx_src = dxbc::Src::R(system_temp_depth_stencil_, dxbc::Src::kXXXX);
z_ddy_src = dxbc::Src::R(system_temp_depth_stencil_, dxbc::Src::kYYYY);
}
// Get the maximum depth slope for polygon offset.
// https://docs.microsoft.com/en-us/windows/desktop/direct3d9/depth-bias
// temp.x if early = ddx(z)
// temp.y if early = ddy(z)
// temp.z = max(|ddx(z)|, |ddy(z)|)
a_.OpMax(temp_z_dest, z_ddx_src.Abs(), z_ddy_src.Abs());
// Calculate the depth bias for the needed faceness.
in_front_face_used_ = true;
a_.OpIf(true, dxbc::Src::V1D(
uint32_t(InOutRegister::kPSInFrontFaceAndSampleIndex),
dxbc::Src::kXXXX));
// temp.x if early = ddx(z)
// temp.y if early = ddy(z)
// temp.z = front face polygon offset
// temp.w = free
a_.OpMAd(
temp_z_dest, temp_z_src,
LoadSystemConstant(SystemConstants::Index::kEdramPolyOffsetFront,
offsetof(SystemConstants, edram_poly_offset_front),
dxbc::Src::kXXXX),
LoadSystemConstant(SystemConstants::Index::kEdramPolyOffsetFront,
offsetof(SystemConstants, edram_poly_offset_front),
dxbc::Src::kYYYY));
a_.OpElse();
// temp.x if early = ddx(z)
// temp.y if early = ddy(z)
// temp.z = back face polygon offset
// temp.w = free
a_.OpMAd(
temp_z_dest, temp_z_src,
LoadSystemConstant(SystemConstants::Index::kEdramPolyOffsetBack,
offsetof(SystemConstants, edram_poly_offset_back),
dxbc::Src::kXXXX),
LoadSystemConstant(SystemConstants::Index::kEdramPolyOffsetBack,
offsetof(SystemConstants, edram_poly_offset_back),
dxbc::Src::kYYYY));
a_.OpEndIf();
// Apply the post-clip and post-viewport polygon offset to the fragment's
// depth. Not clamping yet as this is at the center, which is not
// necessarily covered and not necessarily inside the bounds - derivatives
// scaled by sample positions will be added to this value, and it must be
// linear.
// temp.x if early = ddx(z)
// temp.y if early = ddy(z)
// temp.z = biased depth in the center
in_position_used_ |= 0b0100;
a_.OpAdd(temp_z_dest, temp_z_src, in_position_z);
}
for (uint32_t i = 0; i < 4; ++i) {
// With early depth/stencil, depth/stencil writing may be deferred to the
// end of the shader to prevent writing in case something (like alpha test,
// which is dynamic GPU state) discards the pixel. So, write directly to the
// persistent register, system_temp_depth_stencil_, instead of a local
// temporary register.
dxbc::Dest sample_depth_stencil_dest(
depth_stencil_early ? dxbc::Dest::R(system_temp_depth_stencil_, 1 << i)
: temp_w_dest);
dxbc::Src sample_depth_stencil_src(
depth_stencil_early ? dxbc::Src::R(system_temp_depth_stencil_).Select(i)
: temp_w_src);
// Get if the current sample is covered.
// temp.x if no oDepth and early = ddx(z)
// temp.y if no oDepth and early = ddy(z)
// temp.z if no oDepth = biased depth in the center
// temp.w = coverage of the current sample
a_.OpAnd(temp_w_dest,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(1 << i));
// Check if the current sample is covered.
// temp.x if no oDepth and early = ddx(z)
// temp.y if no oDepth and early = ddy(z)
// temp.z if no oDepth = biased depth in the center
// temp.w = free
a_.OpIf(true, temp_w_src);
uint32_t sample_temp = PushSystemTemp();
dxbc::Dest sample_temp_x_dest(dxbc::Dest::R(sample_temp, 0b0001));
dxbc::Src sample_temp_x_src(dxbc::Src::R(sample_temp, dxbc::Src::kXXXX));
dxbc::Dest sample_temp_y_dest(dxbc::Dest::R(sample_temp, 0b0010));
dxbc::Src sample_temp_y_src(dxbc::Src::R(sample_temp, dxbc::Src::kYYYY));
dxbc::Dest sample_temp_z_dest(dxbc::Dest::R(sample_temp, 0b0100));
dxbc::Src sample_temp_z_src(dxbc::Src::R(sample_temp, dxbc::Src::kZZZZ));
dxbc::Dest sample_temp_w_dest(dxbc::Dest::R(sample_temp, 0b1000));
dxbc::Src sample_temp_w_src(dxbc::Src::R(sample_temp, dxbc::Src::kWWWW));
if (shader_writes_depth) {
// Copy the 24-bit depth common to all samples to sample_depth_stencil.
// temp.w = shader-generated 24-bit depth
assert_false(depth_stencil_early);
a_.OpMov(sample_depth_stencil_dest,
dxbc::Src::R(system_temp_depth_stencil_, dxbc::Src::kXXXX));
} else {
// Adreno 200 doesn't have PA_SC_VPORT_ZMIN/ZMAX, so likely there's no
// need to clamp to the viewport depth bounds, just to 0...1 - thus only
// saturating in the end of the per-sample depth calculation.
switch (i) {
case 0:
// First sample - off-center for MSAA, in the center without it.
// Using ForcedSampleCount 4 for both 2x and 4x MSAA because
// ForcedSampleCount 2 is not supported on Nvidia, thus the position
// of the top-left sample (0 in Xenia) is always that of the top-left
// sample of host 4x MSAA.
// Calculate the depth in the sample 0 for 2x or 4x MSAA.
// temp.x if early = ddx(z)
// temp.y if early = ddy(z)
// temp.z = biased depth in the center
// temp.w if late = unsaturated sample 0 depth at 4x MSAA
a_.OpMAd(
sample_depth_stencil_dest, z_ddx_src,
dxbc::Src::LF(draw_util::kD3D10StandardSamplePositions4x[0][0] *
(1.0f / 16.0f)),
temp_z_src);
a_.OpMAd(
sample_depth_stencil_dest, z_ddy_src,
dxbc::Src::LF(draw_util::kD3D10StandardSamplePositions4x[0][1] *
(1.0f / 16.0f)),
sample_depth_stencil_src);
// Choose between the sample and the center depth depending on whether
// at least 2x MSAA is enabled and saturate.
// temp.x if early = ddx(z)
// temp.y if early = ddy(z)
// temp.z = biased depth in the center
// temp.w if late = sample 0 depth
a_.OpMovC(
sample_depth_stencil_dest,
LoadSystemConstant(SystemConstants::Index::kSampleCountLog2,
offsetof(SystemConstants, sample_count_log2),
dxbc::Src::kYYYY),
sample_depth_stencil_src, temp_z_src, true);
break;
case 1:
// - 2x MSAA: Bottom sample -> bottom-right (3) with Direct3D 11's
// ForcedSampleCount 4.
// - 4x MSAA: Bottom-left Xenia sample -> Direct3D 11 sample 2.
// Check if 4x MSAA is used.
a_.OpIf(true, LoadSystemConstant(
SystemConstants::Index::kSampleCountLog2,
offsetof(SystemConstants, sample_count_log2),
dxbc::Src::kXXXX));
// 4x MSAA.
// temp.x if early = ddx(z)
// temp.y if early = ddy(z)
// temp.z = biased depth in the center
// temp.w if late = saturated sample 1 depth at 4x MSAA
a_.OpMAd(
sample_depth_stencil_dest, z_ddx_src,
dxbc::Src::LF(draw_util::kD3D10StandardSamplePositions4x[2][0] *
(1.0f / 16.0f)),
temp_z_src);
a_.OpMAd(
sample_depth_stencil_dest, z_ddy_src,
dxbc::Src::LF(draw_util::kD3D10StandardSamplePositions4x[2][1] *
(1.0f / 16.0f)),
sample_depth_stencil_src, true);
a_.OpElse();
// 2x MSAA as ForcedSampleCount 4 on the host.
// temp.x if early = ddx(z)
// temp.y if early = ddy(z)
// temp.z = biased depth in the center
// temp.w if late = saturated sample 1 depth at 2x MSAA
a_.OpMAd(
sample_depth_stencil_dest, z_ddx_src,
dxbc::Src::LF(draw_util::kD3D10StandardSamplePositions4x[3][0] *
(1.0f / 16.0f)),
temp_z_src);
a_.OpMAd(
sample_depth_stencil_dest, z_ddy_src,
dxbc::Src::LF(draw_util::kD3D10StandardSamplePositions4x[3][1] *
(1.0f / 16.0f)),
sample_depth_stencil_src, true);
a_.OpEndIf();
break;
default: {
// Xenia samples 2 and 3 (top-right and bottom-right) -> Direct3D 11
// samples 1 and 3.
// temp.x if early = ddx(z)
// temp.y if early = ddy(z)
// temp.z = biased depth in the center
// temp.w if late = saturated sample 2 or 3 depth
const int8_t* sample_position =
draw_util::kD3D10StandardSamplePositions4x[i ^
(((i & 1) ^ (i >> 1)) *
0b11)];
a_.OpMAd(sample_depth_stencil_dest, z_ddx_src,
dxbc::Src::LF(sample_position[0] * (1.0f / 16.0f)),
temp_z_src);
a_.OpMAd(sample_depth_stencil_dest, z_ddy_src,
dxbc::Src::LF(sample_position[1] * (1.0f / 16.0f)),
sample_depth_stencil_src, true);
} break;
}
// Convert the sample's depth to 24-bit, using sample_temp.x as a
// temporary.
// temp.x if early = ddx(z)
// temp.y if early = ddy(z)
// temp.z = biased depth in the center
// temp.w if late = sample's 24-bit Z
ROV_DepthTo24Bit(sample_depth_stencil_src.index_1d_.index_,
sample_depth_stencil_src.swizzle_ & 3,
sample_depth_stencil_src.index_1d_.index_,
sample_depth_stencil_src.swizzle_ & 3, sample_temp, 0);
}
// Load the old depth/stencil value.
// sample_temp.x = old depth/stencil
if (uav_index_edram_ == kBindingIndexUnallocated) {
uav_index_edram_ = uav_count_++;
}
a_.OpLdUAVTyped(
sample_temp_x_dest,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY), 1,
dxbc::Src::U(uav_index_edram_, uint32_t(UAVRegister::kEdram),
dxbc::Src::kXXXX));
// Depth test.
// Extract the old depth part to sample_depth_stencil.
// sample_temp.x = old depth/stencil
// sample_temp.y = old depth
a_.OpUShR(sample_temp_y_dest, sample_temp_x_src, dxbc::Src::LU(8));
// Get the difference between the new and the old depth, > 0 - greater,
// == 0 - equal, < 0 - less.
// sample_temp.x = old depth/stencil
// sample_temp.y = old depth
// sample_temp.z = depth difference
a_.OpIAdd(sample_temp_z_dest, sample_depth_stencil_src, -sample_temp_y_src);
// Check if the depth is "less" or "greater or equal".
// sample_temp.x = old depth/stencil
// sample_temp.y = old depth
// sample_temp.z = depth difference
// sample_temp.w = depth difference less than 0
a_.OpILT(sample_temp_w_dest, sample_temp_z_src, dxbc::Src::LI(0));
// Choose the passed depth function bits for "less" or for "greater".
// sample_temp.x = old depth/stencil
// sample_temp.y = old depth
// sample_temp.z = depth difference
// sample_temp.w = depth function passed bits for "less" or "greater"
a_.OpMovC(sample_temp_w_dest, sample_temp_w_src,
dxbc::Src::LU(kSysFlag_ROVDepthPassIfLess),
dxbc::Src::LU(kSysFlag_ROVDepthPassIfGreater));
// Do the "equal" testing.
// sample_temp.x = old depth/stencil
// sample_temp.y = old depth
// sample_temp.z = depth function passed bits
// sample_temp.w = free
a_.OpMovC(sample_temp_z_dest, sample_temp_z_src, sample_temp_w_src,
dxbc::Src::LU(kSysFlag_ROVDepthPassIfEqual));
// Mask the resulting bits with the ones that should pass.
// sample_temp.x = old depth/stencil
// sample_temp.y = old depth
// sample_temp.z = masked depth function passed bits
a_.OpAnd(sample_temp_z_dest, sample_temp_z_src, LoadFlagsSystemConstant());
// Check if depth test has passed.
// sample_temp.x = old depth/stencil
// sample_temp.y = old depth
// sample_temp.z = free
a_.OpIf(true, sample_temp_z_src);
{
// Extract the depth write flag.
// sample_temp.x = old depth/stencil
// sample_temp.y = old depth
// sample_temp.z = depth write mask
a_.OpAnd(sample_temp_z_dest, LoadFlagsSystemConstant(),
dxbc::Src::LU(kSysFlag_ROVDepthWrite));
// If depth writing is disabled, don't change the depth.
// temp.x if no oDepth and early = ddx(z)
// temp.y if no oDepth and early = ddy(z)
// temp.z if no oDepth = biased depth in the center
// temp.w if late = resulting sample depth after the depth test
// sample_temp.x = old depth/stencil
// sample_temp.y = free
// sample_temp.z = free
a_.OpMovC(sample_depth_stencil_dest, sample_temp_z_src,
sample_depth_stencil_src, sample_temp_y_src);
}
// Depth test has failed.
a_.OpElse();
{
// Exclude the bit from the covered sample mask.
// sample_temp.x = old depth/stencil
// sample_temp.y = old depth
a_.OpAnd(dxbc::Dest::R(system_temp_rov_params_, 0b0001),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(~uint32_t(1 << i)));
}
a_.OpEndIf();
// Create packed depth/stencil, with the stencil value unchanged at this
// point.
// temp.x if no oDepth and early = ddx(z)
// temp.y if no oDepth and early = ddy(z)
// temp.z if no oDepth = biased depth in the center
// temp.w if late = resulting sample depth, current resulting stencil
// sample_temp.x = old depth/stencil
a_.OpBFI(sample_depth_stencil_dest, dxbc::Src::LU(24), dxbc::Src::LU(8),
sample_depth_stencil_src, sample_temp_x_src);
// Stencil test.
// Extract the stencil test bit.
// sample_temp.x = old depth/stencil
// sample_temp.y = stencil test enabled
a_.OpAnd(sample_temp_y_dest, LoadFlagsSystemConstant(),
dxbc::Src::LU(kSysFlag_ROVStencilTest));
// Check if stencil test is enabled.
// sample_temp.x = old depth/stencil
// sample_temp.y = free
a_.OpIf(true, sample_temp_y_src);
{
// Check the current face to get the reference and apply the read mask.
in_front_face_used_ = true;
a_.OpIf(true, dxbc::Src::V1D(
uint32_t(InOutRegister::kPSInFrontFaceAndSampleIndex),
dxbc::Src::kXXXX));
for (uint32_t j = 0; j < 2; ++j) {
if (j) {
// Go to the back face.
a_.OpElse();
}
dxbc::Src stencil_read_mask_src(LoadSystemConstant(
SystemConstants::Index::kEdramStencil,
j ? offsetof(SystemConstants, edram_stencil_back_read_mask)
: offsetof(SystemConstants, edram_stencil_front_read_mask),
dxbc::Src::kXXXX));
// Read-mask the stencil reference.
// sample_temp.x = old depth/stencil
// sample_temp.y = read-masked stencil reference
a_.OpAnd(
sample_temp_y_dest,
LoadSystemConstant(
SystemConstants::Index::kEdramStencil,
j ? offsetof(SystemConstants, edram_stencil_back_reference)
: offsetof(SystemConstants, edram_stencil_front_reference),
dxbc::Src::kXXXX),
stencil_read_mask_src);
// Read-mask the old stencil value (also dropping the depth bits).
// sample_temp.x = old depth/stencil
// sample_temp.y = read-masked stencil reference
// sample_temp.z = read-masked old stencil
a_.OpAnd(sample_temp_z_dest, sample_temp_x_src, stencil_read_mask_src);
}
// Close the face check.
a_.OpEndIf();
// Get the difference between the stencil reference and the old stencil,
// > 0 - greater, == 0 - equal, < 0 - less.
// sample_temp.x = old depth/stencil
// sample_temp.y = stencil difference
// sample_temp.z = free
a_.OpIAdd(sample_temp_y_dest, sample_temp_y_src, -sample_temp_z_src);
// Check if the stencil is "less" or "greater or equal".
// sample_temp.x = old depth/stencil
// sample_temp.y = stencil difference
// sample_temp.z = stencil difference less than 0
a_.OpILT(sample_temp_z_dest, sample_temp_y_src, dxbc::Src::LI(0));
// Choose the passed depth function bits for "less" or for "greater".
// sample_temp.x = old depth/stencil
// sample_temp.y = stencil difference
// sample_temp.z = stencil function passed bits for "less" or "greater"
a_.OpMovC(sample_temp_z_dest, sample_temp_z_src,
dxbc::Src::LU(uint32_t(xenos::CompareFunction::kLess)),
dxbc::Src::LU(uint32_t(xenos::CompareFunction::kGreater)));
// Do the "equal" testing.
// sample_temp.x = old depth/stencil
// sample_temp.y = stencil function passed bits
// sample_temp.z = free
a_.OpMovC(sample_temp_y_dest, sample_temp_y_src, sample_temp_z_src,
dxbc::Src::LU(uint32_t(xenos::CompareFunction::kEqual)));
// Get the comparison function and the operations for the current face.
// sample_temp.x = old depth/stencil
// sample_temp.y = stencil function passed bits
// sample_temp.z = stencil function and operations
in_front_face_used_ = true;
a_.OpMovC(
sample_temp_z_dest,
dxbc::Src::V1D(uint32_t(InOutRegister::kPSInFrontFaceAndSampleIndex),
dxbc::Src::kXXXX),
LoadSystemConstant(
SystemConstants::Index::kEdramStencil,
offsetof(SystemConstants, edram_stencil_front_func_ops),
dxbc::Src::kXXXX),
LoadSystemConstant(
SystemConstants::Index::kEdramStencil,
offsetof(SystemConstants, edram_stencil_back_func_ops),
dxbc::Src::kXXXX));
// Mask the resulting bits with the ones that should pass (the comparison
// function is in the low 3 bits of the constant, and only ANDing 3-bit
// values with it, so safe not to UBFE the function).
// sample_temp.x = old depth/stencil
// sample_temp.y = stencil test result
// sample_temp.z = stencil function and operations
a_.OpAnd(sample_temp_y_dest, sample_temp_y_src, sample_temp_z_src);
// Handle passing and failure of the stencil test, to choose the operation
// and to discard the sample.
// sample_temp.x = old depth/stencil
// sample_temp.y = free
// sample_temp.z = stencil function and operations
a_.OpIf(true, sample_temp_y_src);
{
// Check if depth test has passed for this sample (the sample will only
// be processed if it's covered, so the only thing that could unset the
// bit at this point that matters is the depth test).
// sample_temp.x = old depth/stencil
// sample_temp.y = depth test result
// sample_temp.z = stencil function and operations
a_.OpAnd(sample_temp_y_dest,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(1 << i));
// Choose the bit offset of the stencil operation.
// sample_temp.x = old depth/stencil
// sample_temp.y = sample operation offset
// sample_temp.z = stencil function and operations
a_.OpMovC(sample_temp_y_dest, sample_temp_y_src, dxbc::Src::LU(6),
dxbc::Src::LU(9));
// Extract the stencil operation.
// sample_temp.x = old depth/stencil
// sample_temp.y = stencil operation
// sample_temp.z = free
a_.OpUBFE(sample_temp_y_dest, dxbc::Src::LU(3), sample_temp_y_src,
sample_temp_z_src);
}
// Stencil test has failed.
a_.OpElse();
{
// Extract the stencil fail operation.
// sample_temp.x = old depth/stencil
// sample_temp.y = stencil operation
// sample_temp.z = free
a_.OpUBFE(sample_temp_y_dest, dxbc::Src::LU(3), dxbc::Src::LU(3),
sample_temp_z_src);
// Exclude the bit from the covered sample mask.
// sample_temp.x = old depth/stencil
// sample_temp.y = stencil operation
a_.OpAnd(dxbc::Dest::R(system_temp_rov_params_, 0b0001),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(~uint32_t(1 << i)));
}
// Close the stencil pass check.
a_.OpEndIf();
// Open the stencil operation switch for writing the new stencil (not
// caring about bits 8:31).
// sample_temp.x = old depth/stencil
// sample_temp.y = will contain unmasked new stencil in 0:7 and junk above
a_.OpSwitch(sample_temp_y_src);
{
// Zero.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::StencilOp::kZero)));
a_.OpMov(sample_temp_y_dest, dxbc::Src::LU(0));
a_.OpBreak();
// Replace.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::StencilOp::kReplace)));
in_front_face_used_ = true;
a_.OpMovC(sample_temp_y_dest,
dxbc::Src::V1D(
uint32_t(InOutRegister::kPSInFrontFaceAndSampleIndex),
dxbc::Src::kXXXX),
LoadSystemConstant(
SystemConstants::Index::kEdramStencil,
offsetof(SystemConstants, edram_stencil_front_reference),
dxbc::Src::kXXXX),
LoadSystemConstant(
SystemConstants::Index::kEdramStencil,
offsetof(SystemConstants, edram_stencil_back_reference),
dxbc::Src::kXXXX));
a_.OpBreak();
// Increment and clamp.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::StencilOp::kIncrementClamp)));
{
// Clear the upper bits for saturation.
a_.OpAnd(sample_temp_y_dest, sample_temp_x_src,
dxbc::Src::LU(UINT8_MAX));
// Increment.
a_.OpIAdd(sample_temp_y_dest, sample_temp_y_src, dxbc::Src::LI(1));
// Clamp.
a_.OpIMin(sample_temp_y_dest, sample_temp_y_src,
dxbc::Src::LI(UINT8_MAX));
}
a_.OpBreak();
// Decrement and clamp.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::StencilOp::kDecrementClamp)));
{
// Clear the upper bits for saturation.
a_.OpAnd(sample_temp_y_dest, sample_temp_x_src,
dxbc::Src::LU(UINT8_MAX));
// Increment.
a_.OpIAdd(sample_temp_y_dest, sample_temp_y_src, dxbc::Src::LI(-1));
// Clamp.
a_.OpIMax(sample_temp_y_dest, sample_temp_y_src, dxbc::Src::LI(0));
}
a_.OpBreak();
// Invert.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::StencilOp::kInvert)));
a_.OpNot(sample_temp_y_dest, sample_temp_x_src);
a_.OpBreak();
// Increment and wrap.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::StencilOp::kIncrementWrap)));
a_.OpIAdd(sample_temp_y_dest, sample_temp_x_src, dxbc::Src::LI(1));
a_.OpBreak();
// Decrement and wrap.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::StencilOp::kDecrementWrap)));
a_.OpIAdd(sample_temp_y_dest, sample_temp_x_src, dxbc::Src::LI(-1));
a_.OpBreak();
// Keep.
a_.OpDefault();
a_.OpMov(sample_temp_y_dest, sample_temp_x_src);
a_.OpBreak();
}
// Close the new stencil switch.
a_.OpEndSwitch();
// Select the stencil write mask for the face.
// sample_temp.x = old depth/stencil
// sample_temp.y = unmasked new stencil in 0:7 and junk above
// sample_temp.z = stencil write mask
in_front_face_used_ = true;
a_.OpMovC(
sample_temp_z_dest,
dxbc::Src::V1D(uint32_t(InOutRegister::kPSInFrontFaceAndSampleIndex),
dxbc::Src::kXXXX),
LoadSystemConstant(
SystemConstants::Index::kEdramStencil,
offsetof(SystemConstants, edram_stencil_front_write_mask),
dxbc::Src::kXXXX),
LoadSystemConstant(
SystemConstants::Index::kEdramStencil,
offsetof(SystemConstants, edram_stencil_back_write_mask),
dxbc::Src::kXXXX));
// Apply the write mask to the new stencil, also dropping the upper 24
// bits.
// sample_temp.x = old depth/stencil
// sample_temp.y = masked new stencil
// sample_temp.z = stencil write mask
a_.OpAnd(sample_temp_y_dest, sample_temp_y_src, sample_temp_z_src);
// Invert the write mask for keeping the old stencil and the depth bits.
// sample_temp.x = old depth/stencil
// sample_temp.y = masked new stencil
// sample_temp.z = inverted stencil write mask
a_.OpNot(sample_temp_z_dest, sample_temp_z_src);
// Remove the bits that will be replaced from the combined depth/stencil
// before inserting their new values.
// sample_temp.x = old depth/stencil
// sample_temp.y = masked new stencil
// sample_temp.z = free
// temp.x if no oDepth and early = ddx(z)
// temp.y if no oDepth and early = ddy(z)
// temp.z if no oDepth = biased depth in the center
// temp.w if late = resulting sample depth, inverse-write-masked old
// stencil
a_.OpAnd(sample_depth_stencil_dest, sample_depth_stencil_src,
sample_temp_z_src);
// Merge the old and the new stencil.
// temp.x if no oDepth and early = ddx(z)
// temp.y if no oDepth and early = ddy(z)
// temp.z if no oDepth = biased depth in the center
// temp.w if late = resulting sample depth/stencil
// sample_temp.x = old depth/stencil
// sample_temp.y = free
a_.OpOr(sample_depth_stencil_dest, sample_depth_stencil_src,
sample_temp_y_src);
}
// Close the stencil test check.
a_.OpEndIf();
// Check if the depth/stencil has failed not to modify the depth if it has.
// sample_temp.x = old depth/stencil
// sample_temp.y = whether depth/stencil has passed for this sample
a_.OpAnd(sample_temp_y_dest,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(1 << i));
// If the depth/stencil test has failed, don't change the depth.
// sample_temp.x = old depth/stencil
// sample_temp.y = free
a_.OpIf(false, sample_temp_y_src);
{
// Copy the new stencil over the old depth.
// temp.x if no oDepth and early = ddx(z)
// temp.y if no oDepth and early = ddy(z)
// temp.z if no oDepth = biased depth in the center
// temp.w if late = resulting sample depth/stencil
a_.OpBFI(sample_depth_stencil_dest, dxbc::Src::LU(8), dxbc::Src::LU(0),
sample_depth_stencil_src, sample_temp_x_src);
}
// Close the depth/stencil passing check.
a_.OpEndIf();
// Check if the new depth/stencil is different, and thus needs to be
// written.
// sample_temp.x = old depth/stencil
a_.OpINE(sample_temp_x_dest, sample_depth_stencil_src, sample_temp_x_src);
if (depth_stencil_early &&
!current_shader().implicit_early_z_write_allowed()) {
// Set the sample bit in bits 4:7 of system_temp_rov_params_.x - always
// need to write late in this shader, as it may do something like
// explicitly killing pixels.
a_.OpBFI(dxbc::Dest::R(system_temp_rov_params_, 0b0001), dxbc::Src::LU(1),
dxbc::Src::LU(4 + i), sample_temp_x_src,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX));
} else {
// Check if need to write.
// sample_temp.x = free
a_.OpIf(true, sample_temp_x_src);
{
if (depth_stencil_early) {
// Get if early depth/stencil write is enabled.
// sample_temp.x = whether early depth/stencil write is enabled
a_.OpAnd(sample_temp_x_dest, LoadFlagsSystemConstant(),
dxbc::Src::LU(kSysFlag_ROVDepthStencilEarlyWrite));
// Check if need to write early.
// sample_temp.x = free
a_.OpIf(true, sample_temp_x_src);
}
// Write the new depth/stencil.
if (uav_index_edram_ == kBindingIndexUnallocated) {
uav_index_edram_ = uav_count_++;
}
a_.OpStoreUAVTyped(
dxbc::Dest::U(uav_index_edram_, uint32_t(UAVRegister::kEdram)),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY), 1,
sample_depth_stencil_src);
if (depth_stencil_early) {
// Need to still run the shader to know whether to write the
// depth/stencil value.
a_.OpElse();
// Set the sample bit in bits 4:7 of system_temp_rov_params_.x if need
// to write later (after checking if the sample is not discarded by a
// kill instruction, alphatest or alpha-to-coverage).
a_.OpOr(dxbc::Dest::R(system_temp_rov_params_, 0b0001),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(1 << (4 + i)));
// Close the early depth/stencil check.
a_.OpEndIf();
}
}
// Close the write check.
a_.OpEndIf();
}
// Release sample_temp.
PopSystemTemp();
// Close the sample conditional.
a_.OpEndIf();
// Go to the next sample (samples are at +0, +(80*scale_x), +1,
// +(80*scale_x+1), so need to do +(80*scale_x), -(80*scale_x-1),
// +(80*scale_x) and -(80*scale_x+1) after each sample).
a_.OpIAdd(dxbc::Dest::R(system_temp_rov_params_, 0b0010),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY),
dxbc::Src::LI((i & 1) ? -80 * draw_resolution_scale_x_ + 2 - i
: 80 * draw_resolution_scale_x_));
}
if (ROV_IsDepthStencilEarly()) {
// Check if safe to discard the whole 2x2 quad early, without running the
// translated pixel shader, by checking if coverage is 0 in all pixels in
// the quad and if there are no samples which failed the depth test, but
// where stencil was modified and needs to be written in the end. Must
// reject at 2x2 quad granularity because texture fetches need derivatives.
// temp.x = coverage | deferred depth/stencil write
a_.OpAnd(dxbc::Dest::R(temp, 0b0001),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(0b11111111));
// temp.x = 1.0 if any sample is covered or potentially needs stencil write
// in the end of the shader in the current pixel
a_.OpMovC(dxbc::Dest::R(temp, 0b0001), dxbc::Src::R(temp, dxbc::Src::kXXXX),
dxbc::Src::LF(1.0f), dxbc::Src::LF(0.0f));
// temp.x = 1.0 if any sample is covered or potentially needs stencil write
// in the end of the shader in the current pixel
// temp.y = non-zero if anything is covered in the pixel across X
a_.OpDerivRTXFine(dxbc::Dest::R(temp, 0b0010),
dxbc::Src::R(temp, dxbc::Src::kXXXX));
// temp.x = 1.0 if anything is covered in the current half of the quad
// temp.y = free
a_.OpMovC(dxbc::Dest::R(temp, 0b0001), dxbc::Src::R(temp, dxbc::Src::kYYYY),
dxbc::Src::LF(1.0f), dxbc::Src::R(temp, dxbc::Src::kXXXX));
// temp.x = 1.0 if anything is covered in the current half of the quad
// temp.y = non-zero if anything is covered in the two pixels across Y
a_.OpDerivRTYCoarse(dxbc::Dest::R(temp, 0b0010),
dxbc::Src::R(temp, dxbc::Src::kXXXX));
// temp.x = 1.0 if anything is covered in the current whole quad
// temp.y = free
a_.OpMovC(dxbc::Dest::R(temp, 0b0001), dxbc::Src::R(temp, dxbc::Src::kYYYY),
dxbc::Src::LF(1.0f), dxbc::Src::R(temp, dxbc::Src::kXXXX));
// End the shader if nothing is covered in the 2x2 quad after early
// depth/stencil.
// temp.x = free
a_.OpRetC(false, dxbc::Src::R(temp, dxbc::Src::kXXXX));
}
// Close the large depth/stencil conditional.
a_.OpEndIf();
// Release temp.
PopSystemTemp();
}
void DxbcShaderTranslator::ROV_UnpackColor(
uint32_t rt_index, uint32_t packed_temp, uint32_t packed_temp_components,
uint32_t color_temp, uint32_t temp1, uint32_t temp1_component,
uint32_t temp2, uint32_t temp2_component) {
assert_true(color_temp != packed_temp || packed_temp_components == 0);
dxbc::Src packed_temp_low(
dxbc::Src::R(packed_temp).Select(packed_temp_components));
dxbc::Dest temp1_dest(dxbc::Dest::R(temp1, 1 << temp1_component));
dxbc::Src temp1_src(dxbc::Src::R(temp1).Select(temp1_component));
dxbc::Dest temp2_dest(dxbc::Dest::R(temp2, 1 << temp2_component));
dxbc::Src temp2_src(dxbc::Src::R(temp2).Select(temp2_component));
// Break register dependencies and initialize if there are not enough
// components. The rest of the function will write at least RG (k_32_FLOAT and
// k_32_32_FLOAT handled with the same default label), and if packed_temp is
// the same as color_temp, the packed color won't be touched.
a_.OpMov(dxbc::Dest::R(color_temp, 0b1100),
dxbc::Src::LF(0.0f, 0.0f, 0.0f, 1.0f));
// Choose the packing based on the render target's format.
a_.OpSwitch(
LoadSystemConstant(SystemConstants::Index::kEdramRTFormatFlags,
offsetof(SystemConstants, edram_rt_format_flags) +
sizeof(uint32_t) * rt_index,
dxbc::Src::kXXXX));
// ***************************************************************************
// k_8_8_8_8
// k_8_8_8_8_GAMMA
// ***************************************************************************
for (uint32_t i = 0; i < 2; ++i) {
a_.OpCase(dxbc::Src::LU(ROV_AddColorFormatFlags(
i ? xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA
: xenos::ColorRenderTargetFormat::k_8_8_8_8)));
// Unpack the components.
a_.OpUBFE(dxbc::Dest::R(color_temp), dxbc::Src::LU(8),
dxbc::Src::LU(0, 8, 16, 24), packed_temp_low);
// Convert from fixed-point.
a_.OpUToF(dxbc::Dest::R(color_temp), dxbc::Src::R(color_temp));
// Normalize.
a_.OpMul(dxbc::Dest::R(color_temp), dxbc::Src::R(color_temp),
dxbc::Src::LF(1.0f / 255.0f));
if (i) {
for (uint32_t j = 0; j < 3; ++j) {
PWLGammaToLinear(color_temp, j, color_temp, j, true, temp1,
temp1_component, temp2, temp2_component);
}
}
a_.OpBreak();
}
// ***************************************************************************
// k_2_10_10_10
// k_2_10_10_10_AS_10_10_10_10
// ***************************************************************************
a_.OpCase(dxbc::Src::LU(
ROV_AddColorFormatFlags(xenos::ColorRenderTargetFormat::k_2_10_10_10)));
a_.OpCase(dxbc::Src::LU(ROV_AddColorFormatFlags(
xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10)));
{
// Unpack the components.
a_.OpUBFE(dxbc::Dest::R(color_temp), dxbc::Src::LU(10, 10, 10, 2),
dxbc::Src::LU(0, 10, 20, 30), packed_temp_low);
// Convert from fixed-point.
a_.OpUToF(dxbc::Dest::R(color_temp), dxbc::Src::R(color_temp));
// Normalize.
a_.OpMul(dxbc::Dest::R(color_temp), dxbc::Src::R(color_temp),
dxbc::Src::LF(1.0f / 1023.0f, 1.0f / 1023.0f, 1.0f / 1023.0f,
1.0f / 3.0f));
}
a_.OpBreak();
// ***************************************************************************
// k_2_10_10_10_FLOAT
// k_2_10_10_10_FLOAT_AS_16_16_16_16
// https://github.com/Microsoft/DirectXTex/blob/master/DirectXTex/DirectXTexConvert.cpp
// ***************************************************************************
a_.OpCase(dxbc::Src::LU(ROV_AddColorFormatFlags(
xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT)));
a_.OpCase(dxbc::Src::LU(ROV_AddColorFormatFlags(
xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT_AS_16_16_16_16)));
{
// Unpack the alpha.
a_.OpUBFE(dxbc::Dest::R(color_temp, 0b1000), dxbc::Src::LU(2),
dxbc::Src::LU(30), packed_temp_low);
// Convert the alpha from fixed-point.
a_.OpUToF(dxbc::Dest::R(color_temp, 0b1000),
dxbc::Src::R(color_temp, dxbc::Src::kWWWW));
// Normalize the alpha.
a_.OpMul(dxbc::Dest::R(color_temp, 0b1000),
dxbc::Src::R(color_temp, dxbc::Src::kWWWW),
dxbc::Src::LF(1.0f / 3.0f));
// Process the components in reverse order because color_temp.r stores the
// packed color which shouldn't be touched until G and B are converted if
// packed_temp and color_temp are the same.
for (int32_t i = 2; i >= 0; --i) {
Float7e3To32(a_, dxbc::Dest::R(color_temp, 1 << i), packed_temp,
packed_temp_components, i * 10, color_temp, i, temp1,
temp1_component);
}
}
a_.OpBreak();
// ***************************************************************************
// k_16_16
// k_16_16_16_16 (64bpp)
// ***************************************************************************
for (uint32_t i = 0; i < 2; ++i) {
a_.OpCase(dxbc::Src::LU(ROV_AddColorFormatFlags(
i ? xenos::ColorRenderTargetFormat::k_16_16_16_16
: xenos::ColorRenderTargetFormat::k_16_16)));
dxbc::Dest color_components_dest(
dxbc::Dest::R(color_temp, i ? 0b1111 : 0b0011));
// Unpack the components.
a_.OpIBFE(color_components_dest, dxbc::Src::LU(16),
dxbc::Src::LU(0, 16, 0, 16),
dxbc::Src::R(packed_temp,
0b01010000 + packed_temp_components * 0b01010101));
// Convert from fixed-point.
a_.OpIToF(color_components_dest, dxbc::Src::R(color_temp));
// Normalize.
a_.OpMul(color_components_dest, dxbc::Src::R(color_temp),
dxbc::Src::LF(32.0f / 32767.0f));
a_.OpBreak();
}
// ***************************************************************************
// k_16_16_FLOAT
// k_16_16_16_16_FLOAT (64bpp)
// ***************************************************************************
for (uint32_t i = 0; i < 2; ++i) {
a_.OpCase(dxbc::Src::LU(ROV_AddColorFormatFlags(
i ? xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT
: xenos::ColorRenderTargetFormat::k_16_16_FLOAT)));
dxbc::Dest color_components_dest(
dxbc::Dest::R(color_temp, i ? 0b1111 : 0b0011));
// Unpack the components.
a_.OpUBFE(color_components_dest, dxbc::Src::LU(16),
dxbc::Src::LU(0, 16, 0, 16),
dxbc::Src::R(packed_temp,
0b01010000 + packed_temp_components * 0b01010101));
// Convert from 16-bit float.
a_.OpF16ToF32(color_components_dest, dxbc::Src::R(color_temp));
a_.OpBreak();
}
if (packed_temp != color_temp) {
// Assume k_32_FLOAT or k_32_32_FLOAT for the rest.
a_.OpDefault();
a_.OpMov(
dxbc::Dest::R(color_temp, 0b0011),
dxbc::Src::R(packed_temp, 0b0100 + packed_temp_components * 0b0101));
a_.OpBreak();
}
a_.OpEndSwitch();
}
void DxbcShaderTranslator::ROV_PackPreClampedColor(
uint32_t rt_index, uint32_t color_temp, uint32_t packed_temp,
uint32_t packed_temp_components, uint32_t temp1, uint32_t temp1_component,
uint32_t temp2, uint32_t temp2_component) {
// Packing normalized formats according to the Direct3D 11.3 functional
// specification, but assuming clamping was done by the caller.
assert_true(color_temp != packed_temp || packed_temp_components == 0);
dxbc::Dest packed_dest_low(
dxbc::Dest::R(packed_temp, 1 << packed_temp_components));
dxbc::Src packed_src_low(
dxbc::Src::R(packed_temp).Select(packed_temp_components));
dxbc::Dest temp1_dest(dxbc::Dest::R(temp1, 1 << temp1_component));
dxbc::Src temp1_src(dxbc::Src::R(temp1).Select(temp1_component));
dxbc::Dest temp2_dest(dxbc::Dest::R(temp2, 1 << temp2_component));
dxbc::Src temp2_src(dxbc::Src::R(temp2).Select(temp2_component));
// Break register dependency after 32bpp cases.
a_.OpMov(dxbc::Dest::R(packed_temp, 1 << (packed_temp_components + 1)),
dxbc::Src::LU(0));
// Choose the packing based on the render target's format.
a_.OpSwitch(
LoadSystemConstant(SystemConstants::Index::kEdramRTFormatFlags,
offsetof(SystemConstants, edram_rt_format_flags) +
sizeof(uint32_t) * rt_index,
dxbc::Src::kXXXX));
// ***************************************************************************
// k_8_8_8_8
// k_8_8_8_8_GAMMA
// ***************************************************************************
for (uint32_t i = 0; i < 2; ++i) {
a_.OpCase(dxbc::Src::LU(ROV_AddColorFormatFlags(
i ? xenos::ColorRenderTargetFormat::k_8_8_8_8_GAMMA
: xenos::ColorRenderTargetFormat::k_8_8_8_8)));
for (uint32_t j = 0; j < 4; ++j) {
if (i && j < 3) {
PreSaturatedLinearToPWLGamma(temp1, temp1_component, color_temp, j,
temp1, temp1_component, temp2,
temp2_component);
// Denormalize and add 0.5 for rounding.
a_.OpMAd(temp1_dest, temp1_src, dxbc::Src::LF(255.0f),
dxbc::Src::LF(0.5f));
} else {
// Denormalize and add 0.5 for rounding.
a_.OpMAd(temp1_dest, dxbc::Src::R(color_temp).Select(j),
dxbc::Src::LF(255.0f), dxbc::Src::LF(0.5f));
}
// Convert to fixed-point.
a_.OpFToU(j ? temp1_dest : packed_dest_low, temp1_src);
// Pack the upper components.
if (j) {
a_.OpBFI(packed_dest_low, dxbc::Src::LU(8), dxbc::Src::LU(j * 8),
temp1_src, packed_src_low);
}
}
a_.OpBreak();
}
// ***************************************************************************
// k_2_10_10_10
// k_2_10_10_10_AS_10_10_10_10
// ***************************************************************************
a_.OpCase(dxbc::Src::LU(
ROV_AddColorFormatFlags(xenos::ColorRenderTargetFormat::k_2_10_10_10)));
a_.OpCase(dxbc::Src::LU(ROV_AddColorFormatFlags(
xenos::ColorRenderTargetFormat::k_2_10_10_10_AS_10_10_10_10)));
for (uint32_t i = 0; i < 4; ++i) {
// Denormalize and convert to fixed-point.
a_.OpMAd(temp1_dest, dxbc::Src::R(color_temp).Select(i),
dxbc::Src::LF(i < 3 ? 1023.0f : 3.0f), dxbc::Src::LF(0.5f));
a_.OpFToU(i ? temp1_dest : packed_dest_low, temp1_src);
// Pack the upper components.
if (i) {
a_.OpBFI(packed_dest_low, dxbc::Src::LU(i < 3 ? 10 : 2),
dxbc::Src::LU(i * 10), temp1_src, packed_src_low);
}
}
a_.OpBreak();
// ***************************************************************************
// k_2_10_10_10_FLOAT
// k_2_10_10_10_FLOAT_AS_16_16_16_16
// https://github.com/Microsoft/DirectXTex/blob/master/DirectXTex/DirectXTexConvert.cpp
// ***************************************************************************
a_.OpCase(dxbc::Src::LU(ROV_AddColorFormatFlags(
xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT)));
a_.OpCase(dxbc::Src::LU(ROV_AddColorFormatFlags(
xenos::ColorRenderTargetFormat::k_2_10_10_10_FLOAT_AS_16_16_16_16)));
{
// Convert red directly to the destination, which may be the same as the
// source, but PreClampedFloat32To7e3 allows that.
PreClampedFloat32To7e3(a_, packed_temp, packed_temp_components, color_temp,
0, temp1, temp1_component);
for (uint32_t i = 1; i < 3; ++i) {
// Convert green and blue to a temporary register and insert them into the
// result.
PreClampedFloat32To7e3(a_, temp1, temp1_component, color_temp, i, temp2,
temp2_component);
a_.OpBFI(packed_dest_low, dxbc::Src::LU(10), dxbc::Src::LU(i * 10),
temp1_src, packed_src_low);
}
// Denormalize the alpha and convert it to fixed-point.
a_.OpMAd(temp1_dest, dxbc::Src::R(color_temp, dxbc::Src::kWWWW),
dxbc::Src::LF(3.0f), dxbc::Src::LF(0.5f));
a_.OpFToU(temp1_dest, temp1_src);
// Pack the alpha.
a_.OpBFI(packed_dest_low, dxbc::Src::LU(2), dxbc::Src::LU(30), temp1_src,
packed_src_low);
}
a_.OpBreak();
// ***************************************************************************
// k_16_16
// k_16_16_16_16 (64bpp)
// ***************************************************************************
for (uint32_t i = 0; i < 2; ++i) {
a_.OpCase(dxbc::Src::LU(ROV_AddColorFormatFlags(
i ? xenos::ColorRenderTargetFormat::k_16_16_16_16
: xenos::ColorRenderTargetFormat::k_16_16)));
for (uint32_t j = 0; j < (uint32_t(2) << i); ++j) {
// Denormalize and convert to fixed-point, making 0.5 with the proper sign
// in temp2.
a_.OpGE(temp2_dest, dxbc::Src::R(color_temp).Select(j),
dxbc::Src::LF(0.0f));
a_.OpMovC(temp2_dest, temp2_src, dxbc::Src::LF(0.5f),
dxbc::Src::LF(-0.5f));
a_.OpMAd(temp1_dest, dxbc::Src::R(color_temp).Select(j),
dxbc::Src::LF(32767.0f / 32.0f), temp2_src);
dxbc::Dest packed_dest_half(
dxbc::Dest::R(packed_temp, 1 << (packed_temp_components + (j >> 1))));
// Convert to fixed-point.
a_.OpFToI((j & 1) ? temp1_dest : packed_dest_half, temp1_src);
// Pack green or alpha.
if (j & 1) {
a_.OpBFI(packed_dest_half, dxbc::Src::LU(16), dxbc::Src::LU(16),
temp1_src,
dxbc::Src::R(packed_temp)
.Select(packed_temp_components + (j >> 1)));
}
}
a_.OpBreak();
}
// ***************************************************************************
// k_16_16_FLOAT
// k_16_16_16_16_FLOAT (64bpp)
// ***************************************************************************
for (uint32_t i = 0; i < 2; ++i) {
a_.OpCase(dxbc::Src::LU(ROV_AddColorFormatFlags(
i ? xenos::ColorRenderTargetFormat::k_16_16_16_16_FLOAT
: xenos::ColorRenderTargetFormat::k_16_16_FLOAT)));
for (uint32_t j = 0; j < (uint32_t(2) << i); ++j) {
dxbc::Dest packed_dest_half(
dxbc::Dest::R(packed_temp, 1 << (packed_temp_components + (j >> 1))));
// Convert to 16-bit float.
a_.OpF32ToF16((j & 1) ? temp1_dest : packed_dest_half,
dxbc::Src::R(color_temp).Select(j));
// Pack green or alpha.
if (j & 1) {
a_.OpBFI(packed_dest_half, dxbc::Src::LU(16), dxbc::Src::LU(16),
temp1_src,
dxbc::Src::R(packed_temp)
.Select(packed_temp_components + (j >> 1)));
}
}
a_.OpBreak();
}
if (packed_temp != color_temp) {
// Assume k_32_FLOAT or k_32_32_FLOAT for the rest.
a_.OpDefault();
a_.OpMov(dxbc::Dest::R(packed_temp, 0b11 << packed_temp_components),
dxbc::Src::R(color_temp, 0b0100 << (packed_temp_components * 2)));
a_.OpBreak();
}
a_.OpEndSwitch();
}
void DxbcShaderTranslator::ROV_HandleColorBlendFactorCases(
uint32_t src_temp, uint32_t dst_temp, uint32_t factor_temp) {
dxbc::Dest factor_dest(dxbc::Dest::R(factor_temp, 0b0111));
dxbc::Src one_src(dxbc::Src::LF(1.0f));
// kOne.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOne)));
a_.OpMov(factor_dest, one_src);
a_.OpBreak();
// kSrcColor
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kSrcColor)));
if (factor_temp != src_temp) {
a_.OpMov(factor_dest, dxbc::Src::R(src_temp));
}
a_.OpBreak();
// kOneMinusSrcColor
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOneMinusSrcColor)));
a_.OpAdd(factor_dest, one_src, -dxbc::Src::R(src_temp));
a_.OpBreak();
// kSrcAlpha
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kSrcAlpha)));
a_.OpMov(factor_dest, dxbc::Src::R(src_temp, dxbc::Src::kWWWW));
a_.OpBreak();
// kOneMinusSrcAlpha
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOneMinusSrcAlpha)));
a_.OpAdd(factor_dest, one_src, -dxbc::Src::R(src_temp, dxbc::Src::kWWWW));
a_.OpBreak();
// kDstColor
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kDstColor)));
if (factor_temp != dst_temp) {
a_.OpMov(factor_dest, dxbc::Src::R(dst_temp));
}
a_.OpBreak();
// kOneMinusDstColor
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOneMinusDstColor)));
a_.OpAdd(factor_dest, one_src, -dxbc::Src::R(dst_temp));
a_.OpBreak();
// kDstAlpha
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kDstAlpha)));
a_.OpMov(factor_dest, dxbc::Src::R(dst_temp, dxbc::Src::kWWWW));
a_.OpBreak();
// kOneMinusDstAlpha
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOneMinusDstAlpha)));
a_.OpAdd(factor_dest, one_src, -dxbc::Src::R(dst_temp, dxbc::Src::kWWWW));
a_.OpBreak();
// Factors involving the constant.
// kConstantColor
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kConstantColor)));
a_.OpMov(factor_dest,
LoadSystemConstant(SystemConstants::Index::kEdramBlendConstant,
offsetof(SystemConstants, edram_blend_constant),
dxbc::Src::kXYZW));
a_.OpBreak();
// kOneMinusConstantColor
a_.OpCase(
dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOneMinusConstantColor)));
a_.OpAdd(factor_dest, one_src,
-LoadSystemConstant(SystemConstants::Index::kEdramBlendConstant,
offsetof(SystemConstants, edram_blend_constant),
dxbc::Src::kXYZW));
a_.OpBreak();
// kConstantAlpha
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kConstantAlpha)));
a_.OpMov(factor_dest,
LoadSystemConstant(SystemConstants::Index::kEdramBlendConstant,
offsetof(SystemConstants, edram_blend_constant),
dxbc::Src::kWWWW));
a_.OpBreak();
// kOneMinusConstantAlpha
a_.OpCase(
dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOneMinusConstantAlpha)));
a_.OpAdd(factor_dest, one_src,
-LoadSystemConstant(SystemConstants::Index::kEdramBlendConstant,
offsetof(SystemConstants, edram_blend_constant),
dxbc::Src::kWWWW));
a_.OpBreak();
// kSrcAlphaSaturate
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kSrcAlphaSaturate)));
a_.OpAdd(dxbc::Dest::R(factor_temp, 0b0001), one_src,
-dxbc::Src::R(dst_temp, dxbc::Src::kWWWW));
a_.OpMin(factor_dest, dxbc::Src::R(src_temp, dxbc::Src::kWWWW),
dxbc::Src::R(factor_temp, dxbc::Src::kXXXX));
a_.OpBreak();
// kZero default.
a_.OpDefault();
a_.OpMov(factor_dest, dxbc::Src::LF(0.0f));
a_.OpBreak();
}
void DxbcShaderTranslator::ROV_HandleAlphaBlendFactorCases(
uint32_t src_temp, uint32_t dst_temp, uint32_t factor_temp,
uint32_t factor_component) {
dxbc::Dest factor_dest(dxbc::Dest::R(factor_temp, 1 << factor_component));
dxbc::Src one_src(dxbc::Src::LF(1.0f));
// kOne, kSrcAlphaSaturate.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOne)));
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kSrcAlphaSaturate)));
a_.OpMov(factor_dest, one_src);
a_.OpBreak();
// kSrcColor, kSrcAlpha.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kSrcColor)));
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kSrcAlpha)));
if (factor_temp != src_temp || factor_component != 3) {
a_.OpMov(factor_dest, dxbc::Src::R(src_temp, dxbc::Src::kWWWW));
}
a_.OpBreak();
// kOneMinusSrcColor, kOneMinusSrcAlpha.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOneMinusSrcColor)));
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOneMinusSrcAlpha)));
a_.OpAdd(factor_dest, one_src, -dxbc::Src::R(src_temp, dxbc::Src::kWWWW));
a_.OpBreak();
// kDstColor, kDstAlpha.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kDstColor)));
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kDstAlpha)));
if (factor_temp != dst_temp || factor_component != 3) {
a_.OpMov(factor_dest, dxbc::Src::R(dst_temp, dxbc::Src::kWWWW));
}
a_.OpBreak();
// kOneMinusDstColor, kOneMinusDstAlpha.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOneMinusDstColor)));
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOneMinusDstAlpha)));
a_.OpAdd(factor_dest, one_src, -dxbc::Src::R(dst_temp, dxbc::Src::kWWWW));
a_.OpBreak();
// Factors involving the constant.
// kConstantColor, kConstantAlpha.
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kConstantColor)));
a_.OpCase(dxbc::Src::LU(uint32_t(xenos::BlendFactor::kConstantAlpha)));
a_.OpMov(factor_dest,
LoadSystemConstant(SystemConstants::Index::kEdramBlendConstant,
offsetof(SystemConstants, edram_blend_constant),
dxbc::Src::kWWWW));
a_.OpBreak();
// kOneMinusConstantColor, kOneMinusConstantAlpha.
a_.OpCase(
dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOneMinusConstantColor)));
a_.OpCase(
dxbc::Src::LU(uint32_t(xenos::BlendFactor::kOneMinusConstantAlpha)));
a_.OpAdd(factor_dest, one_src,
-LoadSystemConstant(SystemConstants::Index::kEdramBlendConstant,
offsetof(SystemConstants, edram_blend_constant),
dxbc::Src::kWWWW));
a_.OpBreak();
// kZero default.
a_.OpDefault();
a_.OpMov(factor_dest, dxbc::Src::LF(0.0f));
a_.OpBreak();
}
void DxbcShaderTranslator::CompletePixelShader_WriteToRTVs() {
uint32_t shader_writes_color_targets =
current_shader().writes_color_targets();
if (!shader_writes_color_targets) {
return;
}
uint32_t gamma_temp = PushSystemTemp();
for (uint32_t i = 0; i < 4; ++i) {
if (!(shader_writes_color_targets & (1 << i))) {
continue;
}
uint32_t system_temp_color = system_temps_color_[i];
// Apply the exponent bias after alpha to coverage because it needs the
// unbiased alpha from the shader.
a_.OpMul(dxbc::Dest::R(system_temp_color), dxbc::Src::R(system_temp_color),
LoadSystemConstant(
SystemConstants::Index::kColorExpBias,
offsetof(SystemConstants, color_exp_bias) + sizeof(float) * i,
dxbc::Src::kXXXX));
if (!gamma_render_target_as_srgb_) {
// Convert to gamma space - this is incorrect, since it must be done after
// blending on the Xbox 360, but this is just one of many blending issues
// in the RTV path.
a_.OpAnd(dxbc::Dest::R(gamma_temp, 0b0001), LoadFlagsSystemConstant(),
dxbc::Src::LU(kSysFlag_ConvertColor0ToGamma << i));
a_.OpIf(true, dxbc::Src::R(gamma_temp, dxbc::Src::kXXXX));
// Saturate before the gamma conversion.
a_.OpMov(dxbc::Dest::R(system_temp_color, 0b0111),
dxbc::Src::R(system_temp_color), true);
for (uint32_t j = 0; j < 3; ++j) {
PreSaturatedLinearToPWLGamma(system_temp_color, j, system_temp_color, j,
gamma_temp, 0, gamma_temp, 1);
}
a_.OpEndIf();
}
// Copy the color from a readable temp register to an output register.
a_.OpMov(dxbc::Dest::O(i), dxbc::Src::R(system_temp_color));
}
// Release gamma_temp.
PopSystemTemp();
}
void DxbcShaderTranslator::CompletePixelShader_DSV_DepthTo24Bit() {
bool shader_writes_depth = current_shader().writes_depth();
if (!DSV_IsWritingFloat24Depth()) {
if (shader_writes_depth) {
// If not converting, but the shader writes depth explicitly, for float24,
// need to scale it from guest 0...1 to host 0...0.5 to support
// reinterpretation round trips as viewport scaling doesn't apply to
// oDepth.
a_.OpAnd(dxbc::Dest::R(system_temp_depth_stencil_, 0b0010),
LoadFlagsSystemConstant(), dxbc::Src::LU(kSysFlag_DepthFloat24));
a_.OpIf(true, dxbc::Src::R(system_temp_depth_stencil_, dxbc::Src::kYYYY));
a_.OpMul(dxbc::Dest::R(system_temp_depth_stencil_, 0b0001),
dxbc::Src::R(system_temp_depth_stencil_, dxbc::Src::kXXXX),
dxbc::Src::LF(0.5f));
a_.OpEndIf();
// Write the depth from the temporary to the system depth output.
a_.OpMov(dxbc::Dest::ODepth(),
dxbc::Src::R(system_temp_depth_stencil_, dxbc::Src::kXXXX));
}
return;
}
uint32_t temp;
if (shader_writes_depth) {
// The depth is already written to system_temp_depth_stencil_.x and clamped
// to 0...1 with NaNs dropped (saturating in StoreResult); yzw are free.
temp = system_temp_depth_stencil_;
} else {
// Need a temporary variable; remap the sample's depth input from host
// 0...0.5 back to guest 0...1 for conversion purposes to it and saturate it
// (in Direct3D 11, depth is clamped to the viewport bounds after the pixel
// shader, and SV_Position.z contains the unclamped depth, which may be
// outside the viewport's depth range if it's biased); though it will be
// clamped to the viewport bounds anyway, but to be able to make the
// assumption of it being clamped while working with the bit representation.
temp = PushSystemTemp();
in_position_used_ |= 0b0100;
a_.OpMul(dxbc::Dest::R(temp, 0b0001),
dxbc::Src::V1D(uint32_t(InOutRegister::kPSInPosition),
dxbc::Src::kZZZZ),
dxbc::Src::LF(2.0f), true);
}
dxbc::Dest temp_x_dest(dxbc::Dest::R(temp, 0b0001));
dxbc::Src temp_x_src(dxbc::Src::R(temp, dxbc::Src::kXXXX));
dxbc::Dest temp_y_dest(dxbc::Dest::R(temp, 0b0010));
dxbc::Src temp_y_src(dxbc::Src::R(temp, dxbc::Src::kYYYY));
if (GetDxbcShaderModification().pixel.depth_stencil_mode ==
Modification::DepthStencilMode::kFloat24Truncating) {
// Simplified conversion, always less than or equal to the original value -
// just drop the lower bits.
// The float32 exponent bias is 127.
// After saturating, the exponent range is -127...0.
// The smallest normalized 20e4 exponent is -14 - should drop 3 mantissa
// bits at -14 or above.
// The smallest denormalized 20e4 number is -34 - should drop 23 mantissa
// bits at -34.
// Anything smaller than 2^-34 becomes 0.
dxbc::Dest truncate_dest(shader_writes_depth ? dxbc::Dest::ODepth()
: dxbc::Dest::ODepthLE());
// Check if the number is representable as a float24 after truncation - the
// exponent is at least -34.
a_.OpUGE(temp_y_dest, temp_x_src, dxbc::Src::LU(0x2E800000));
a_.OpIf(true, temp_y_src);
{
// Extract the biased float32 exponent to temp.y.
// temp.y = 113+ at exponent -14+.
// temp.y = 93 at exponent -34.
a_.OpUBFE(temp_y_dest, dxbc::Src::LU(8), dxbc::Src::LU(23), temp_x_src);
// Convert exponent to the unclamped number of bits to truncate.
// 116 - 113 = 3.
// 116 - 93 = 23.
// temp.y = 3+ at exponent -14+.
// temp.y = 23 at exponent -34.
a_.OpIAdd(temp_y_dest, dxbc::Src::LI(116), -temp_y_src);
// Clamp the truncated bit count to drop 3 bits of any normal number.
// Exponents below -34 are handled separately.
// temp.y = 3 at exponent -14.
// temp.y = 23 at exponent -34.
a_.OpIMax(temp_y_dest, temp_y_src, dxbc::Src::LI(3));
// Truncate the mantissa - fill the low bits with zeros.
// temp.x = result in 0...1 range
a_.OpBFI(temp_x_dest, temp_y_src, dxbc::Src::LU(0), dxbc::Src::LU(0),
temp_x_src);
// Remap from guest 0...1 to host 0...0.5.
a_.OpMul(truncate_dest, temp_x_src, dxbc::Src::LF(0.5f));
}
// The number is not representable as float24 after truncation - zero.
a_.OpElse();
a_.OpMov(truncate_dest, dxbc::Src::LF(0.0f));
// Close the non-zero result check.
a_.OpEndIf();
} else {
// Properly convert to 20e4, with rounding to the nearest even (the bias was
// pre-applied by multiplying by 2), then convert back restoring the bias.
PreClampedDepthTo20e4(a_, temp, 0, temp, 0, temp, 1, false);
Depth20e4To32(a_, dxbc::Dest::ODepth(), temp, 0, 0, temp, 0, temp, 1, true);
}
if (!shader_writes_depth) {
// Release temp.
PopSystemTemp();
}
}
void DxbcShaderTranslator::CompletePixelShader_AlphaToMaskSample(
bool initialize, uint32_t sample_index, float threshold_base,
dxbc::Src threshold_offset, float threshold_offset_scale,
uint32_t coverage_temp, uint32_t coverage_temp_component, uint32_t temp,
uint32_t temp_component) {
dxbc::Dest temp_dest(dxbc::Dest::R(temp, 1 << temp_component));
dxbc::Src temp_src(dxbc::Src::R(temp).Select(temp_component));
// Calculate the threshold.
a_.OpMAd(temp_dest, threshold_offset, dxbc::Src::LF(-threshold_offset_scale),
dxbc::Src::LF(threshold_base));
// Check if alpha of oC0 is at or greater than the threshold (handling NaN
// according to the Direct3D 11.3 functional specification, as not covered).
a_.OpGE(temp_dest, dxbc::Src::R(system_temps_color_[0], dxbc::Src::kWWWW),
temp_src);
dxbc::Dest coverage_dest(
dxbc::Dest::R(coverage_temp, 1 << coverage_temp_component));
dxbc::Src coverage_src(
dxbc::Src::R(coverage_temp).Select(coverage_temp_component));
if (edram_rov_used_) {
assert_true(coverage_temp != temp ||
coverage_temp_component != temp_component);
// Keep all bits in but the ones that need to be removed in case of failure.
// For ROV, the test must effect not only the coverage bits, but also the
// deferred depth/stencil write bits since the coverage is zeroed for
// samples that have failed the depth/stencil test, but stencil may still
// require writing - but if the sample is discarded by alpha to coverage, it
// must not be written at all.
a_.OpOr(temp_dest, temp_src,
dxbc::Src::LU(~(uint32_t(0b00010001) << sample_index)));
// Clear the coverage for samples that have failed the test.
a_.OpAnd(coverage_dest, coverage_src, temp_src);
} else {
if (initialize) {
// First sample tested - initialize.
assert_true(coverage_temp != temp ||
coverage_temp_component != temp_component);
a_.OpAnd(coverage_dest, temp_src,
dxbc::Src::LU(uint32_t(1) << sample_index));
} else {
// Not first sample tested - add.
a_.OpAnd(temp_dest, temp_src, dxbc::Src::LU(uint32_t(1) << sample_index));
a_.OpOr(coverage_dest, coverage_src, temp_src);
}
}
}
void DxbcShaderTranslator::CompletePixelShader_AlphaToMask() {
// Check if alpha to coverage can be done at all in this shader.
if (!current_shader().writes_color_target(0)) {
return;
}
if (!edram_rov_used_) {
// Initialize the output coverage for the case if alpha to mask is not
// enabled - it needs to be written on every execution path.
a_.OpMov(dxbc::Dest::OMask(), dxbc::Src::LU(UINT32_MAX));
}
// Check if alpha to coverage is enabled.
dxbc::Src alpha_to_mask_constant_src(LoadSystemConstant(
SystemConstants::Index::kAlphaToMask,
offsetof(SystemConstants, alpha_to_mask), dxbc::Src::kXXXX));
a_.OpIf(true, alpha_to_mask_constant_src);
uint32_t temp = PushSystemTemp();
dxbc::Dest temp_x_dest(dxbc::Dest::R(temp, 0b0001));
dxbc::Src temp_x_src(dxbc::Src::R(temp, dxbc::Src::kXXXX));
// Get the dithering threshold offset index for the pixel, Y - low bit of
// offset index, X - high bit, and extract the offset and convert it to
// floating-point. With resolution scaling, still using host pixels, to
// preserve the idea of dithering.
// temp.x = alpha to coverage offset as float 0.0...3.0.
in_position_used_ |= 0b0011;
a_.OpFToU(dxbc::Dest::R(temp, 0b0011),
dxbc::Src::V1D(uint32_t(InOutRegister::kPSInPosition)));
a_.OpAnd(dxbc::Dest::R(temp, 0b0010), dxbc::Src::R(temp, dxbc::Src::kYYYY),
dxbc::Src::LU(1));
a_.OpBFI(temp_x_dest, dxbc::Src::LU(1), dxbc::Src::LU(1), temp_x_src,
dxbc::Src::R(temp, dxbc::Src::kYYYY));
a_.OpIShL(temp_x_dest, temp_x_src, dxbc::Src::LU(1));
a_.OpUBFE(temp_x_dest, dxbc::Src::LU(2), temp_x_src,
alpha_to_mask_constant_src);
a_.OpUToF(temp_x_dest, temp_x_src);
// Write the result to temp.z for RTV or to system_temp_rov_params_.x for ROV.
// temp.x = alpha to coverage offset as float 0.0...3.0.
// temp.z = without ROV, accumulated coverage.
uint32_t coverage_temp = edram_rov_used_ ? system_temp_rov_params_ : temp;
uint32_t coverage_temp_component = edram_rov_used_ ? 0 : 2;
// Check if MSAA is enabled.
a_.OpIf(true, LoadSystemConstant(SystemConstants::Index::kSampleCountLog2,
offsetof(SystemConstants, sample_count_log2),
dxbc::Src::kYYYY));
{
// Check if MSAA is 4x or 2x.
a_.OpIf(true,
LoadSystemConstant(SystemConstants::Index::kSampleCountLog2,
offsetof(SystemConstants, sample_count_log2),
dxbc::Src::kXXXX));
// 4x MSAA.
// Sample 0 must be checked first - CompletePixelShader_AlphaToMaskSample
// initializes the result for sample index 0.
CompletePixelShader_AlphaToMaskSample(true, 0, 0.75f, temp_x_src,
1.0f / 16.0f, coverage_temp,
coverage_temp_component, temp, 1);
CompletePixelShader_AlphaToMaskSample(false, 1, 0.25f, temp_x_src,
1.0f / 16.0f, coverage_temp,
coverage_temp_component, temp, 1);
CompletePixelShader_AlphaToMaskSample(false, 2, 0.5f, temp_x_src,
1.0f / 16.0f, coverage_temp,
coverage_temp_component, temp, 1);
CompletePixelShader_AlphaToMaskSample(false, 3, 1.0f, temp_x_src,
1.0f / 16.0f, coverage_temp,
coverage_temp_component, temp, 1);
// 2x MSAA.
// With ROV, using guest sample indices.
// Without ROV:
// - Native 2x: top (0 in Xenia) is 1 in D3D10.1+, bottom (1 in Xenia) is 0.
// - 2x as 4x: top is 0, bottom is 3.
a_.OpElse();
CompletePixelShader_AlphaToMaskSample(
true, (!edram_rov_used_ && msaa_2x_supported_) ? 1 : 0, 0.5f,
temp_x_src, 1.0f / 8.0f, coverage_temp, coverage_temp_component, temp,
1);
CompletePixelShader_AlphaToMaskSample(
false, edram_rov_used_ ? 1 : (msaa_2x_supported_ ? 0 : 3), 1.0f,
temp_x_src, 1.0f / 8.0f, coverage_temp, coverage_temp_component, temp,
1);
// Close the 4x check.
a_.OpEndIf();
}
// MSAA is disabled.
a_.OpElse();
CompletePixelShader_AlphaToMaskSample(true, 0, 1.0f, temp_x_src, 1.0f / 4.0f,
coverage_temp, coverage_temp_component,
temp, 1);
// Close the 2x/4x check.
a_.OpEndIf();
// Check if any sample is still covered and return to avoid unneeded work (the
// driver's shader compiler may place return after a discard, but it will
// likely not place one during SV_Coverage assignment - that's what the AMD
// compiler does, at least). Then, if needed, write the coverage value.
if (edram_rov_used_) {
// The mask includes both 0:3 and 4:7 parts because there may be samples
// which passed alpha to coverage, but not stencil test, and the stencil
// buffer needs to be modified - in this case, samples would be dropped in
// 0:3, but not in 4:7).
a_.OpAnd(temp_x_dest,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(0b11111111));
a_.OpRetC(false, temp_x_src);
} else {
dxbc::Src coverage_src(
dxbc::Src::R(coverage_temp, coverage_temp_component));
a_.OpDiscard(false, coverage_src);
a_.OpMov(dxbc::Dest::OMask(), coverage_src);
}
// Release temp.
PopSystemTemp();
// Close the alpha to coverage check.
a_.OpEndIf();
}
void DxbcShaderTranslator::CompletePixelShader_WriteToROV() {
uint32_t temp = PushSystemTemp();
dxbc::Dest temp_x_dest(dxbc::Dest::R(temp, 0b0001));
dxbc::Src temp_x_src(dxbc::Src::R(temp, dxbc::Src::kXXXX));
dxbc::Dest temp_y_dest(dxbc::Dest::R(temp, 0b0010));
dxbc::Src temp_y_src(dxbc::Src::R(temp, dxbc::Src::kYYYY));
dxbc::Dest temp_z_dest(dxbc::Dest::R(temp, 0b0100));
dxbc::Src temp_z_src(dxbc::Src::R(temp, dxbc::Src::kZZZZ));
dxbc::Dest temp_w_dest(dxbc::Dest::R(temp, 0b1000));
dxbc::Src temp_w_src(dxbc::Src::R(temp, dxbc::Src::kWWWW));
// Do late depth/stencil test (which includes writing) if needed or deferred
// depth writing.
if (ROV_IsDepthStencilEarly()) {
// Write modified depth/stencil.
for (uint32_t i = 0; i < 4; ++i) {
// Get if need to write to temp.x.
// temp.x = whether the depth sample needs to be written.
a_.OpAnd(temp_x_dest,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(1 << (4 + i)));
// Check if need to write.
// temp.x = free.
a_.OpIf(true, temp_x_src);
{
// Write the new depth/stencil.
if (uav_index_edram_ == kBindingIndexUnallocated) {
uav_index_edram_ = uav_count_++;
}
a_.OpStoreUAVTyped(
dxbc::Dest::U(uav_index_edram_, uint32_t(UAVRegister::kEdram)),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY), 1,
dxbc::Src::R(system_temp_depth_stencil_).Select(i));
}
// Close the write check.
a_.OpEndIf();
// Go to the next sample (samples are at +0, +(80*scale_x), +1,
// +(80*scale_x+1), so need to do +(80*scale_x), -(80*scale_x-1),
// +(80*scale_x) and -(80*scale_x+1) after each sample).
if (i < 3) {
a_.OpIAdd(dxbc::Dest::R(system_temp_rov_params_, 0b0010),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kYYYY),
dxbc::Src::LI((i & 1) ? -80 * draw_resolution_scale_x_ + 2 - i
: 80 * draw_resolution_scale_x_));
}
}
} else {
ROV_DepthStencilTest();
}
if (!is_depth_only_pixel_shader_) {
// Check if any sample is still covered after depth testing and writing,
// skip color writing completely in this case.
// temp.x = whether any sample is still covered.
a_.OpAnd(temp_x_dest,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(0b1111));
// temp.x = free.
a_.OpRetC(false, temp_x_src);
}
// Write color values.
uint32_t shader_writes_color_targets =
current_shader().writes_color_targets();
for (uint32_t i = 0; i < 4; ++i) {
if (!(shader_writes_color_targets & (1 << i))) {
continue;
}
// This includes a swizzle to choose XY for even render targets or ZW for
// odd ones - use SelectFromSwizzled and SwizzleSwizzled.
dxbc::Src keep_mask_src(
LoadSystemConstant(SystemConstants::Index::kEdramRTKeepMask,
offsetof(SystemConstants, edram_rt_keep_mask) +
sizeof(uint32_t) * 2 * i,
0b0100));
// Check if color writing is disabled - special keep mask constant case,
// both 32bpp parts are forced UINT32_MAX, but also check whether the shader
// has written anything to this target at all.
// Combine both parts of the keep mask to check if both are 0xFFFFFFFF.
// temp.x = whether all bits need to be kept.
a_.OpAnd(temp_x_dest, keep_mask_src.SelectFromSwizzled(0),
keep_mask_src.SelectFromSwizzled(1));
// Flip the bits so both UINT32_MAX would result in 0 - not writing.
// temp.x = whether any bits need to be written.
a_.OpNot(temp_x_dest, temp_x_src);
// Get the bits that will be used for checking wherther the render target
// has been written to on the taken execution path - if the write mask is
// empty, AND zero with the test bit to always get zero.
// temp.x = bits for checking whether the render target has been written to.
a_.OpMovC(temp_x_dest, temp_x_src,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(0));
// Check if the render target was written to on the execution path.
// temp.x = whether anything was written and needs to be stored.
a_.OpAnd(temp_x_dest, temp_x_src, dxbc::Src::LU(1 << (8 + i)));
// Check if need to write anything to the render target.
// temp.x = free.
a_.OpIf(true, temp_x_src);
// Apply the exponent bias after alpha to coverage because it needs the
// unbiased alpha from the shader.
a_.OpMul(dxbc::Dest::R(system_temps_color_[i]),
dxbc::Src::R(system_temps_color_[i]),
LoadSystemConstant(
SystemConstants::Index::kColorExpBias,
offsetof(SystemConstants, color_exp_bias) + sizeof(float) * i,
dxbc::Src::kXXXX));
// Add the EDRAM bases of the render target to system_temp_rov_params_.zw.
a_.OpIAdd(dxbc::Dest::R(system_temp_rov_params_, 0b1100),
dxbc::Src::R(system_temp_rov_params_),
LoadSystemConstant(
SystemConstants::Index::kEdramRTBaseDwordsScaled,
offsetof(SystemConstants, edram_rt_base_dwords_scaled) +
sizeof(uint32_t) * i,
dxbc::Src::kXXXX));
dxbc::Src rt_blend_factors_ops_src(LoadSystemConstant(
SystemConstants::Index::kEdramRTBlendFactorsOps,
offsetof(SystemConstants, edram_rt_blend_factors_ops) +
sizeof(uint32_t) * i,
dxbc::Src::kXXXX));
dxbc::Src rt_clamp_vec_src(LoadSystemConstant(
SystemConstants::Index::kEdramRTClamp,
offsetof(SystemConstants, edram_rt_clamp) + sizeof(float) * 4 * i,
dxbc::Src::kXYZW));
dxbc::Src rt_format_flags_src(LoadSystemConstant(
SystemConstants::Index::kEdramRTFormatFlags,
offsetof(SystemConstants, edram_rt_format_flags) + sizeof(uint32_t) * i,
dxbc::Src::kXXXX));
// Get if not blending to pack the color once for all 4 samples.
// temp.x = whether blending is disabled.
a_.OpIEq(temp_x_dest, rt_blend_factors_ops_src, dxbc::Src::LU(0x00010001));
// Check if not blending.
// temp.x = free.
a_.OpIf(true, temp_x_src);
{
// Clamp the color to the render target's representable range - will be
// packed.
a_.OpMax(dxbc::Dest::R(system_temps_color_[i]),
dxbc::Src::R(system_temps_color_[i]),
rt_clamp_vec_src.Swizzle(0b01000000));
a_.OpMin(dxbc::Dest::R(system_temps_color_[i]),
dxbc::Src::R(system_temps_color_[i]),
rt_clamp_vec_src.Swizzle(0b11101010));
// Pack the color once if blending.
// temp.xy = packed color.
ROV_PackPreClampedColor(i, system_temps_color_[i], temp, 0, temp, 2, temp,
3);
}
// Blending is enabled.
a_.OpElse();
{
// Get if the blending source color is fixed-point for clamping if it is.
// temp.x = whether color is fixed-point.
a_.OpAnd(temp_x_dest, rt_format_flags_src,
dxbc::Src::LU(kRTFormatFlag_FixedPointColor));
// Check if the blending source color is fixed-point and needs clamping.
// temp.x = free.
a_.OpIf(true, temp_x_src);
{
// Clamp the blending source color if needed.
a_.OpMax(dxbc::Dest::R(system_temps_color_[i], 0b0111),
dxbc::Src::R(system_temps_color_[i]),
rt_clamp_vec_src.Select(0));
a_.OpMin(dxbc::Dest::R(system_temps_color_[i], 0b0111),
dxbc::Src::R(system_temps_color_[i]),
rt_clamp_vec_src.Select(2));
}
// Close the fixed-point color check.
a_.OpEndIf();
// Get if the blending source alpha is fixed-point for clamping if it is.
// temp.x = whether alpha is fixed-point.
a_.OpAnd(temp_x_dest, rt_format_flags_src,
dxbc::Src::LU(kRTFormatFlag_FixedPointAlpha));
// Check if the blending source alpha is fixed-point and needs clamping.
// temp.x = free.
a_.OpIf(true, temp_x_src);
{
// Clamp the blending source alpha if needed.
a_.OpMax(dxbc::Dest::R(system_temps_color_[i], 0b1000),
dxbc::Src::R(system_temps_color_[i], dxbc::Src::kWWWW),
rt_clamp_vec_src.Select(1));
a_.OpMin(dxbc::Dest::R(system_temps_color_[i], 0b1000),
dxbc::Src::R(system_temps_color_[i], dxbc::Src::kWWWW),
rt_clamp_vec_src.Select(3));
}
// Close the fixed-point alpha check.
a_.OpEndIf();
// Break register dependency in the color sample raster operation.
// temp.xy = 0 instead of packed color.
a_.OpMov(dxbc::Dest::R(temp, 0b0011), dxbc::Src::LU(0));
}
a_.OpEndIf();
// Blend, mask and write all samples.
for (uint32_t j = 0; j < 4; ++j) {
// Get if the sample is covered.
// temp.z = whether the sample is covered.
a_.OpAnd(temp_z_dest,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kXXXX),
dxbc::Src::LU(1 << j));
// Check if the sample is covered.
// temp.z = free.
a_.OpIf(true, temp_z_src);
// Only temp.xy are used at this point (containing the packed color from
// the shader if not blending).
// ***********************************************************************
// Color sample raster operation.
// ***********************************************************************
// ***********************************************************************
// Checking if color loading must be done - if any component needs to be
// kept or if blending is enabled.
// ***********************************************************************
// Get if need to keep any components to temp.z.
// temp.z = whether any components must be kept (OR of keep masks).
a_.OpOr(temp_z_dest, keep_mask_src.SelectFromSwizzled(0),
keep_mask_src.SelectFromSwizzled(1));
// Blending isn't done if it's 1 * source + 0 * destination. But since the
// previous color also needs to be loaded if any original components need
// to be kept, force the blend control to something with blending in this
// case in temp.z.
// temp.z = blending mode used to check if need to load.
a_.OpMovC(temp_z_dest, temp_z_src, dxbc::Src::LU(0),
rt_blend_factors_ops_src);
// Get if the blend control register requires loading the color to temp.z.
// temp.z = whether need to load the color.
a_.OpINE(temp_z_dest, temp_z_src, dxbc::Src::LU(0x00010001));
// Check if need to do something with the previous color.
// temp.z = free.
a_.OpIf(true, temp_z_src);
{
// *********************************************************************
// Loading the previous color to temp.zw.
// *********************************************************************
// Get if the format is 64bpp to temp.z.
// temp.z = whether the render target is 64bpp.
a_.OpAnd(temp_z_dest, rt_format_flags_src,
dxbc::Src::LU(kRTFormatFlag_64bpp));
// Check if the format is 64bpp.
// temp.z = free.
a_.OpIf(true, temp_z_src);
{
// Load the lower 32 bits of the 64bpp color to temp.z.
// temp.z = lower 32 bits of the packed color.
if (uav_index_edram_ == kBindingIndexUnallocated) {
uav_index_edram_ = uav_count_++;
}
a_.OpLdUAVTyped(
temp_z_dest,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kWWWW), 1,
dxbc::Src::U(uav_index_edram_, uint32_t(UAVRegister::kEdram),
dxbc::Src::kXXXX));
// Get the address of the upper 32 bits of the color to temp.w.
// temp.w = address of the upper 32 bits of the packed color.
a_.OpIAdd(temp_w_dest,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kWWWW),
dxbc::Src::LU(1));
// Load the upper 32 bits of the 64bpp color to temp.w.
// temp.zw = packed destination color/alpha.
if (uav_index_edram_ == kBindingIndexUnallocated) {
uav_index_edram_ = uav_count_++;
}
a_.OpLdUAVTyped(
temp_w_dest, temp_w_src, 1,
dxbc::Src::U(uav_index_edram_, uint32_t(UAVRegister::kEdram),
dxbc::Src::kXXXX));
}
// The color is 32bpp.
a_.OpElse();
{
// Load the 32bpp color to temp.z.
// temp.z = packed 32bpp destination color.
if (uav_index_edram_ == kBindingIndexUnallocated) {
uav_index_edram_ = uav_count_++;
}
a_.OpLdUAVTyped(
temp_z_dest,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kZZZZ), 1,
dxbc::Src::U(uav_index_edram_, uint32_t(UAVRegister::kEdram),
dxbc::Src::kXXXX));
// Break register dependency in temp.w if the color is 32bpp.
// temp.zw = packed destination color/alpha.
a_.OpMov(temp_w_dest, dxbc::Src::LU(0));
}
// Close the color format check.
a_.OpEndIf();
uint32_t color_temp = PushSystemTemp();
dxbc::Dest color_temp_rgb_dest(dxbc::Dest::R(color_temp, 0b0111));
dxbc::Dest color_temp_a_dest(dxbc::Dest::R(color_temp, 0b1000));
dxbc::Src color_temp_src(dxbc::Src::R(color_temp));
dxbc::Src color_temp_a_src(dxbc::Src::R(color_temp, dxbc::Src::kWWWW));
// Get if blending is enabled to color_temp.x.
// color_temp.x = whether blending is enabled.
a_.OpINE(dxbc::Dest::R(color_temp, 0b0001), rt_blend_factors_ops_src,
dxbc::Src::LU(0x00010001));
// Check if need to blend.
// color_temp.x = free.
a_.OpIf(true, dxbc::Src::R(color_temp, dxbc::Src::kXXXX));
{
// Now, when blending is enabled, temp.xy are used as scratch since
// the color is packed after blending.
// Unpack the destination color to color_temp, using temp.xy as temps.
// The destination color never needs clamping because out-of-range
// values can't be loaded.
// color_temp.xyzw = destination color/alpha.
ROV_UnpackColor(i, temp, 2, color_temp, temp, 0, temp, 1);
// *******************************************************************
// Color blending.
// *******************************************************************
// Extract the color min/max bit to temp.x.
// temp.x = whether min/max should be used for color.
a_.OpAnd(temp_x_dest, rt_blend_factors_ops_src,
dxbc::Src::LU(1 << (5 + 1)));
// Check if need to do blend the color with factors.
// temp.x = free.
a_.OpIf(false, temp_x_src);
{
uint32_t blend_src_temp = PushSystemTemp();
dxbc::Dest blend_src_temp_rgb_dest(
dxbc::Dest::R(blend_src_temp, 0b0111));
dxbc::Src blend_src_temp_src(dxbc::Src::R(blend_src_temp));
// Extract the source color factor to temp.x.
// temp.x = source color factor index.
a_.OpAnd(temp_x_dest, rt_blend_factors_ops_src,
dxbc::Src::LU((1 << 5) - 1));
// Check if the source color factor is not zero - if it is, the
// source must be ignored completely, and Infinity and NaN in it
// shouldn't affect blending.
a_.OpIf(true, temp_x_src);
{
// Open the switch for choosing the source color blend factor.
// temp.x = free.
a_.OpSwitch(temp_x_src);
// Write the source color factor to blend_src_temp.xyz.
// blend_src_temp.xyz = unclamped source color factor.
ROV_HandleColorBlendFactorCases(system_temps_color_[i],
color_temp, blend_src_temp);
// Close the source color factor switch.
a_.OpEndSwitch();
// Get if the render target color is fixed-point and the source
// color factor needs clamping to temp.x.
// temp.x = whether color is fixed-point.
a_.OpAnd(temp_x_dest, rt_format_flags_src,
dxbc::Src::LU(kRTFormatFlag_FixedPointColor));
// Check if the source color factor needs clamping.
a_.OpIf(true, temp_x_src);
{
// Clamp the source color factor in blend_src_temp.xyz.
// blend_src_temp.xyz = source color factor.
a_.OpMax(blend_src_temp_rgb_dest, blend_src_temp_src,
rt_clamp_vec_src.Select(0));
a_.OpMin(blend_src_temp_rgb_dest, blend_src_temp_src,
rt_clamp_vec_src.Select(2));
}
// Close the source color factor clamping check.
a_.OpEndIf();
// Apply the factor to the source color.
// blend_src_temp.xyz = unclamped source color part without
// addition sign.
a_.OpMul(blend_src_temp_rgb_dest,
dxbc::Src::R(system_temps_color_[i]),
blend_src_temp_src);
// Check if the source color part needs clamping after the
// multiplication.
// temp.x = free.
a_.OpIf(true, temp_x_src);
{
// Clamp the source color part.
// blend_src_temp.xyz = source color part without addition sign.
a_.OpMax(blend_src_temp_rgb_dest, blend_src_temp_src,
rt_clamp_vec_src.Select(0));
a_.OpMin(blend_src_temp_rgb_dest, blend_src_temp_src,
rt_clamp_vec_src.Select(2));
}
// Close the source color part clamping check.
a_.OpEndIf();
// Extract the source color sign to temp.x.
// temp.x = source color sign as zero for 1 and non-zero for -1.
a_.OpAnd(temp_x_dest, rt_blend_factors_ops_src,
dxbc::Src::LU(1 << (5 + 2)));
// Apply the source color sign.
// blend_src_temp.xyz = source color part.
// temp.x = free.
a_.OpMovC(blend_src_temp_rgb_dest, temp_x_src,
-blend_src_temp_src, blend_src_temp_src);
}
// The source color factor is zero.
a_.OpElse();
{
// Write zero to the source color part.
// blend_src_temp.xyz = source color part.
// temp.x = free.
a_.OpMov(blend_src_temp_rgb_dest, dxbc::Src::LF(0.0f));
}
// Close the source color factor zero check.
a_.OpEndIf();
// Extract the destination color factor to temp.x.
// temp.x = destination color factor index.
a_.OpUBFE(temp_x_dest, dxbc::Src::LU(5), dxbc::Src::LU(8),
rt_blend_factors_ops_src);
// Check if the destination color factor is not zero.
a_.OpIf(true, temp_x_src);
{
uint32_t blend_dest_factor_temp = PushSystemTemp();
dxbc::Src blend_dest_factor_temp_src(
dxbc::Src::R(blend_dest_factor_temp));
// Open the switch for choosing the destination color blend
// factor.
// temp.x = free.
a_.OpSwitch(temp_x_src);
// Write the destination color factor to
// blend_dest_factor_temp.xyz.
// blend_dest_factor_temp.xyz = unclamped destination color
// factor.
ROV_HandleColorBlendFactorCases(
system_temps_color_[i], color_temp, blend_dest_factor_temp);
// Close the destination color factor switch.
a_.OpEndSwitch();
// Get if the render target color is fixed-point and the
// destination color factor needs clamping to temp.x.
// temp.x = whether color is fixed-point.
a_.OpAnd(temp_x_dest, rt_format_flags_src,
dxbc::Src::LU(kRTFormatFlag_FixedPointColor));
// Check if the destination color factor needs clamping.
a_.OpIf(true, temp_x_src);
{
// Clamp the destination color factor in
// blend_dest_factor_temp.xyz.
// blend_dest_factor_temp.xyz = destination color factor.
a_.OpMax(dxbc::Dest::R(blend_dest_factor_temp, 0b0111),
blend_dest_factor_temp_src,
rt_clamp_vec_src.Select(0));
a_.OpMin(dxbc::Dest::R(blend_dest_factor_temp, 0b0111),
blend_dest_factor_temp_src,
rt_clamp_vec_src.Select(2));
}
// Close the destination color factor clamping check.
a_.OpEndIf();
// Apply the factor to the destination color in color_temp.xyz.
// color_temp.xyz = unclamped destination color part without
// addition sign.
// blend_dest_temp.xyz = free.
a_.OpMul(color_temp_rgb_dest, color_temp_src,
blend_dest_factor_temp_src);
// Release blend_dest_factor_temp.
PopSystemTemp();
// Check if the destination color part needs clamping after the
// multiplication.
// temp.x = free.
a_.OpIf(true, temp_x_src);
{
// Clamp the destination color part.
// color_temp.xyz = destination color part without addition
// sign.
a_.OpMax(color_temp_rgb_dest, color_temp_src,
rt_clamp_vec_src.Select(0));
a_.OpMin(color_temp_rgb_dest, color_temp_src,
rt_clamp_vec_src.Select(2));
}
// Close the destination color part clamping check.
a_.OpEndIf();
// Extract the destination color sign to temp.x.
// temp.x = destination color sign as zero for 1 and non-zero for
// -1.
a_.OpAnd(temp_x_dest, rt_blend_factors_ops_src,
dxbc::Src::LU(1 << 5));
// Select the sign for destination multiply-add as 1.0 or -1.0 to
// temp.x.
// temp.x = destination color sign as float.
a_.OpMovC(temp_x_dest, temp_x_src, dxbc::Src::LF(-1.0f),
dxbc::Src::LF(1.0f));
// Perform color blending to color_temp.xyz.
// color_temp.xyz = unclamped blended color.
// blend_src_temp.xyz = free.
// temp.x = free.
a_.OpMAd(color_temp_rgb_dest, color_temp_src, temp_x_src,
blend_src_temp_src);
}
// The destination color factor is zero.
a_.OpElse();
{
// Write the source color part without applying the destination
// color.
// color_temp.xyz = unclamped blended color.
// blend_src_temp.xyz = free.
// temp.x = free.
a_.OpMov(color_temp_rgb_dest, blend_src_temp_src);
}
// Close the destination color factor zero check.
a_.OpEndIf();
// Release blend_src_temp.
PopSystemTemp();
// Clamp the color in color_temp.xyz before packing.
// color_temp.xyz = blended color.
a_.OpMax(color_temp_rgb_dest, color_temp_src,
rt_clamp_vec_src.Select(0));
a_.OpMin(color_temp_rgb_dest, color_temp_src,
rt_clamp_vec_src.Select(2));
}
// Need to do min/max for color.
a_.OpElse();
{
// Extract the color min (0) or max (1) bit to temp.x
// temp.x = whether min or max should be used for color.
a_.OpAnd(temp_x_dest, rt_blend_factors_ops_src,
dxbc::Src::LU(1 << 5));
// Check if need to do min or max for color.
// temp.x = free.
a_.OpIf(true, temp_x_src);
{
// Choose max of the colors without applying the factors to
// color_temp.xyz.
// color_temp.xyz = blended color.
a_.OpMax(color_temp_rgb_dest,
dxbc::Src::R(system_temps_color_[i]), color_temp_src);
}
// Need to do min.
a_.OpElse();
{
// Choose min of the colors without applying the factors to
// color_temp.xyz.
// color_temp.xyz = blended color.
a_.OpMin(color_temp_rgb_dest,
dxbc::Src::R(system_temps_color_[i]), color_temp_src);
}
// Close the min or max check.
a_.OpEndIf();
}
// Close the color factor blending or min/max check.
a_.OpEndIf();
// *******************************************************************
// Alpha blending.
// *******************************************************************
// Extract the alpha min/max bit to temp.x.
// temp.x = whether min/max should be used for alpha.
a_.OpAnd(temp_x_dest, rt_blend_factors_ops_src,
dxbc::Src::LU(1 << (21 + 1)));
// Check if need to do blend the color with factors.
// temp.x = free.
a_.OpIf(false, temp_x_src);
{
// Extract the source alpha factor to temp.x.
// temp.x = source alpha factor index.
a_.OpUBFE(temp_x_dest, dxbc::Src::LU(5), dxbc::Src::LU(16),
rt_blend_factors_ops_src);
// Check if the source alpha factor is not zero.
a_.OpIf(true, temp_x_src);
{
// Open the switch for choosing the source alpha blend factor.
// temp.x = free.
a_.OpSwitch(temp_x_src);
// Write the source alpha factor to temp.x.
// temp.x = unclamped source alpha factor.
ROV_HandleAlphaBlendFactorCases(system_temps_color_[i],
color_temp, temp, 0);
// Close the source alpha factor switch.
a_.OpEndSwitch();
// Get if the render target alpha is fixed-point and the source
// alpha factor needs clamping to temp.y.
// temp.y = whether alpha is fixed-point.
a_.OpAnd(temp_y_dest, rt_format_flags_src,
dxbc::Src::LU(kRTFormatFlag_FixedPointAlpha));
// Check if the source alpha factor needs clamping.
a_.OpIf(true, temp_y_src);
{
// Clamp the source alpha factor in temp.x.
// temp.x = source alpha factor.
a_.OpMax(temp_x_dest, temp_x_src, rt_clamp_vec_src.Select(1));
a_.OpMin(temp_x_dest, temp_x_src, rt_clamp_vec_src.Select(3));
}
// Close the source alpha factor clamping check.
a_.OpEndIf();
// Apply the factor to the source alpha.
// temp.x = unclamped source alpha part without addition sign.
a_.OpMul(temp_x_dest,
dxbc::Src::R(system_temps_color_[i], dxbc::Src::kWWWW),
temp_x_src);
// Check if the source alpha part needs clamping after the
// multiplication.
// temp.y = free.
a_.OpIf(true, temp_y_src);
{
// Clamp the source alpha part.
// temp.x = source alpha part without addition sign.
a_.OpMax(temp_x_dest, temp_x_src, rt_clamp_vec_src.Select(1));
a_.OpMin(temp_x_dest, temp_x_src, rt_clamp_vec_src.Select(3));
}
// Close the source alpha part clamping check.
a_.OpEndIf();
// Extract the source alpha sign to temp.y.
// temp.y = source alpha sign as zero for 1 and non-zero for -1.
a_.OpAnd(temp_y_dest, rt_blend_factors_ops_src,
dxbc::Src::LU(1 << (21 + 2)));
// Apply the source alpha sign.
// temp.x = source alpha part.
a_.OpMovC(temp_x_dest, temp_y_src, -temp_x_src, temp_x_src);
}
// The source alpha factor is zero.
a_.OpElse();
{
// Write zero to the source alpha part.
// temp.x = source alpha part.
a_.OpMov(temp_x_dest, dxbc::Src::LF(0.0f));
}
// Close the source alpha factor zero check.
a_.OpEndIf();
// Extract the destination alpha factor to temp.y.
// temp.y = destination alpha factor index.
a_.OpUBFE(temp_y_dest, dxbc::Src::LU(5), dxbc::Src::LU(24),
rt_blend_factors_ops_src);
// Check if the destination alpha factor is not zero.
a_.OpIf(true, temp_y_src);
{
// Open the switch for choosing the destination alpha blend
// factor.
// temp.y = free.
a_.OpSwitch(temp_y_src);
// Write the destination alpha factor to temp.y.
// temp.y = unclamped destination alpha factor.
ROV_HandleAlphaBlendFactorCases(system_temps_color_[i],
color_temp, temp, 1);
// Close the destination alpha factor switch.
a_.OpEndSwitch();
// Get if the render target alpha is fixed-point and the
// destination alpha factor needs clamping.
// alpha_is_fixed_temp.x = whether alpha is fixed-point.
uint32_t alpha_is_fixed_temp = PushSystemTemp();
a_.OpAnd(dxbc::Dest::R(alpha_is_fixed_temp, 0b0001),
rt_format_flags_src,
dxbc::Src::LU(kRTFormatFlag_FixedPointAlpha));
// Check if the destination alpha factor needs clamping.
a_.OpIf(true,
dxbc::Src::R(alpha_is_fixed_temp, dxbc::Src::kXXXX));
{
// Clamp the destination alpha factor in temp.y.
// temp.y = destination alpha factor.
a_.OpMax(temp_y_dest, temp_y_src, rt_clamp_vec_src.Select(1));
a_.OpMin(temp_y_dest, temp_y_src, rt_clamp_vec_src.Select(3));
}
// Close the destination alpha factor clamping check.
a_.OpEndIf();
// Apply the factor to the destination alpha in color_temp.w.
// color_temp.w = unclamped destination alpha part without
// addition sign.
a_.OpMul(color_temp_a_dest, color_temp_a_src, temp_y_src);
// Check if the destination alpha part needs clamping after the
// multiplication.
// alpha_is_fixed_temp.x = free.
a_.OpIf(true,
dxbc::Src::R(alpha_is_fixed_temp, dxbc::Src::kXXXX));
// Release alpha_is_fixed_temp.
PopSystemTemp();
{
// Clamp the destination alpha part.
// color_temp.w = destination alpha part without addition sign.
a_.OpMax(color_temp_a_dest, color_temp_a_src,
rt_clamp_vec_src.Select(1));
a_.OpMin(color_temp_a_dest, color_temp_a_src,
rt_clamp_vec_src.Select(3));
}
// Close the destination alpha factor clamping check.
a_.OpEndIf();
// Extract the destination alpha sign to temp.y.
// temp.y = destination alpha sign as zero for 1 and non-zero for
// -1.
a_.OpAnd(temp_y_dest, rt_blend_factors_ops_src,
dxbc::Src::LU(1 << 21));
// Select the sign for destination multiply-add as 1.0 or -1.0 to
// temp.y.
// temp.y = destination alpha sign as float.
a_.OpMovC(temp_y_dest, temp_y_src, dxbc::Src::LF(-1.0f),
dxbc::Src::LF(1.0f));
// Perform alpha blending to color_temp.w.
// color_temp.w = unclamped blended alpha.
// temp.xy = free.
a_.OpMAd(color_temp_a_dest, color_temp_a_src, temp_y_src,
temp_x_src);
}
// The destination alpha factor is zero.
a_.OpElse();
{
// Write the source alpha part without applying the destination
// alpha.
// color_temp.w = unclamped blended alpha.
// temp.xy = free.
a_.OpMov(color_temp_a_dest, temp_x_src);
}
// Close the destination alpha factor zero check.
a_.OpEndIf();
// Clamp the alpha in color_temp.w before packing.
// color_temp.w = blended alpha.
a_.OpMax(color_temp_a_dest, color_temp_a_src,
rt_clamp_vec_src.Select(1));
a_.OpMin(color_temp_a_dest, color_temp_a_src,
rt_clamp_vec_src.Select(3));
}
// Need to do min/max for alpha.
a_.OpElse();
{
// Extract the alpha min (0) or max (1) bit to temp.x.
// temp.x = whether min or max should be used for alpha.
a_.OpAnd(temp_x_dest, rt_blend_factors_ops_src,
dxbc::Src::LU(1 << 21));
// Check if need to do min or max for alpha.
// temp.x = free.
a_.OpIf(true, temp_x_src);
{
// Choose max of the alphas without applying the factors to
// color_temp.w.
// color_temp.w = blended alpha.
a_.OpMax(color_temp_a_dest,
dxbc::Src::R(system_temps_color_[i], dxbc::Src::kWWWW),
color_temp_a_src);
}
// Need to do min.
a_.OpElse();
{
// Choose min of the alphas without applying the factors to
// color_temp.w.
// color_temp.w = blended alpha.
a_.OpMin(color_temp_a_dest,
dxbc::Src::R(system_temps_color_[i], dxbc::Src::kWWWW),
color_temp_a_src);
}
// Close the min or max check.
a_.OpEndIf();
}
// Close the alpha factor blending or min/max check.
a_.OpEndIf();
// Pack the new color/alpha to temp.xy.
// temp.xy = packed new color/alpha.
uint32_t color_pack_temp = PushSystemTemp();
ROV_PackPreClampedColor(i, color_temp, temp, 0, color_pack_temp, 0,
color_pack_temp, 1);
// Release color_pack_temp.
PopSystemTemp();
}
// Close the blending check.
a_.OpEndIf();
// *********************************************************************
// Write mask application
// *********************************************************************
// Apply the keep mask to the previous packed color/alpha in temp.zw.
// temp.zw = masked packed old color/alpha.
a_.OpAnd(dxbc::Dest::R(temp, 0b1100), dxbc::Src::R(temp),
keep_mask_src.SwizzleSwizzled(0b0100 << 4));
// Invert the keep mask into color_temp.xy.
// color_temp.xy = inverted keep mask (write mask).
a_.OpNot(dxbc::Dest::R(color_temp, 0b0011), keep_mask_src);
// Release color_temp.
PopSystemTemp();
// Apply the write mask to the new color/alpha in temp.xy.
// temp.xy = masked packed new color/alpha.
a_.OpAnd(dxbc::Dest::R(temp, 0b0011), dxbc::Src::R(temp),
dxbc::Src::R(color_temp));
// Combine the masked colors into temp.xy.
// temp.xy = packed resulting color/alpha.
// temp.zw = free.
a_.OpOr(dxbc::Dest::R(temp, 0b0011), dxbc::Src::R(temp),
dxbc::Src::R(temp, 0b1110));
}
// Close the previous color load check.
a_.OpEndIf();
// ***********************************************************************
// Writing the color
// ***********************************************************************
// Get if the format is 64bpp to temp.z.
// temp.z = whether the render target is 64bpp.
a_.OpAnd(temp_z_dest, rt_format_flags_src,
dxbc::Src::LU(kRTFormatFlag_64bpp));
// Check if the format is 64bpp.
// temp.z = free.
a_.OpIf(true, temp_z_src);
{
// Store the lower 32 bits of the 64bpp color.
if (uav_index_edram_ == kBindingIndexUnallocated) {
uav_index_edram_ = uav_count_++;
}
a_.OpStoreUAVTyped(
dxbc::Dest::U(uav_index_edram_, uint32_t(UAVRegister::kEdram)),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kWWWW), 1,
temp_x_src);
// Get the address of the upper 32 bits of the color to temp.z (can't
// use temp.x because components when not blending, packing is done once
// for all samples, so it has to be preserved).
a_.OpIAdd(temp_z_dest,
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kWWWW),
dxbc::Src::LU(1));
// Store the upper 32 bits of the 64bpp color.
if (uav_index_edram_ == kBindingIndexUnallocated) {
uav_index_edram_ = uav_count_++;
}
a_.OpStoreUAVTyped(
dxbc::Dest::U(uav_index_edram_, uint32_t(UAVRegister::kEdram)),
temp_z_src, 1, temp_y_src);
}
// The color is 32bpp.
a_.OpElse();
{
// Store the 32bpp color.
if (uav_index_edram_ == kBindingIndexUnallocated) {
uav_index_edram_ = uav_count_++;
}
a_.OpStoreUAVTyped(
dxbc::Dest::U(uav_index_edram_, uint32_t(UAVRegister::kEdram)),
dxbc::Src::R(system_temp_rov_params_, dxbc::Src::kZZZZ), 1,
temp_x_src);
}
// Close the 64bpp/32bpp conditional.
a_.OpEndIf();
// ***********************************************************************
// End of color sample raster operation.
// ***********************************************************************
// Close the sample covered check.
a_.OpEndIf();
// Go to the next sample (samples are at +0, +(80*scale_x), +1,
// +(80*scale_x+1), so need to do +(80*scale_x), -(80*scale_x-1),
// +(80*scale_x) and -(80*scale_x+1) after each sample).
int32_t next_sample_distance =
(j & 1) ? -80 * draw_resolution_scale_x_ + 2 - j
: 80 * draw_resolution_scale_x_;
a_.OpIAdd(
dxbc::Dest::R(system_temp_rov_params_, 0b1100),
dxbc::Src::R(system_temp_rov_params_),
dxbc::Src::LI(0, 0, next_sample_distance, next_sample_distance));
}
// Revert adding the EDRAM bases of the render target to
// system_temp_rov_params_.zw.
a_.OpIAdd(dxbc::Dest::R(system_temp_rov_params_, 0b1100),
dxbc::Src::R(system_temp_rov_params_),
-LoadSystemConstant(
SystemConstants::Index::kEdramRTBaseDwordsScaled,
offsetof(SystemConstants, edram_rt_base_dwords_scaled) +
sizeof(uint32_t) * i,
dxbc::Src::kXXXX));
// Close the render target write check.
a_.OpEndIf();
}
// Release temp.
PopSystemTemp();
}
void DxbcShaderTranslator::CompletePixelShader() {
if (is_depth_only_pixel_shader_) {
// The depth-only shader only needs to do the depth test and to write the
// depth to the ROV.
if (edram_rov_used_) {
CompletePixelShader_WriteToROV();
}
return;
}
if (current_shader().writes_color_target(0)) {
// Alpha test.
// X - mask, then masked result (SGPR for loading, VGPR for masking).
// Y - operation result (SGPR for mask operations, VGPR for alpha
// operations).
uint32_t alpha_test_temp = PushSystemTemp();
dxbc::Dest alpha_test_mask_dest(dxbc::Dest::R(alpha_test_temp, 0b0001));
dxbc::Src alpha_test_mask_src(
dxbc::Src::R(alpha_test_temp, dxbc::Src::kXXXX));
dxbc::Dest alpha_test_op_dest(dxbc::Dest::R(alpha_test_temp, 0b0010));
dxbc::Src alpha_test_op_src(
dxbc::Src::R(alpha_test_temp, dxbc::Src::kYYYY));
// Extract the comparison mask to check if the test needs to be done at all.
// Don't care about flow control being somewhat dynamic - early Z is forced
// using a special version of the shader anyway.
a_.OpUBFE(alpha_test_mask_dest, dxbc::Src::LU(3),
dxbc::Src::LU(kSysFlag_AlphaPassIfLess_Shift),
LoadFlagsSystemConstant());
// Compare the mask to ALWAYS to check if the test shouldn't be done (will
// pass even for NaNs, though the expected behavior in this case hasn't been
// checked, but let's assume this means "always", not "less, equal or
// greater".
// TODO(Triang3l): Check how alpha test works with NaN on Direct3D 9.
a_.OpINE(alpha_test_op_dest, alpha_test_mask_src, dxbc::Src::LU(0b111));
// Don't do the test if the mode is "always".
a_.OpIf(true, alpha_test_op_src);
{
// Do the test. Can't use subtraction and sign because of float specials.
dxbc::Src alpha_src(
dxbc::Src::R(system_temps_color_[0], dxbc::Src::kWWWW));
dxbc::Src alpha_test_reference_src(LoadSystemConstant(
SystemConstants::Index::kAlphaTestReference,
offsetof(SystemConstants, alpha_test_reference), dxbc::Src::kXXXX));
// Less than.
a_.OpLT(alpha_test_op_dest, alpha_src, alpha_test_reference_src);
a_.OpOr(alpha_test_op_dest, alpha_test_op_src,
dxbc::Src::LU(~uint32_t(1 << 0)));
a_.OpAnd(alpha_test_mask_dest, alpha_test_mask_src, alpha_test_op_src);
// Equals to.
a_.OpEq(alpha_test_op_dest, alpha_src, alpha_test_reference_src);
a_.OpOr(alpha_test_op_dest, alpha_test_op_src,
dxbc::Src::LU(~uint32_t(1 << 1)));
a_.OpAnd(alpha_test_mask_dest, alpha_test_mask_src, alpha_test_op_src);
// Greater than.
a_.OpLT(alpha_test_op_dest, alpha_test_reference_src, alpha_src);
a_.OpOr(alpha_test_op_dest, alpha_test_op_src,
dxbc::Src::LU(~uint32_t(1 << 2)));
a_.OpAnd(alpha_test_mask_dest, alpha_test_mask_src, alpha_test_op_src);
// Discard the pixel if it has failed the test.
if (edram_rov_used_) {
a_.OpRetC(false, alpha_test_mask_src);
} else {
a_.OpDiscard(false, alpha_test_mask_src);
}
}
// Close the "not always" check.
a_.OpEndIf();
// Release alpha_test_temp.
PopSystemTemp();
}
// Discard samples with alpha to coverage.
CompletePixelShader_AlphaToMask();
// Write the values to the render targets. Not applying the exponent bias yet
// because the original 0 to 1 alpha value is needed for alpha to coverage,
// which is done differently for ROV and RTV/DSV.
if (edram_rov_used_) {
CompletePixelShader_WriteToROV();
} else {
CompletePixelShader_WriteToRTVs();
CompletePixelShader_DSV_DepthTo24Bit();
}
}
void DxbcShaderTranslator::PreClampedFloat32To7e3(
dxbc::Assembler& a, uint32_t f10_temp, uint32_t f10_temp_component,
uint32_t f32_temp, uint32_t f32_temp_component, uint32_t temp_temp,
uint32_t temp_temp_component) {
assert_true(temp_temp != f10_temp ||
temp_temp_component != f10_temp_component);
assert_true(temp_temp != f32_temp ||
temp_temp_component != f32_temp_component);
// Source and destination may be the same.
dxbc::Dest f10_dest(dxbc::Dest::R(f10_temp, 1 << f10_temp_component));
dxbc::Src f10_src(dxbc::Src::R(f10_temp).Select(f10_temp_component));
dxbc::Src f32_src(dxbc::Src::R(f32_temp).Select(f32_temp_component));
dxbc::Dest temp_dest(dxbc::Dest::R(temp_temp, 1 << temp_temp_component));
dxbc::Src temp_src(dxbc::Src::R(temp_temp).Select(temp_temp_component));
// https://github.com/Microsoft/DirectXTex/blob/master/DirectXTex/DirectXTexConvert.cpp
// Assuming the color is already clamped to [0, 31.875].
// Check if the number is too small to be represented as normalized 7e3.
// temp = f32 < 2^-2
a.OpULT(temp_dest, f32_src, dxbc::Src::LU(0x3E800000));
// Handle denormalized numbers separately.
a.OpIf(true, temp_src);
{
// temp = f32 >> 23
a.OpUShR(temp_dest, f32_src, dxbc::Src::LU(23));
// temp = 125 - (f32 >> 23)
a.OpIAdd(temp_dest, dxbc::Src::LI(125), -temp_src);
// Don't allow the shift to overflow, since in DXBC the lower 5 bits of the
// shift amount are used.
// temp = min(125 - (f32 >> 23), 24)
a.OpUMin(temp_dest, temp_src, dxbc::Src::LU(24));
// biased_f32 = (f32 & 0x7FFFFF) | 0x800000
a.OpBFI(f10_dest, dxbc::Src::LU(9), dxbc::Src::LU(23), dxbc::Src::LU(1),
f32_src);
// biased_f32 = ((f32 & 0x7FFFFF) | 0x800000) >> min(125 - (f32 >> 23), 24)
a.OpUShR(f10_dest, f10_src, temp_src);
}
// Not denormalized?
a.OpElse();
{
// Bias the exponent.
// biased_f32 = f32 + (-124 << 23)
// (left shift of a negative value is undefined behavior)
a.OpIAdd(f10_dest, f32_src, dxbc::Src::LU(0xC2000000u));
}
// Close the denormal check.
a.OpEndIf();
// Build the 7e3 number.
// temp = (biased_f32 >> 16) & 1
a.OpUBFE(temp_dest, dxbc::Src::LU(1), dxbc::Src::LU(16), f10_src);
// f10 = biased_f32 + 0x7FFF
a.OpIAdd(f10_dest, f10_src, dxbc::Src::LU(0x7FFF));
// f10 = biased_f32 + 0x7FFF + ((biased_f32 >> 16) & 1)
a.OpIAdd(f10_dest, f10_src, temp_src);
// f24 = ((biased_f32 + 0x7FFF + ((biased_f32 >> 16) & 1)) >> 16) & 0x3FF
a.OpUBFE(f10_dest, dxbc::Src::LU(10), dxbc::Src::LU(16), f10_src);
}
void DxbcShaderTranslator::UnclampedFloat32To7e3(
dxbc::Assembler& a, uint32_t f10_temp, uint32_t f10_temp_component,
uint32_t f32_temp, uint32_t f32_temp_component, uint32_t temp_temp,
uint32_t temp_temp_component) {
// Source and destination might be the same or different, just like in
// PreClampedFloat32To7e3 - clamp to the destination and use it as source.
a.OpMax(dxbc::Dest::R(f10_temp, 1 << f10_temp_component),
dxbc::Src::R(f32_temp).Select(f32_temp_component),
dxbc::Src::LF(0.0f));
a.OpMin(dxbc::Dest::R(f10_temp, 1 << f10_temp_component),
dxbc::Src::R(f10_temp).Select(f10_temp_component),
dxbc::Src::LF(31.875f));
PreClampedFloat32To7e3(a, f10_temp, f10_temp_component, f10_temp,
f10_temp_component, temp_temp, temp_temp_component);
}
void DxbcShaderTranslator::Float7e3To32(
dxbc::Assembler& a, const dxbc::Dest& f32, uint32_t f10_temp,
uint32_t f10_temp_component, uint32_t f10_shift, uint32_t temp1_temp,
uint32_t temp1_temp_component, uint32_t temp2_temp,
uint32_t temp2_temp_component) {
assert_true(f10_shift <= (32 - 10));
assert_true(temp1_temp != temp2_temp ||
temp1_temp_component != temp2_temp_component);
// Source may be the same as temp1 or temp2.
dxbc::Dest exponent_dest(
dxbc::Dest::R(temp1_temp, 1 << temp1_temp_component));
dxbc::Src exponent_src(dxbc::Src::R(temp1_temp).Select(temp1_temp_component));
dxbc::Dest mantissa_dest(
dxbc::Dest::R(temp2_temp, 1 << temp2_temp_component));
dxbc::Src mantissa_src(dxbc::Src::R(temp2_temp).Select(temp2_temp_component));
// https://github.com/Microsoft/DirectXTex/blob/master/DirectXTex/DirectXTexConvert.cpp
if (!(f10_temp == temp1_temp && f10_temp_component == temp1_temp_component)) {
// Unpack the exponent before the mantissa if that doesn't overwrite the
// source.
a.OpUBFE(exponent_dest, dxbc::Src::LU(3), dxbc::Src::LU(f10_shift + 7),
dxbc::Src::R(f10_temp).Select(f10_temp_component));
}
// Unpack the mantissa.
a.OpUBFE(mantissa_dest, dxbc::Src::LU(7), dxbc::Src::LU(f10_shift),
dxbc::Src::R(f10_temp).Select(f10_temp_component));
if (f10_temp == temp1_temp && f10_temp_component == temp1_temp_component) {
// Unpack the exponent after the mantissa if doing that before the mantissa
// would overwrite the source.
a.OpUBFE(exponent_dest, dxbc::Src::LU(3), dxbc::Src::LU(f10_shift + 7),
dxbc::Src::R(f10_temp).Select(f10_temp_component));
}
// Check if the number is denormalized.
a.OpIf(false, exponent_src);
{
// Check if the number is non-zero (if the mantissa isn't zero - the
// exponent is known to be zero at this point).
a.OpIf(true, mantissa_src);
{
// Normalize the mantissa.
// Note that HLSL firstbithigh(x) is compiled to DXBC like:
// `x ? 31 - firstbit_hi(x) : -1`
// (returns the index from the LSB, not the MSB, but -1 for zero too).
// exponent = firstbit_hi(mantissa)
a.OpFirstBitHi(exponent_dest, mantissa_src);
// exponent = 7 - firstbithigh(mantissa)
// Or:
// exponent = 7 - (31 - firstbit_hi(mantissa))
a.OpIAdd(exponent_dest, exponent_src, dxbc::Src::LI(7 - 31));
// mantissa = mantissa << (7 - firstbithigh(mantissa))
// AND 0x7F not needed after this - BFI will do it.
a.OpIShL(mantissa_dest, mantissa_src, exponent_src);
// Get the normalized exponent.
// exponent = 1 - (7 - firstbithigh(mantissa))
a.OpIAdd(exponent_dest, dxbc::Src::LI(1), -exponent_src);
}
// The number is zero.
a.OpElse();
{
// Set the unbiased exponent to -124 for zero - 124 will be added later,
// resulting in zero float32.
a.OpMov(exponent_dest, dxbc::Src::LI(-124));
}
// Close the non-zero check.
a.OpEndIf();
}
// Close the denormal check.
a.OpEndIf();
// Bias the exponent and move it to the correct location in f32.
a.OpIMAd(exponent_dest, exponent_src, dxbc::Src::LI(1 << 23),
dxbc::Src::LI(124 << 23));
// Combine the mantissa and the exponent.
a.OpBFI(f32, dxbc::Src::LU(7), dxbc::Src::LU(23 - 7), mantissa_src,
exponent_src);
}
void DxbcShaderTranslator::PreClampedDepthTo20e4(
dxbc::Assembler& a, uint32_t f24_temp, uint32_t f24_temp_component,
uint32_t f32_temp, uint32_t f32_temp_component, uint32_t temp_temp,
uint32_t temp_temp_component, bool remap_from_0_to_0_5) {
assert_true(temp_temp != f24_temp ||
temp_temp_component != f24_temp_component);
assert_true(temp_temp != f32_temp ||
temp_temp_component != f32_temp_component);
// Source and destination may be the same.
dxbc::Dest f24_dest(dxbc::Dest::R(f24_temp, 1 << f24_temp_component));
dxbc::Src f24_src(dxbc::Src::R(f24_temp).Select(f24_temp_component));
dxbc::Src f32_src(dxbc::Src::R(f32_temp).Select(f32_temp_component));
dxbc::Dest temp_dest(dxbc::Dest::R(temp_temp, 1 << temp_temp_component));
dxbc::Src temp_src(dxbc::Src::R(temp_temp).Select(temp_temp_component));
// CFloat24 from d3dref9.dll +
// https://github.com/Microsoft/DirectXTex/blob/master/DirectXTex/DirectXTexConvert.cpp
// Assuming the depth is already clamped to [0, 2) (in all places, the depth
// is written with the saturate flag set).
uint32_t remap_bias = uint32_t(remap_from_0_to_0_5);
// Check if the number is too small to be represented as normalized 20e4.
// temp = f32 < 2^-14
a.OpULT(temp_dest, f32_src, dxbc::Src::LU(0x38800000 - (remap_bias << 23)));
// Handle denormalized numbers separately.
a.OpIf(true, temp_src);
{
// temp = f32 >> 23
a.OpUShR(temp_dest, f32_src, dxbc::Src::LU(23));
// temp = 113 - (f32 >> 23)
a.OpIAdd(temp_dest, dxbc::Src::LI(113 - remap_bias), -temp_src);
// Don't allow the shift to overflow, since in DXBC the lower 5 bits of the
// shift amount are used (otherwise 0 becomes 8).
// temp = min(113 - (f32 >> 23), 24)
a.OpUMin(temp_dest, temp_src, dxbc::Src::LU(24));
// biased_f32 = (f32 & 0x7FFFFF) | 0x800000
a.OpBFI(f24_dest, dxbc::Src::LU(9), dxbc::Src::LU(23), dxbc::Src::LU(1),
f32_src);
// biased_f32 = ((f32 & 0x7FFFFF) | 0x800000) >> min(113 - (f32 >> 23), 24)
a.OpUShR(f24_dest, f24_src, temp_src);
}
// Not denormalized?
a.OpElse();
{
// Bias the exponent.
// biased_f32 = f32 + (-112 << 23)
// (left shift of a negative value is undefined behavior)
a.OpIAdd(f24_dest, f32_src,
dxbc::Src::LU(0xC8000000u + (remap_bias << 23)));
}
// Close the denormal check.
a.OpEndIf();
// Build the 20e4 number.
// temp = (biased_f32 >> 3) & 1
a.OpUBFE(temp_dest, dxbc::Src::LU(1), dxbc::Src::LU(3), f24_src);
// f24 = biased_f32 + 3
a.OpIAdd(f24_dest, f24_src, dxbc::Src::LU(3));
// f24 = biased_f32 + 3 + ((biased_f32 >> 3) & 1)
a.OpIAdd(f24_dest, f24_src, temp_src);
// f24 = ((biased_f32 + 3 + ((biased_f32 >> 3) & 1)) >> 3) & 0xFFFFFF
a.OpUBFE(f24_dest, dxbc::Src::LU(24), dxbc::Src::LU(3), f24_src);
}
void DxbcShaderTranslator::Depth20e4To32(
dxbc::Assembler& a, const dxbc::Dest& f32, uint32_t f24_temp,
uint32_t f24_temp_component, uint32_t f24_shift, uint32_t temp1_temp,
uint32_t temp1_temp_component, uint32_t temp2_temp,
uint32_t temp2_temp_component, bool remap_to_0_to_0_5) {
assert_true(f24_shift <= (32 - 24));
assert_true(temp1_temp != temp2_temp ||
temp1_temp_component != temp2_temp_component);
// Source may be the same as temp1 or temp2.
dxbc::Dest exponent_dest(
dxbc::Dest::R(temp1_temp, 1 << temp1_temp_component));
dxbc::Src exponent_src(dxbc::Src::R(temp1_temp).Select(temp1_temp_component));
dxbc::Dest mantissa_dest(
dxbc::Dest::R(temp2_temp, 1 << temp2_temp_component));
dxbc::Src mantissa_src(dxbc::Src::R(temp2_temp).Select(temp2_temp_component));
// CFloat24 from d3dref9.dll +
// https://github.com/Microsoft/DirectXTex/blob/master/DirectXTex/DirectXTexConvert.cpp
uint32_t remap_bias = uint32_t(remap_to_0_to_0_5);
if (!(f24_temp == temp1_temp && f24_temp_component == temp1_temp_component)) {
// Unpack the exponent before the mantissa if that doesn't overwrite the
// source.
a.OpUBFE(exponent_dest, dxbc::Src::LU(4), dxbc::Src::LU(f24_shift + 20),
dxbc::Src::R(f24_temp).Select(f24_temp_component));
}
// Unpack the mantissa.
a.OpUBFE(mantissa_dest, dxbc::Src::LU(20), dxbc::Src::LU(f24_shift),
dxbc::Src::R(f24_temp).Select(f24_temp_component));
if (f24_temp == temp1_temp && f24_temp_component == temp1_temp_component) {
// Unpack the exponent after the mantissa if doing that before the mantissa
// would overwrite the source.
a.OpUBFE(exponent_dest, dxbc::Src::LU(4), dxbc::Src::LU(f24_shift + 20),
dxbc::Src::R(f24_temp).Select(f24_temp_component));
}
// Check if the number is denormalized.
a.OpIf(false, exponent_src);
{
// Check if the number is non-zero (if the mantissa isn't zero - the
// exponent is known to be zero at this point).
a.OpIf(true, mantissa_src);
{
// Normalize the mantissa.
// Note that HLSL firstbithigh(x) is compiled to DXBC like:
// `x ? 31 - firstbit_hi(x) : -1`
// (returns the index from the LSB, not the MSB, but -1 for zero too).
// exponent = firstbit_hi(mantissa)
a.OpFirstBitHi(exponent_dest, mantissa_src);
// exponent = 20 - firstbithigh(mantissa)
// Or:
// exponent = 20 - (31 - firstbit_hi(mantissa))
a.OpIAdd(exponent_dest, exponent_src, dxbc::Src::LI(20 - 31));
// mantissa = mantissa << (20 - firstbithigh(mantissa))
// AND 0xFFFFF not needed after this - BFI will do it.
a.OpIShL(mantissa_dest, mantissa_src, exponent_src);
// Get the normalized exponent.
// exponent = 1 - (20 - firstbithigh(mantissa))
a.OpIAdd(exponent_dest, dxbc::Src::LI(1), -exponent_src);
}
// The number is zero.
a.OpElse();
{
// Set the unbiased exponent to -112 for zero - 112 will be added later
// (taking the range remap bias into account), resulting in zero float32.
a.OpMov(exponent_dest, dxbc::Src::LI(-int32_t(112 - remap_bias)));
}
// Close the non-zero check.
a.OpEndIf();
}
// Close the denormal check.
a.OpEndIf();
// Bias the exponent and move it to the correct location in f32, and also
// remap from guest 0...1 to host 0...0.5 if needed.
a.OpIMAd(exponent_dest, exponent_src, dxbc::Src::LI(1 << 23),
dxbc::Src::LI((112 - remap_bias) << 23));
// Combine the mantissa and the exponent.
a.OpBFI(f32, dxbc::Src::LU(20), dxbc::Src::LU(23 - 20), mantissa_src,
exponent_src);
}
void DxbcShaderTranslator::ROV_DepthTo24Bit(uint32_t d24_temp,
uint32_t d24_temp_component,
uint32_t d32_temp,
uint32_t d32_temp_component,
uint32_t temp_temp,
uint32_t temp_temp_component) {
assert_true(temp_temp != d32_temp ||
temp_temp_component != d32_temp_component);
// Source and destination may be the same.
a_.OpAnd(dxbc::Dest::R(temp_temp, 1 << temp_temp_component),
LoadFlagsSystemConstant(), dxbc::Src::LU(kSysFlag_DepthFloat24));
// Convert according to the format.
a_.OpIf(true, dxbc::Src::R(temp_temp).Select(temp_temp_component));
{
// 20e4 conversion.
PreClampedDepthTo20e4(a_, d24_temp, d24_temp_component, d32_temp,
d32_temp_component, temp_temp, temp_temp_component,
false);
}
a_.OpElse();
{
// Unorm24 conversion.
dxbc::Dest d24_dest(dxbc::Dest::R(d24_temp, 1 << d24_temp_component));
dxbc::Src d24_src(dxbc::Src::R(d24_temp).Select(d24_temp_component));
a_.OpMul(d24_dest, dxbc::Src::R(d32_temp).Select(d32_temp_component),
dxbc::Src::LF(float(0xFFFFFF)));
// Round to the nearest even integer. This seems to be the correct way:
// rounding towards zero gives 0xFF instead of 0x100 in clear shaders in,
// for instance, 4D5307E6, but other clear shaders in it are also broken if
// 0.5 is added before ftou instead of round_ne.
a_.OpRoundNE(d24_dest, d24_src);
// Convert to fixed-point.
a_.OpFToU(d24_dest, d24_src);
}
a_.OpEndIf();
}
} // namespace gpu
} // namespace xe