[GPU] Select promoted tfetch1D layouts at runtime

This keeps 545407D4's 2D use working while fixing wide 1D bindings in 555308CE.

Co-authored-by: Herman S. <429230+has207@users.noreply.github.com>
This commit is contained in:
goldislead
2026-08-13 20:40:16 -07:00
committed by Radosław Gliński
parent 95545f8e76
commit 7cd47947b0
4 changed files with 90 additions and 14 deletions

View File

@@ -885,7 +885,10 @@ void DxbcShaderTranslator::ProcessTextureFetchInstruction(
size_needed_components |= 0b0001;
break;
case xenos::FetchOpDimension::k2D:
if (instr.attributes.unnormalized_coordinates) {
// A tfetch1D promoted by its source swizzle may still use a 1D fetch
// constant. Its size interpretation is selected below at runtime.
if (instr.dimension == xenos::FetchOpDimension::k1D ||
instr.attributes.unnormalized_coordinates) {
size_needed_components |= 0b0011;
}
break;
@@ -937,6 +940,26 @@ void DxbcShaderTranslator::ProcessTextureFetchInstruction(
a_.OpUBFE(dxbc::Dest::R(size_and_is_3d_temp, size_needed_components),
dxbc::Src::LU(13, 13, 0, 0), dxbc::Src::LU(0, 13, 0, 0),
RequestTextureFetchConstantWord(tfetch_index, 2));
if (instr.dimension == xenos::FetchOpDimension::k1D) {
assert_true((size_needed_components & 0b0011) == 0b0011);
size_1d_width_minus_1_temp = PushSystemTemp();
a_.OpUBFE(dxbc::Dest::R(size_1d_width_minus_1_temp, 0b0001),
dxbc::Src::LU(xenos::kTexture1DMaxWidthLog2),
dxbc::Src::LU(0),
RequestTextureFetchConstantWord(tfetch_index, 2));
a_.OpMov(dxbc::Dest::R(size_1d_width_minus_1_temp, 0b0010),
dxbc::Src::LU(0));
a_.OpUBFE(dxbc::Dest::R(size_and_is_3d_temp, 0b1000),
dxbc::Src::LU(2), dxbc::Src::LU(9),
RequestTextureFetchConstantWord(tfetch_index, 5));
a_.OpIEq(dxbc::Dest::R(size_and_is_3d_temp, 0b1000),
dxbc::Src::R(size_and_is_3d_temp, dxbc::Src::kWWWW),
dxbc::Src::LU(uint32_t(xenos::DataDimension::k1D)));
a_.OpMovC(dxbc::Dest::R(size_and_is_3d_temp, 0b0011),
dxbc::Src::R(size_and_is_3d_temp, dxbc::Src::kWWWW),
dxbc::Src::R(size_1d_width_minus_1_temp),
dxbc::Src::R(size_and_is_3d_temp));
}
break;
case xenos::FetchOpDimension::k3DOrStacked:
// tfetch3D is used for both stacked and 3D - first, check if 3D.
@@ -1262,7 +1285,7 @@ void DxbcShaderTranslator::ProcessTextureFetchInstruction(
}
}
}
switch (coordinate_dimension) {
switch (instr.dimension) {
case xenos::FetchOpDimension::k1D: {
// Check if the fetch constant's actual dimension is k1D (word 5, bits
// 9-10). If not, skip wide 1D handling as size bits differ per
@@ -1315,6 +1338,9 @@ void DxbcShaderTranslator::ProcessTextureFetchInstruction(
dxbc::Src::LF(float(xenos::kTexture2DCubeMaxWidthHeight)));
a_.OpRoundPI(dxbc::Dest::R(coord_and_sampler_temp, 0b0100),
dxbc::Src::R(coord_and_sampler_temp, dxbc::Src::kZZZZ));
a_.OpMin(dxbc::Dest::R(coord_and_sampler_temp, 0b0100),
dxbc::Src::R(coord_and_sampler_temp, dxbc::Src::kZZZZ),
dxbc::Src::LF(float(xenos::kTexture1DWideMaxRows)));
// coord.y = (row_index + 0.5) / num_rows - sample at the center of
// the row, not its edge. At the edge, linear filtering would blend
// 50/50 with the previous row (texels 8192 apart), and even point
@@ -1333,15 +1359,23 @@ void DxbcShaderTranslator::ProcessTextureFetchInstruction(
a_.OpElse();
{
// Normal 1D texture - pad to 2D array coordinates.
a_.OpMov(dxbc::Dest::R(coord_and_sampler_temp, 0b0110),
dxbc::Src::LF(0.0f));
a_.OpMov(
dxbc::Dest::R(coord_and_sampler_temp,
coordinate_dimension == xenos::FetchOpDimension::k1D
? 0b0110
: 0b0100),
dxbc::Src::LF(0.0f));
}
a_.OpEndIf();
a_.OpElse();
{
// Non-1D texture bound to 1D fetch - just pad coordinates.
a_.OpMov(dxbc::Dest::R(coord_and_sampler_temp, 0b0110),
dxbc::Src::LF(0.0f));
// Keep Y when the source swizzle promoted the fetch to 2D.
a_.OpMov(
dxbc::Dest::R(coord_and_sampler_temp,
coordinate_dimension == xenos::FetchOpDimension::k1D
? 0b0110
: 0b0100),
dxbc::Src::LF(0.0f));
}
a_.OpEndIf();
} break;

View File

@@ -903,7 +903,10 @@ void SpirvShaderTranslator::ProcessTextureFetchInstruction(
size_needed_components |= 0b0001;
break;
case xenos::FetchOpDimension::k2D:
if (instr.attributes.unnormalized_coordinates) {
// A tfetch1D promoted by its source swizzle may still use a 1D fetch
// constant. Its size interpretation is selected below at runtime.
if (instr.dimension == xenos::FetchOpDimension::k1D ||
instr.attributes.unnormalized_coordinates) {
size_needed_components |= 0b0011;
}
break;
@@ -1002,6 +1005,38 @@ void SpirvShaderTranslator::ProcessTextureFetchInstruction(
spv::OpBitFieldUExtract, type_uint_, fetch_constant_word_2,
width_height_bit_count, width_height_bit_count);
}
if (instr.dimension == xenos::FetchOpDimension::k1D) {
assert_true((size_needed_components & 0b11) == 0b11);
id_vector_temp_.clear();
id_vector_temp_.push_back(const_int_0_);
id_vector_temp_.push_back(builder_->makeIntConstant(
int((fetch_constant_word_0_index + 5) >> 2)));
id_vector_temp_.push_back(builder_->makeIntConstant(
int((fetch_constant_word_0_index + 5) & 3)));
spv::Id fetch_constant_word_5 = builder_->createLoad(
builder_->createAccessChain(spv::StorageClassUniform,
uniform_fetch_constants_,
id_vector_temp_),
spv::NoPrecision);
spv::Id data_is_1d = builder_->createBinOp(
spv::OpIEqual, type_bool_,
builder_->createTriOp(spv::OpBitFieldUExtract, type_uint_,
fetch_constant_word_5,
builder_->makeUintConstant(9),
builder_->makeUintConstant(2)),
builder_->makeUintConstant(
static_cast<unsigned int>(xenos::DataDimension::k1D)));
spv::Id width_1d_minus_1 = builder_->createTriOp(
spv::OpBitFieldUExtract, type_uint_, fetch_constant_word_2,
const_uint_0_,
builder_->makeUintConstant(xenos::kTexture1DMaxWidthLog2));
size[0] =
builder_->createTriOp(spv::OpSelect, type_uint_, data_is_1d,
width_1d_minus_1, size[0]);
size[1] = builder_->createTriOp(spv::OpSelect, type_uint_,
data_is_1d, const_uint_0_, size[1]);
size_1d_width_minus_1_uint = width_1d_minus_1;
}
assert_zero(size_needed_components & 0b100);
} break;
case xenos::FetchOpDimension::k3DOrStacked: {
@@ -1341,12 +1376,15 @@ void SpirvShaderTranslator::ProcessTextureFetchInstruction(
spv::Id row_width_float = builder_->makeFloatConstant(
float(xenos::kTexture2DCubeMaxWidthHeight));
// num_rows = ceil(original_width / row_width)
spv::Id num_rows = builder_->createUnaryBuiltinCall(
type_float_, ext_inst_glsl_std_450_, GLSLstd450Ceil,
builder_->createNoContractionBinOp(spv::OpFDiv, type_float_,
original_width_float,
row_width_float));
// Keep this in sync with the texture cache's materialized row cap.
spv::Id num_rows = builder_->createBinBuiltinCall(
type_float_, ext_inst_glsl_std_450_, GLSLstd450NMin,
builder_->createUnaryBuiltinCall(
type_float_, ext_inst_glsl_std_450_, GLSLstd450Ceil,
builder_->createNoContractionBinOp(spv::OpFDiv, type_float_,
original_width_float,
row_width_float)),
builder_->makeFloatConstant(float(xenos::kTexture1DWideMaxRows)));
// linear_x = coord * original_width (denormalize to texel space)
spv::Id linear_x = builder_->createNoContractionBinOp(

View File

@@ -1013,6 +1013,7 @@ void TextureCache::BindingInfoFromFetchConstant(
uint32_t total_width = width_minus_1 + 1;
uint32_t row_width = xenos::kTexture2DCubeMaxWidthHeight;
uint32_t num_rows = (total_width + row_width - 1) / row_width;
num_rows = std::min(num_rows, xenos::kTexture1DWideMaxRows);
width_minus_1 = row_width - 1;
height_minus_1 = num_rows - 1;
// Disable mipmaps for wide 1D textures. The shader's coordinate remapping

View File

@@ -1185,6 +1185,9 @@ constexpr uint32_t kTextureSubresourceAlignmentBytes =
// Texture fetch constant size field widths.
constexpr uint32_t kTexture1DMaxWidthLog2 = 24;
constexpr uint32_t kTexture1DMaxWidth = 1 << kTexture1DMaxWidthLog2;
// Limit the number of rows materialized when wide 1D textures are mapped to
// 2D. Some games use very large widths with much less data behind them.
constexpr uint32_t kTexture1DWideMaxRows = 32;
constexpr uint32_t kTexture2DCubeMaxWidthHeightLog2 = 13;
constexpr uint32_t kTexture2DCubeMaxWidthHeight =
1 << kTexture2DCubeMaxWidthHeightLog2;