Files
Xenia-Canary/src/xenia/gpu/d3d12/texture_cache.cc

2806 lines
107 KiB
C++

/**
******************************************************************************
* Xenia : Xbox 360 Emulator Research Project *
******************************************************************************
* Copyright 2018 Ben Vanik. All rights reserved. *
* Released under the BSD license - see LICENSE in the root for more details. *
******************************************************************************
*/
#include "xenia/gpu/d3d12/texture_cache.h"
#include "third_party/xxhash/xxhash.h"
#include <algorithm>
#include <cstring>
#include "xenia/base/assert.h"
#include "xenia/base/clock.h"
#include "xenia/base/cvar.h"
#include "xenia/base/logging.h"
#include "xenia/base/math.h"
#include "xenia/base/profiling.h"
#include "xenia/gpu/d3d12/d3d12_command_processor.h"
#include "xenia/gpu/gpu_flags.h"
#include "xenia/gpu/texture_info.h"
#include "xenia/gpu/texture_util.h"
#include "xenia/ui/d3d12/d3d12_util.h"
#include "xenia/ui/d3d12/pools.h"
DEFINE_int32(d3d12_resolution_scale, 1,
"Scale of rendering width and height (currently only 1 and 2 "
"are available).",
"D3D12");
DEFINE_int32(d3d12_texture_cache_limit_soft, 384,
"Maximum host texture memory usage (in megabytes) above which old "
"textures will be destroyed (lifetime configured with "
"d3d12_texture_cache_limit_soft_lifetime). If using 2x resolution "
"scale, 1.25x of this is used.",
"D3D12");
DEFINE_int32(d3d12_texture_cache_limit_soft_lifetime, 30,
"Seconds a texture should be unused to be considered old enough "
"to be deleted if texture memory usage exceeds "
"d3d12_texture_cache_limit_soft.",
"D3D12");
DEFINE_int32(d3d12_texture_cache_limit_hard, 768,
"Maximum host texture memory usage (in megabytes) above which "
"textures will be destroyed as soon as possible. If using 2x "
"resolution scale, 1.25x of this is used.",
"D3D12");
namespace xe {
namespace gpu {
namespace d3d12 {
// Generated with `xb buildhlsl`.
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_128bpb_2x_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_128bpb_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_16bpb_2x_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_16bpb_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_32bpb_2x_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_32bpb_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_64bpb_2x_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_64bpb_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_8bpb_2x_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_8bpb_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_ctx1_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_depth_float_2x_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_depth_float_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_depth_unorm_2x_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_depth_unorm_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_dxn_rg8_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_dxt1_rgba8_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_dxt3_rgba8_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_dxt3a_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_dxt3aas1111_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_dxt5_rgba8_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_dxt5a_r8_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_r10g11b11_rgba16_2x_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_r10g11b11_rgba16_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_r10g11b11_rgba16_snorm_2x_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_r10g11b11_rgba16_snorm_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_r11g11b10_rgba16_2x_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_r11g11b10_rgba16_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_r11g11b10_rgba16_snorm_2x_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_load_r11g11b10_rgba16_snorm_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_tile_128bpp_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_tile_16bpp_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_tile_16bpp_rgba_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_tile_32bpp_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_tile_64bpp_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_tile_8bpp_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_tile_r10g11b11_rgba16_cs.h"
#include "xenia/gpu/d3d12/shaders/dxbc/texture_tile_r11g11b10_rgba16_cs.h"
constexpr uint32_t TextureCache::Texture::kCachedSRVDescriptorSwizzleMissing;
constexpr uint32_t TextureCache::SRVDescriptorCachePage::kHeapSize;
constexpr uint32_t TextureCache::LoadConstants::kGuestPitchTiled;
constexpr uint32_t TextureCache::kScaledResolveBufferSizeLog2;
constexpr uint32_t TextureCache::kScaledResolveBufferSize;
constexpr uint32_t TextureCache::kScaledResolveHeapSizeLog2;
constexpr uint32_t TextureCache::kScaledResolveHeapSize;
// For formats with less than 4 components, assuming the last component is
// replicated into the non-existent ones, similar to what is done for unused
// components of operands in shaders.
// For DXT3A and DXT5A, RRRR swizzle is specified in:
// http://fileadmin.cs.lth.se/cs/Personal/Michael_Doggett/talks/unc-xenos-doggett.pdf
// Halo 3 also expects replicated components in k_8 sprites.
// DXN is read as RG in Halo 3, but as RA in Call of Duty.
// TODO(Triang3l): Find out the correct contents of unused texture components.
const TextureCache::HostFormat TextureCache::host_formats_[64] = {
// k_1_REVERSE
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_1
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_8
{DXGI_FORMAT_R8_TYPELESS,
DXGI_FORMAT_R8_UNORM,
LoadMode::k8bpb,
DXGI_FORMAT_R8_SNORM,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8_UNORM,
ResolveTileMode::k8bpp,
{0, 0, 0, 0}},
// k_1_5_5_5
// Red and blue swapped in the load shader for simplicity.
{DXGI_FORMAT_B5G5R5A1_UNORM,
DXGI_FORMAT_B5G5R5A1_UNORM,
LoadMode::k16bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8G8B8A8_UNORM,
ResolveTileMode::k16bppRGBA,
{0, 1, 2, 3}},
// k_5_6_5
// Red and blue swapped in the load shader for simplicity.
{DXGI_FORMAT_B5G6R5_UNORM,
DXGI_FORMAT_B5G6R5_UNORM,
LoadMode::k16bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_B5G6R5_UNORM,
ResolveTileMode::k16bpp,
{0, 1, 2, 2}},
// k_6_5_5
// On the host, green bits in blue, blue bits in green.
{DXGI_FORMAT_B5G6R5_UNORM,
DXGI_FORMAT_B5G6R5_UNORM,
LoadMode::k16bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_B5G6R5_UNORM,
ResolveTileMode::k16bpp,
{0, 2, 1, 1}},
// k_8_8_8_8
{DXGI_FORMAT_R8G8B8A8_TYPELESS,
DXGI_FORMAT_R8G8B8A8_UNORM,
LoadMode::k32bpb,
DXGI_FORMAT_R8G8B8A8_SNORM,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8G8B8A8_UNORM,
ResolveTileMode::k32bpp,
{0, 1, 2, 3}},
// k_2_10_10_10
{DXGI_FORMAT_R10G10B10A2_TYPELESS,
DXGI_FORMAT_R10G10B10A2_UNORM,
LoadMode::k32bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R10G10B10A2_UNORM,
ResolveTileMode::k32bpp,
{0, 1, 2, 3}},
// k_8_A
{DXGI_FORMAT_R8_TYPELESS,
DXGI_FORMAT_R8_UNORM,
LoadMode::k8bpb,
DXGI_FORMAT_R8_SNORM,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8_UNORM,
ResolveTileMode::k8bpp,
{0, 0, 0, 0}},
// k_8_B
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_8_8
{DXGI_FORMAT_R8G8_TYPELESS,
DXGI_FORMAT_R8G8_UNORM,
LoadMode::k16bpb,
DXGI_FORMAT_R8G8_SNORM,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8G8_UNORM,
ResolveTileMode::k16bpp,
{0, 1, 1, 1}},
// k_Cr_Y1_Cb_Y0_REP
// Red and blue probably must be swapped, similar to k_Y1_Cr_Y0_Cb_REP.
{DXGI_FORMAT_G8R8_G8B8_UNORM,
DXGI_FORMAT_G8R8_G8B8_UNORM,
LoadMode::k32bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{2, 1, 0, 3}},
// k_Y1_Cr_Y0_Cb_REP
// Used for videos in NBA 2K9. Red and blue must be swapped.
// TODO(Triang3l): D3DFMT_G8R8_G8B8 is DXGI_FORMAT_R8G8_B8G8_UNORM * 255.0f,
// watch out for num_format int, division in shaders, etc., in NBA 2K9 it
// works as is. Also need to decompress if the size is uneven, but should be
// a very rare case.
{DXGI_FORMAT_R8G8_B8G8_UNORM,
DXGI_FORMAT_R8G8_B8G8_UNORM,
LoadMode::k32bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{2, 1, 0, 3}},
// k_16_16_EDRAM
// Not usable as a texture, also has -32...32 range.
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 1, 1}},
// k_8_8_8_8_A
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 3}},
// k_4_4_4_4
// Red and blue swapped in the load shader for simplicity.
{DXGI_FORMAT_B4G4R4A4_UNORM,
DXGI_FORMAT_B4G4R4A4_UNORM,
LoadMode::k16bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8G8B8A8_UNORM,
ResolveTileMode::k16bppRGBA,
{0, 1, 2, 3}},
// k_10_11_11
{DXGI_FORMAT_R16G16B16A16_TYPELESS,
DXGI_FORMAT_R16G16B16A16_UNORM,
LoadMode::kR11G11B10ToRGBA16,
DXGI_FORMAT_R16G16B16A16_SNORM,
LoadMode::kR11G11B10ToRGBA16SNorm,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16G16B16A16_UNORM,
ResolveTileMode::kR11G11B10AsRGBA16,
{0, 1, 2, 2}},
// k_11_11_10
{DXGI_FORMAT_R16G16B16A16_TYPELESS,
DXGI_FORMAT_R16G16B16A16_UNORM,
LoadMode::kR10G11B11ToRGBA16,
DXGI_FORMAT_R16G16B16A16_SNORM,
LoadMode::kR10G11B11ToRGBA16SNorm,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16G16B16A16_UNORM,
ResolveTileMode::kR10G11B11AsRGBA16,
{0, 1, 2, 2}},
// k_DXT1
{DXGI_FORMAT_BC1_UNORM,
DXGI_FORMAT_BC1_UNORM,
LoadMode::k64bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8G8B8A8_UNORM,
LoadMode::kDXT1ToRGBA8,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 3}},
// k_DXT2_3
{DXGI_FORMAT_BC2_UNORM,
DXGI_FORMAT_BC2_UNORM,
LoadMode::k128bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8G8B8A8_UNORM,
LoadMode::kDXT3ToRGBA8,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 3}},
// k_DXT4_5
{DXGI_FORMAT_BC3_UNORM,
DXGI_FORMAT_BC3_UNORM,
LoadMode::k128bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8G8B8A8_UNORM,
LoadMode::kDXT5ToRGBA8,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 3}},
// k_16_16_16_16_EDRAM
// Not usable as a texture, also has -32...32 range.
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 3}},
// R32_FLOAT for depth because shaders would require an additional SRV to
// sample stencil, which we don't provide.
// k_24_8
{DXGI_FORMAT_R32_FLOAT,
DXGI_FORMAT_R32_FLOAT,
LoadMode::kDepthUnorm,
DXGI_FORMAT_R32_FLOAT,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_24_8_FLOAT
{DXGI_FORMAT_R32_FLOAT,
DXGI_FORMAT_R32_FLOAT,
LoadMode::kDepthFloat,
DXGI_FORMAT_R32_FLOAT,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_16
{DXGI_FORMAT_R16_TYPELESS,
DXGI_FORMAT_R16_UNORM,
LoadMode::k16bpb,
DXGI_FORMAT_R16_SNORM,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16_UNORM,
ResolveTileMode::k16bpp,
{0, 0, 0, 0}},
// k_16_16
// The resolve format being unorm is correct (with snorm distortion effects
// in Halo 3 cause stretching of one corner of the screen).
{DXGI_FORMAT_R16G16_TYPELESS,
DXGI_FORMAT_R16G16_UNORM,
LoadMode::k32bpb,
DXGI_FORMAT_R16G16_SNORM,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16G16_UNORM,
ResolveTileMode::k32bpp,
{0, 1, 1, 1}},
// k_16_16_16_16
// The resolve format being unorm is correct (with snorm distortion effects
// in Halo 3 cause stretching of one corner of the screen).
{DXGI_FORMAT_R16G16B16A16_TYPELESS,
DXGI_FORMAT_R16G16B16A16_UNORM,
LoadMode::k64bpb,
DXGI_FORMAT_R16G16B16A16_SNORM,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16G16B16A16_UNORM,
ResolveTileMode::k64bpp,
{0, 1, 2, 3}},
// k_16_EXPAND
{DXGI_FORMAT_R16_FLOAT,
DXGI_FORMAT_R16_FLOAT,
LoadMode::k16bpb,
DXGI_FORMAT_R16_FLOAT,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16_FLOAT,
ResolveTileMode::k16bpp,
{0, 0, 0, 0}},
// k_16_16_EXPAND
{DXGI_FORMAT_R16G16_FLOAT,
DXGI_FORMAT_R16G16_FLOAT,
LoadMode::k32bpb,
DXGI_FORMAT_R16G16_FLOAT,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16G16_FLOAT,
ResolveTileMode::k32bpp,
{0, 1, 1, 1}},
// k_16_16_16_16_EXPAND
{DXGI_FORMAT_R16G16B16A16_FLOAT,
DXGI_FORMAT_R16G16B16A16_FLOAT,
LoadMode::k64bpb,
DXGI_FORMAT_R16G16B16A16_FLOAT,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16G16B16A16_FLOAT,
ResolveTileMode::k64bpp,
{0, 1, 2, 3}},
// k_16_FLOAT
{DXGI_FORMAT_R16_FLOAT,
DXGI_FORMAT_R16_FLOAT,
LoadMode::k16bpb,
DXGI_FORMAT_R16_FLOAT,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16_FLOAT,
ResolveTileMode::k16bpp,
{0, 0, 0, 0}},
// k_16_16_FLOAT
{DXGI_FORMAT_R16G16_FLOAT,
DXGI_FORMAT_R16G16_FLOAT,
LoadMode::k32bpb,
DXGI_FORMAT_R16G16_FLOAT,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16G16_FLOAT,
ResolveTileMode::k32bpp,
{0, 1, 1, 1}},
// k_16_16_16_16_FLOAT
{DXGI_FORMAT_R16G16B16A16_FLOAT,
DXGI_FORMAT_R16G16B16A16_FLOAT,
LoadMode::k64bpb,
DXGI_FORMAT_R16G16B16A16_FLOAT,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16G16B16A16_FLOAT,
ResolveTileMode::k64bpp,
{0, 1, 2, 3}},
// k_32
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_32_32
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 1, 1}},
// k_32_32_32_32
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 3}},
// k_32_FLOAT
{DXGI_FORMAT_R32_FLOAT,
DXGI_FORMAT_R32_FLOAT,
LoadMode::k32bpb,
DXGI_FORMAT_R32_FLOAT,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R32_FLOAT,
ResolveTileMode::k32bpp,
{0, 0, 0, 0}},
// k_32_32_FLOAT
{DXGI_FORMAT_R32G32_FLOAT,
DXGI_FORMAT_R32G32_FLOAT,
LoadMode::k64bpb,
DXGI_FORMAT_R32G32_FLOAT,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R32G32_FLOAT,
ResolveTileMode::k64bpp,
{0, 1, 1, 1}},
// k_32_32_32_32_FLOAT
{DXGI_FORMAT_R32G32B32A32_FLOAT,
DXGI_FORMAT_R32G32B32A32_FLOAT,
LoadMode::k128bpb,
DXGI_FORMAT_R32G32B32A32_FLOAT,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R32G32B32A32_FLOAT,
ResolveTileMode::k128bpp,
{0, 1, 2, 3}},
// k_32_AS_8
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_32_AS_8_8
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 1, 1}},
// k_16_MPEG
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_16_16_MPEG
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 1, 1}},
// k_8_INTERLACED
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_32_AS_8_INTERLACED
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_32_AS_8_8_INTERLACED
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 1, 1}},
// k_16_INTERLACED
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_16_MPEG_INTERLACED
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_16_16_MPEG_INTERLACED
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 1, 1}},
// k_DXN
{DXGI_FORMAT_BC5_UNORM,
DXGI_FORMAT_BC5_UNORM,
LoadMode::k128bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8G8_UNORM,
LoadMode::kDXNToRG8,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 1, 1}},
// k_8_8_8_8_AS_16_16_16_16
{DXGI_FORMAT_R8G8B8A8_TYPELESS,
DXGI_FORMAT_R8G8B8A8_UNORM,
LoadMode::k32bpb,
DXGI_FORMAT_R8G8B8A8_SNORM,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8G8B8A8_UNORM,
ResolveTileMode::k32bpp,
{0, 1, 2, 3}},
// k_DXT1_AS_16_16_16_16
{DXGI_FORMAT_BC1_UNORM,
DXGI_FORMAT_BC1_UNORM,
LoadMode::k64bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8G8B8A8_UNORM,
LoadMode::kDXT1ToRGBA8,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 3}},
// k_DXT2_3_AS_16_16_16_16
{DXGI_FORMAT_BC2_UNORM,
DXGI_FORMAT_BC2_UNORM,
LoadMode::k128bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8G8B8A8_UNORM,
LoadMode::kDXT3ToRGBA8,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 3}},
// k_DXT4_5_AS_16_16_16_16
{DXGI_FORMAT_BC3_UNORM,
DXGI_FORMAT_BC3_UNORM,
LoadMode::k128bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8G8B8A8_UNORM,
LoadMode::kDXT5ToRGBA8,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 3}},
// k_2_10_10_10_AS_16_16_16_16
{DXGI_FORMAT_R10G10B10A2_UNORM,
DXGI_FORMAT_R10G10B10A2_UNORM,
LoadMode::k32bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R10G10B10A2_UNORM,
ResolveTileMode::k32bpp,
{0, 1, 2, 3}},
// k_10_11_11_AS_16_16_16_16
{DXGI_FORMAT_R16G16B16A16_TYPELESS,
DXGI_FORMAT_R16G16B16A16_UNORM,
LoadMode::kR11G11B10ToRGBA16,
DXGI_FORMAT_R16G16B16A16_SNORM,
LoadMode::kR11G11B10ToRGBA16SNorm,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16G16B16A16_UNORM,
ResolveTileMode::kR11G11B10AsRGBA16,
{0, 1, 2, 2}},
// k_11_11_10_AS_16_16_16_16
{DXGI_FORMAT_R16G16B16A16_TYPELESS,
DXGI_FORMAT_R16G16B16A16_UNORM,
LoadMode::kR10G11B11ToRGBA16,
DXGI_FORMAT_R16G16B16A16_SNORM,
LoadMode::kR10G11B11ToRGBA16SNorm,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R16G16B16A16_UNORM,
ResolveTileMode::kR10G11B11AsRGBA16,
{0, 1, 2, 2}},
// k_32_32_32_FLOAT
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 2}},
// k_DXT3A
// R8_UNORM has the same size as BC2, but doesn't have the 4x4 size
// alignment requirement.
{DXGI_FORMAT_R8_UNORM,
DXGI_FORMAT_R8_UNORM,
LoadMode::kDXT3A,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_DXT5A
{DXGI_FORMAT_BC4_UNORM,
DXGI_FORMAT_BC4_UNORM,
LoadMode::k64bpb,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_R8_UNORM,
LoadMode::kDXT5AToR8,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 0, 0, 0}},
// k_CTX1
{DXGI_FORMAT_R8G8_UNORM,
DXGI_FORMAT_R8G8_UNORM,
LoadMode::kCTX1,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 1, 1}},
// k_DXT3A_AS_1_1_1_1
{DXGI_FORMAT_B4G4R4A4_UNORM,
DXGI_FORMAT_B4G4R4A4_UNORM,
LoadMode::kDXT3AAs1111,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 3}},
// k_8_8_8_8_GAMMA_EDRAM
// Not usable as a texture.
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 3}},
// k_2_10_10_10_FLOAT_EDRAM
// Not usable as a texture.
{DXGI_FORMAT_UNKNOWN,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
LoadMode::kUnknown,
DXGI_FORMAT_UNKNOWN,
ResolveTileMode::kUnknown,
{0, 1, 2, 3}},
};
const char* const TextureCache::dimension_names_[4] = {"1D", "2D", "3D",
"cube"};
const TextureCache::LoadModeInfo TextureCache::load_mode_info_[] = {
{texture_load_8bpb_cs, sizeof(texture_load_8bpb_cs),
texture_load_8bpb_2x_cs, sizeof(texture_load_8bpb_2x_cs)},
{texture_load_16bpb_cs, sizeof(texture_load_16bpb_cs),
texture_load_16bpb_2x_cs, sizeof(texture_load_16bpb_2x_cs)},
{texture_load_32bpb_cs, sizeof(texture_load_32bpb_cs),
texture_load_32bpb_2x_cs, sizeof(texture_load_32bpb_2x_cs)},
{texture_load_64bpb_cs, sizeof(texture_load_64bpb_cs),
texture_load_64bpb_2x_cs, sizeof(texture_load_64bpb_2x_cs)},
{texture_load_128bpb_cs, sizeof(texture_load_128bpb_cs),
texture_load_128bpb_2x_cs, sizeof(texture_load_128bpb_2x_cs)},
{texture_load_r11g11b10_rgba16_cs, sizeof(texture_load_r11g11b10_rgba16_cs),
texture_load_r11g11b10_rgba16_2x_cs,
sizeof(texture_load_r11g11b10_rgba16_2x_cs)},
{texture_load_r11g11b10_rgba16_snorm_cs,
sizeof(texture_load_r11g11b10_rgba16_snorm_cs),
texture_load_r11g11b10_rgba16_snorm_2x_cs,
sizeof(texture_load_r11g11b10_rgba16_snorm_2x_cs)},
{texture_load_r10g11b11_rgba16_cs, sizeof(texture_load_r10g11b11_rgba16_cs),
texture_load_r10g11b11_rgba16_2x_cs,
sizeof(texture_load_r10g11b11_rgba16_2x_cs)},
{texture_load_r10g11b11_rgba16_snorm_cs,
sizeof(texture_load_r10g11b11_rgba16_snorm_cs),
texture_load_r10g11b11_rgba16_snorm_2x_cs,
sizeof(texture_load_r10g11b11_rgba16_snorm_2x_cs)},
{texture_load_dxt1_rgba8_cs, sizeof(texture_load_dxt1_rgba8_cs), nullptr,
0},
{texture_load_dxt3_rgba8_cs, sizeof(texture_load_dxt3_rgba8_cs), nullptr,
0},
{texture_load_dxt5_rgba8_cs, sizeof(texture_load_dxt5_rgba8_cs), nullptr,
0},
{texture_load_dxn_rg8_cs, sizeof(texture_load_dxn_rg8_cs), nullptr, 0},
{texture_load_dxt3a_cs, sizeof(texture_load_dxt3a_cs), nullptr, 0},
{texture_load_dxt3aas1111_cs, sizeof(texture_load_dxt3aas1111_cs), nullptr,
0},
{texture_load_dxt5a_r8_cs, sizeof(texture_load_dxt5a_r8_cs), nullptr, 0},
{texture_load_ctx1_cs, sizeof(texture_load_ctx1_cs), nullptr, 0},
{texture_load_depth_unorm_cs, sizeof(texture_load_depth_unorm_cs),
texture_load_depth_unorm_2x_cs, sizeof(texture_load_depth_unorm_2x_cs)},
{texture_load_depth_float_cs, sizeof(texture_load_depth_float_cs),
texture_load_depth_float_2x_cs, sizeof(texture_load_depth_float_2x_cs)},
};
const TextureCache::ResolveTileModeInfo
TextureCache::resolve_tile_mode_info_[] = {
{texture_tile_8bpp_cs, sizeof(texture_tile_8bpp_cs),
DXGI_FORMAT_R8_UINT, 0},
{texture_tile_16bpp_cs, sizeof(texture_tile_16bpp_cs),
DXGI_FORMAT_R16_UINT, 1},
{texture_tile_32bpp_cs, sizeof(texture_tile_32bpp_cs),
DXGI_FORMAT_UNKNOWN, 0},
{texture_tile_64bpp_cs, sizeof(texture_tile_64bpp_cs),
DXGI_FORMAT_UNKNOWN, 0},
{texture_tile_128bpp_cs, sizeof(texture_tile_128bpp_cs),
DXGI_FORMAT_UNKNOWN, 0},
{texture_tile_16bpp_rgba_cs, sizeof(texture_tile_16bpp_rgba_cs),
DXGI_FORMAT_R16_UINT, 1},
{texture_tile_r11g11b10_rgba16_cs,
sizeof(texture_tile_r11g11b10_rgba16_cs), DXGI_FORMAT_UNKNOWN, 0},
{texture_tile_r10g11b11_rgba16_cs,
sizeof(texture_tile_r10g11b11_rgba16_cs), DXGI_FORMAT_UNKNOWN, 0},
};
TextureCache::TextureCache(D3D12CommandProcessor* command_processor,
RegisterFile* register_file,
SharedMemory* shared_memory)
: command_processor_(command_processor),
register_file_(register_file),
shared_memory_(shared_memory) {}
TextureCache::~TextureCache() { Shutdown(); }
bool TextureCache::Initialize() {
auto provider = command_processor_->GetD3D12Context()->GetD3D12Provider();
auto device = provider->GetDevice();
// Try to create the tiled buffer 2x resolution scaling.
// Not currently supported with the RTV/DSV output path for various reasons.
// As of November 27th, 2018, PIX doesn't support tiled buffers.
if (cvars::d3d12_resolution_scale >= 2 &&
command_processor_->IsROVUsedForEDRAM() &&
provider->GetTiledResourcesTier() >= 1 &&
provider->GetGraphicsAnalysis() == nullptr &&
provider->GetVirtualAddressBitsPerResource() >=
kScaledResolveBufferSizeLog2) {
D3D12_RESOURCE_DESC scaled_resolve_buffer_desc;
ui::d3d12::util::FillBufferResourceDesc(
scaled_resolve_buffer_desc, kScaledResolveBufferSize,
D3D12_RESOURCE_FLAG_ALLOW_UNORDERED_ACCESS);
scaled_resolve_buffer_state_ = D3D12_RESOURCE_STATE_UNORDERED_ACCESS;
if (FAILED(device->CreateReservedResource(
&scaled_resolve_buffer_desc, scaled_resolve_buffer_state_, nullptr,
IID_PPV_ARGS(&scaled_resolve_buffer_)))) {
XELOGE(
"Texture cache: Failed to create the 2 GB tiled buffer for 2x "
"resolution scale - switching to 1x");
}
const uint32_t scaled_resolve_page_dword_count =
(512 * 1024 * 1024) / 4096 / 32;
scaled_resolve_pages_ = new uint32_t[scaled_resolve_page_dword_count];
std::memset(scaled_resolve_pages_, 0,
scaled_resolve_page_dword_count * sizeof(uint32_t));
std::memset(scaled_resolve_pages_l2_, 0, sizeof(scaled_resolve_pages_l2_));
}
std::memset(scaled_resolve_heaps_, 0, sizeof(scaled_resolve_heaps_));
scaled_resolve_heap_count_ = 0;
// Create the loading root signature.
D3D12_ROOT_PARAMETER root_parameters[2];
// Parameter 0 is constants (changed very often when untiling).
root_parameters[0].ParameterType = D3D12_ROOT_PARAMETER_TYPE_CBV;
root_parameters[0].Descriptor.ShaderRegister = 0;
root_parameters[0].Descriptor.RegisterSpace = 0;
root_parameters[0].ShaderVisibility = D3D12_SHADER_VISIBILITY_ALL;
// Parameter 1 is source and target.
D3D12_DESCRIPTOR_RANGE root_copy_ranges[2];
root_copy_ranges[0].RangeType = D3D12_DESCRIPTOR_RANGE_TYPE_SRV;
root_copy_ranges[0].NumDescriptors = 1;
root_copy_ranges[0].BaseShaderRegister = 0;
root_copy_ranges[0].RegisterSpace = 0;
root_copy_ranges[0].OffsetInDescriptorsFromTableStart = 0;
root_copy_ranges[1].RangeType = D3D12_DESCRIPTOR_RANGE_TYPE_UAV;
root_copy_ranges[1].NumDescriptors = 1;
root_copy_ranges[1].BaseShaderRegister = 0;
root_copy_ranges[1].RegisterSpace = 0;
root_copy_ranges[1].OffsetInDescriptorsFromTableStart = 1;
root_parameters[1].ParameterType = D3D12_ROOT_PARAMETER_TYPE_DESCRIPTOR_TABLE;
root_parameters[1].DescriptorTable.NumDescriptorRanges = 2;
root_parameters[1].DescriptorTable.pDescriptorRanges = root_copy_ranges;
root_parameters[1].ShaderVisibility = D3D12_SHADER_VISIBILITY_ALL;
D3D12_ROOT_SIGNATURE_DESC root_signature_desc;
root_signature_desc.NumParameters = UINT(xe::countof(root_parameters));
root_signature_desc.pParameters = root_parameters;
root_signature_desc.NumStaticSamplers = 0;
root_signature_desc.pStaticSamplers = nullptr;
root_signature_desc.Flags = D3D12_ROOT_SIGNATURE_FLAG_NONE;
load_root_signature_ =
ui::d3d12::util::CreateRootSignature(provider, root_signature_desc);
if (load_root_signature_ == nullptr) {
XELOGE("Failed to create the texture loading root signature");
Shutdown();
return false;
}
// Create the tiling root signature (almost the same, but with root constants
// in parameter 0).
root_parameters[0].ParameterType = D3D12_ROOT_PARAMETER_TYPE_32BIT_CONSTANTS;
root_parameters[0].Constants.ShaderRegister = 0;
root_parameters[0].Constants.RegisterSpace = 0;
root_parameters[0].Constants.Num32BitValues =
sizeof(ResolveTileConstants) / sizeof(uint32_t);
resolve_tile_root_signature_ =
ui::d3d12::util::CreateRootSignature(provider, root_signature_desc);
if (resolve_tile_root_signature_ == nullptr) {
XELOGE("Failed to create the texture tiling root signature");
Shutdown();
return false;
}
// Create the loading and tiling pipelines.
for (uint32_t i = 0; i < uint32_t(LoadMode::kCount); ++i) {
const LoadModeInfo& mode_info = load_mode_info_[i];
load_pipelines_[i] = ui::d3d12::util::CreateComputePipeline(
device, mode_info.shader, mode_info.shader_size, load_root_signature_);
if (load_pipelines_[i] == nullptr) {
XELOGE("Failed to create the texture loading pipeline for mode %u", i);
Shutdown();
return false;
}
if (IsResolutionScale2X() && mode_info.shader_2x != nullptr) {
load_pipelines_2x_[i] = ui::d3d12::util::CreateComputePipeline(
device, mode_info.shader_2x, mode_info.shader_2x_size,
load_root_signature_);
if (load_pipelines_2x_[i] == nullptr) {
XELOGE(
"Failed to create the 2x-scaled texture loading pipeline for mode "
"%u",
i);
Shutdown();
return false;
}
}
}
for (uint32_t i = 0; i < uint32_t(ResolveTileMode::kCount); ++i) {
const ResolveTileModeInfo& mode_info = resolve_tile_mode_info_[i];
resolve_tile_pipelines_[i] = ui::d3d12::util::CreateComputePipeline(
device, mode_info.shader, mode_info.shader_size,
resolve_tile_root_signature_);
if (resolve_tile_pipelines_[i] == nullptr) {
XELOGE("Failed to create the texture tiling pipeline for mode %u", i);
Shutdown();
return false;
}
}
// Create a heap with null SRV descriptors, since it's faster to copy a
// descriptor than to create an SRV, and null descriptors are used a lot (for
// the signed version when only unsigned is used, for instance).
D3D12_DESCRIPTOR_HEAP_DESC null_srv_descriptor_heap_desc;
null_srv_descriptor_heap_desc.Type = D3D12_DESCRIPTOR_HEAP_TYPE_CBV_SRV_UAV;
null_srv_descriptor_heap_desc.NumDescriptors =
uint32_t(NullSRVDescriptorIndex::kCount);
null_srv_descriptor_heap_desc.Flags = D3D12_DESCRIPTOR_HEAP_FLAG_NONE;
null_srv_descriptor_heap_desc.NodeMask = 0;
if (FAILED(device->CreateDescriptorHeap(
&null_srv_descriptor_heap_desc,
IID_PPV_ARGS(&null_srv_descriptor_heap_)))) {
XELOGE("Failed to create the descriptor heap for null SRVs");
Shutdown();
return false;
}
null_srv_descriptor_heap_start_ =
null_srv_descriptor_heap_->GetCPUDescriptorHandleForHeapStart();
D3D12_SHADER_RESOURCE_VIEW_DESC null_srv_desc;
null_srv_desc.Format = DXGI_FORMAT_R8G8B8A8_UNORM;
null_srv_desc.Shader4ComponentMapping =
D3D12_ENCODE_SHADER_4_COMPONENT_MAPPING(
D3D12_SHADER_COMPONENT_MAPPING_FORCE_VALUE_0,
D3D12_SHADER_COMPONENT_MAPPING_FORCE_VALUE_0,
D3D12_SHADER_COMPONENT_MAPPING_FORCE_VALUE_0,
D3D12_SHADER_COMPONENT_MAPPING_FORCE_VALUE_0);
null_srv_desc.ViewDimension = D3D12_SRV_DIMENSION_TEXTURE2DARRAY;
null_srv_desc.Texture2DArray.MostDetailedMip = 0;
null_srv_desc.Texture2DArray.MipLevels = 1;
null_srv_desc.Texture2DArray.FirstArraySlice = 0;
null_srv_desc.Texture2DArray.ArraySize = 1;
null_srv_desc.Texture2DArray.PlaneSlice = 0;
null_srv_desc.Texture2DArray.ResourceMinLODClamp = 0.0f;
device->CreateShaderResourceView(
nullptr, &null_srv_desc,
provider->OffsetViewDescriptor(
null_srv_descriptor_heap_start_,
uint32_t(NullSRVDescriptorIndex::k2DArray)));
null_srv_desc.ViewDimension = D3D12_SRV_DIMENSION_TEXTURE3D;
null_srv_desc.Texture3D.MostDetailedMip = 0;
null_srv_desc.Texture3D.MipLevels = 1;
null_srv_desc.Texture3D.ResourceMinLODClamp = 0.0f;
device->CreateShaderResourceView(
nullptr, &null_srv_desc,
provider->OffsetViewDescriptor(null_srv_descriptor_heap_start_,
uint32_t(NullSRVDescriptorIndex::k3D)));
null_srv_desc.ViewDimension = D3D12_SRV_DIMENSION_TEXTURECUBE;
null_srv_desc.TextureCube.MostDetailedMip = 0;
null_srv_desc.TextureCube.MipLevels = 1;
null_srv_desc.TextureCube.ResourceMinLODClamp = 0.0f;
device->CreateShaderResourceView(
nullptr, &null_srv_desc,
provider->OffsetViewDescriptor(null_srv_descriptor_heap_start_,
uint32_t(NullSRVDescriptorIndex::kCube)));
if (IsResolutionScale2X()) {
scaled_resolve_global_watch_handle_ = shared_memory_->RegisterGlobalWatch(
ScaledResolveGlobalWatchCallbackThunk, this);
}
texture_current_usage_time_ = xe::Clock::QueryHostUptimeMillis();
return true;
}
void TextureCache::Shutdown() {
ClearCache();
if (scaled_resolve_global_watch_handle_ != nullptr) {
shared_memory_->UnregisterGlobalWatch(scaled_resolve_global_watch_handle_);
scaled_resolve_global_watch_handle_ = nullptr;
}
ui::d3d12::util::ReleaseAndNull(null_srv_descriptor_heap_);
for (uint32_t i = 0; i < uint32_t(ResolveTileMode::kCount); ++i) {
ui::d3d12::util::ReleaseAndNull(resolve_tile_pipelines_[i]);
}
ui::d3d12::util::ReleaseAndNull(resolve_tile_root_signature_);
for (uint32_t i = 0; i < uint32_t(LoadMode::kCount); ++i) {
ui::d3d12::util::ReleaseAndNull(load_pipelines_2x_[i]);
ui::d3d12::util::ReleaseAndNull(load_pipelines_[i]);
}
ui::d3d12::util::ReleaseAndNull(load_root_signature_);
if (scaled_resolve_pages_ != nullptr) {
delete[] scaled_resolve_pages_;
scaled_resolve_pages_ = nullptr;
}
// First free the buffer to detach it from the heaps.
ui::d3d12::util::ReleaseAndNull(scaled_resolve_buffer_);
for (uint32_t i = 0; i < xe::countof(scaled_resolve_heaps_); ++i) {
ui::d3d12::util::ReleaseAndNull(scaled_resolve_heaps_[i]);
}
scaled_resolve_heap_count_ = 0;
COUNT_profile_set("gpu/texture_cache/scaled_resolve_buffer_used_mb", 0);
}
void TextureCache::ClearCache() {
// Destroy all the textures.
for (auto texture_pair : textures_) {
Texture* texture = texture_pair.second;
shared_memory_->UnwatchMemoryRange(texture->base_watch_handle);
shared_memory_->UnwatchMemoryRange(texture->mip_watch_handle);
texture->resource->Release();
delete texture;
}
textures_.clear();
COUNT_profile_set("gpu/texture_cache/textures", 0);
textures_total_size_ = 0;
COUNT_profile_set("gpu/texture_cache/total_size_mb", 0);
texture_used_first_ = texture_used_last_ = nullptr;
// Clear texture descriptor cache.
srv_descriptor_cache_free_.clear();
for (auto& page : srv_descriptor_cache_) {
page.heap->Release();
}
srv_descriptor_cache_.clear();
}
void TextureCache::TextureFetchConstantWritten(uint32_t index) {
texture_keys_in_sync_ &= ~(1u << index);
}
void TextureCache::BeginFrame() {
// In case there was a failure creating something in the previous frame, make
// sure bindings are reset so a new attempt will surely be made if the texture
// is requested again.
ClearBindings();
std::memset(unsupported_format_features_used_, 0,
sizeof(unsupported_format_features_used_));
texture_current_usage_time_ = xe::Clock::QueryHostUptimeMillis();
// If memory usage is too high, destroy unused textures.
uint64_t completed_frame = command_processor_->GetCompletedFrame();
uint32_t limit_soft_mb = cvars::d3d12_texture_cache_limit_soft;
uint32_t limit_hard_mb = cvars::d3d12_texture_cache_limit_hard;
if (IsResolutionScale2X()) {
limit_soft_mb += limit_soft_mb >> 2;
limit_hard_mb += limit_hard_mb >> 2;
}
uint32_t limit_soft_lifetime =
std::max(cvars::d3d12_texture_cache_limit_soft_lifetime, 0) * 1000;
bool destroyed_any = false;
while (texture_used_first_ != nullptr) {
uint64_t total_size_mb = textures_total_size_ >> 20;
bool limit_hard_exceeded = total_size_mb >= limit_hard_mb;
if (total_size_mb < limit_soft_mb && !limit_hard_exceeded) {
break;
}
Texture* texture = texture_used_first_;
if (texture->last_usage_frame > completed_frame) {
break;
}
if (!limit_hard_exceeded &&
(texture->last_usage_time + limit_soft_lifetime) >
texture_current_usage_time_) {
break;
}
destroyed_any = true;
// Remove the texture from the map.
auto found_range = textures_.equal_range(texture->key.GetMapKey());
for (auto iter = found_range.first; iter != found_range.second; ++iter) {
if (iter->second == texture) {
textures_.erase(iter);
break;
}
}
// Unlink the texture.
texture_used_first_ = texture->used_next;
if (texture_used_first_ != nullptr) {
texture_used_first_->used_previous = nullptr;
} else {
texture_used_last_ = nullptr;
}
// Exclude the texture from the memory usage counter.
textures_total_size_ -= texture->resource_size;
// Destroy the texture.
if (texture->cached_srv_descriptor_swizzle !=
Texture::kCachedSRVDescriptorSwizzleMissing) {
srv_descriptor_cache_free_.push_back(texture->cached_srv_descriptor);
}
shared_memory_->UnwatchMemoryRange(texture->base_watch_handle);
shared_memory_->UnwatchMemoryRange(texture->mip_watch_handle);
texture->resource->Release();
delete texture;
}
if (destroyed_any) {
COUNT_profile_set("gpu/texture_cache/textures", textures_.size());
COUNT_profile_set("gpu/texture_cache/total_size_mb",
uint32_t(textures_total_size_ >> 20));
}
}
void TextureCache::EndFrame() {
// Report used unsupported texture formats.
bool unsupported_header_written = false;
for (uint32_t i = 0; i < 64; ++i) {
uint32_t unsupported_features = unsupported_format_features_used_[i];
if (unsupported_features == 0) {
continue;
}
if (!unsupported_header_written) {
XELOGE("Unsupported texture formats used in the frame:");
unsupported_header_written = true;
}
XELOGE("* %s%s%s%s", FormatInfo::Get(TextureFormat(i))->name,
unsupported_features & kUnsupportedResourceBit ? " resource" : "",
unsupported_features & kUnsupportedUnormBit ? " unorm" : "",
unsupported_features & kUnsupportedSnormBit ? " snorm" : "");
unsupported_format_features_used_[i] = 0;
}
}
void TextureCache::RequestTextures(uint32_t used_vertex_texture_mask,
uint32_t used_pixel_texture_mask) {
auto& regs = *register_file_;
#if FINE_GRAINED_DRAW_SCOPES
SCOPE_profile_cpu_f("gpu");
#endif // FINE_GRAINED_DRAW_SCOPES
if (texture_invalidated_.exchange(false, std::memory_order_acquire)) {
// Clear the bindings not only for this draw call, but entirely, because
// loading may be needed in some draw call later, which may have the same
// key for some binding as before the invalidation, but texture_invalidated_
// being false (menu background in Halo 3).
std::memset(texture_bindings_, 0, sizeof(texture_bindings_));
texture_keys_in_sync_ = 0;
}
// Update the texture keys and the textures.
uint32_t used_texture_mask =
used_vertex_texture_mask | used_pixel_texture_mask;
uint32_t index = 0;
while (xe::bit_scan_forward(used_texture_mask, &index)) {
uint32_t index_bit = 1u << index;
used_texture_mask &= ~index_bit;
if (texture_keys_in_sync_ & index_bit) {
continue;
}
TextureBinding& binding = texture_bindings_[index];
const auto& fetch = regs.Get<xenos::xe_gpu_texture_fetch_t>(
XE_GPU_REG_SHADER_CONSTANT_FETCH_00_0 + index * 6);
TextureKey old_key = binding.key;
bool old_has_unsigned = binding.has_unsigned;
bool old_has_signed = binding.has_signed;
BindingInfoFromFetchConstant(fetch, binding.key, &binding.swizzle,
&binding.has_unsigned, &binding.has_signed);
texture_keys_in_sync_ |= index_bit;
if (binding.key.IsInvalid()) {
binding.texture = nullptr;
binding.texture_signed = nullptr;
continue;
}
// Check if need to load the unsigned and the signed versions of the texture
// (if the format is emulated with different host bit representations for
// signed and unsigned - otherwise only the unsigned one is loaded).
bool key_changed = binding.key != old_key;
bool load_unsigned_data = false, load_signed_data = false;
if (IsSignedVersionSeparate(binding.key.format)) {
// Can reuse previously loaded unsigned/signed versions if the key is the
// same and the texture was previously bound as unsigned/signed
// respectively (checking the previous values of has_unsigned/has_signed
// rather than binding.texture != nullptr and binding.texture_signed !=
// nullptr also prevents repeated attempts to load the texture if it has
// failed to load).
if (binding.has_unsigned) {
if (key_changed || !old_has_unsigned) {
binding.texture = FindOrCreateTexture(binding.key);
load_unsigned_data = true;
}
} else {
binding.texture = nullptr;
}
if (binding.has_signed) {
if (key_changed || !old_has_signed) {
TextureKey signed_key = binding.key;
signed_key.signed_separate = 1;
binding.texture_signed = FindOrCreateTexture(signed_key);
load_signed_data = true;
}
} else {
binding.texture_signed = nullptr;
}
} else {
if (key_changed) {
binding.texture = FindOrCreateTexture(binding.key);
load_unsigned_data = true;
}
binding.texture_signed = nullptr;
}
if (load_unsigned_data && binding.texture != nullptr) {
LoadTextureData(binding.texture);
}
if (load_signed_data && binding.texture_signed != nullptr) {
LoadTextureData(binding.texture_signed);
}
}
// Transition the textures to the needed usage.
used_texture_mask = used_vertex_texture_mask | used_pixel_texture_mask;
while (xe::bit_scan_forward(used_texture_mask, &index)) {
uint32_t index_bit = 1u << index;
used_texture_mask &= ~index_bit;
D3D12_RESOURCE_STATES state = D3D12_RESOURCE_STATES(0);
if (used_vertex_texture_mask & index_bit) {
state |= D3D12_RESOURCE_STATE_NON_PIXEL_SHADER_RESOURCE;
}
if (used_pixel_texture_mask & index_bit) {
state |= D3D12_RESOURCE_STATE_PIXEL_SHADER_RESOURCE;
}
TextureBinding& binding = texture_bindings_[index];
if (binding.texture != nullptr) {
// Will be referenced by the command list, so mark as used.
MarkTextureUsed(binding.texture);
command_processor_->PushTransitionBarrier(binding.texture->resource,
binding.texture->state, state);
binding.texture->state = state;
}
if (binding.texture_signed != nullptr) {
MarkTextureUsed(binding.texture_signed);
command_processor_->PushTransitionBarrier(
binding.texture_signed->resource, binding.texture_signed->state,
state);
binding.texture_signed->state = state;
}
}
}
uint64_t TextureCache::GetDescriptorHashForActiveTextures(
const D3D12Shader::TextureSRV* texture_srvs,
uint32_t texture_srv_count) const {
XXH64_state_t hash_state;
XXH64_reset(&hash_state, 0);
for (uint32_t i = 0; i < texture_srv_count; ++i) {
const D3D12Shader::TextureSRV& texture_srv = texture_srvs[i];
// There can be multiple SRVs of the same texture.
XXH64_update(&hash_state, &texture_srv.dimension,
sizeof(texture_srv.dimension));
XXH64_update(&hash_state, &texture_srv.is_signed,
sizeof(texture_srv.is_signed));
XXH64_update(&hash_state, &texture_srv.is_sign_required,
sizeof(texture_srv.is_sign_required));
const TextureBinding& binding =
texture_bindings_[texture_srv.fetch_constant];
XXH64_update(&hash_state, &binding.key, sizeof(binding.key));
XXH64_update(&hash_state, &binding.swizzle, sizeof(binding.swizzle));
XXH64_update(&hash_state, &binding.has_unsigned,
sizeof(binding.has_unsigned));
XXH64_update(&hash_state, &binding.has_signed, sizeof(binding.has_signed));
}
return XXH64_digest(&hash_state);
}
void TextureCache::WriteTextureSRV(const D3D12Shader::TextureSRV& texture_srv,
D3D12_CPU_DESCRIPTOR_HANDLE handle) {
D3D12_SHADER_RESOURCE_VIEW_DESC desc;
desc.Format = DXGI_FORMAT_UNKNOWN;
Dimension binding_dimension;
uint32_t mip_max_level, array_size;
Texture* texture = nullptr;
ID3D12Resource* resource = nullptr;
const TextureBinding& binding = texture_bindings_[texture_srv.fetch_constant];
if (!binding.key.IsInvalid()) {
TextureFormat format = binding.key.format;
if (IsSignedVersionSeparate(format) && texture_srv.is_signed) {
texture = binding.texture_signed;
} else {
texture = binding.texture;
}
if (texture != nullptr) {
resource = texture->resource;
}
if (texture_srv.is_signed) {
// Not supporting signed compressed textures - hopefully DXN and DXT5A are
// not used as signed.
if (binding.has_signed || texture_srv.is_sign_required) {
desc.Format = host_formats_[uint32_t(format)].dxgi_format_snorm;
if (desc.Format == DXGI_FORMAT_UNKNOWN) {
unsupported_format_features_used_[uint32_t(format)] |=
kUnsupportedSnormBit;
}
}
} else {
if (binding.has_unsigned || texture_srv.is_sign_required) {
desc.Format = GetDXGIUnormFormat(binding.key);
if (desc.Format == DXGI_FORMAT_UNKNOWN) {
unsupported_format_features_used_[uint32_t(format)] |=
kUnsupportedUnormBit;
}
}
}
binding_dimension = binding.key.dimension;
mip_max_level = binding.key.mip_max_level;
array_size = binding.key.depth;
// XE_GPU_SWIZZLE and D3D12_SHADER_COMPONENT_MAPPING are the same except for
// one bit.
desc.Shader4ComponentMapping =
binding.swizzle |
D3D12_SHADER_COMPONENT_MAPPING_ALWAYS_SET_BIT_AVOIDING_ZEROMEM_MISTAKES;
} else {
binding_dimension = Dimension::k2D;
mip_max_level = 0;
array_size = 1;
desc.Shader4ComponentMapping = D3D12_ENCODE_SHADER_4_COMPONENT_MAPPING(
D3D12_SHADER_COMPONENT_MAPPING_FORCE_VALUE_0,
D3D12_SHADER_COMPONENT_MAPPING_FORCE_VALUE_0,
D3D12_SHADER_COMPONENT_MAPPING_FORCE_VALUE_0,
D3D12_SHADER_COMPONENT_MAPPING_FORCE_VALUE_0);
}
if (desc.Format == DXGI_FORMAT_UNKNOWN) {
// A null descriptor must still have a valid format.
desc.Format = DXGI_FORMAT_R8G8B8A8_UNORM;
resource = nullptr;
}
NullSRVDescriptorIndex null_descriptor_index;
switch (texture_srv.dimension) {
case TextureDimension::k3D:
desc.ViewDimension = D3D12_SRV_DIMENSION_TEXTURE3D;
desc.Texture3D.MostDetailedMip = 0;
desc.Texture3D.MipLevels = mip_max_level + 1;
desc.Texture3D.ResourceMinLODClamp = 0.0f;
if (binding_dimension != Dimension::k3D) {
// Create a null descriptor so it's safe to sample this texture even
// though it has different dimensions.
resource = nullptr;
}
null_descriptor_index = NullSRVDescriptorIndex::k3D;
break;
case TextureDimension::kCube:
desc.ViewDimension = D3D12_SRV_DIMENSION_TEXTURECUBE;
desc.TextureCube.MostDetailedMip = 0;
desc.TextureCube.MipLevels = mip_max_level + 1;
desc.TextureCube.ResourceMinLODClamp = 0.0f;
if (binding_dimension != Dimension::kCube) {
resource = nullptr;
}
null_descriptor_index = NullSRVDescriptorIndex::kCube;
break;
default:
desc.ViewDimension = D3D12_SRV_DIMENSION_TEXTURE2DARRAY;
desc.Texture2DArray.MostDetailedMip = 0;
desc.Texture2DArray.MipLevels = mip_max_level + 1;
desc.Texture2DArray.FirstArraySlice = 0;
desc.Texture2DArray.ArraySize = array_size;
desc.Texture2DArray.PlaneSlice = 0;
desc.Texture2DArray.ResourceMinLODClamp = 0.0f;
if (binding_dimension == Dimension::k3D ||
binding_dimension == Dimension::kCube) {
resource = nullptr;
}
null_descriptor_index = NullSRVDescriptorIndex::k2DArray;
break;
}
auto provider = command_processor_->GetD3D12Context()->GetD3D12Provider();
auto device = provider->GetDevice();
if (resource == nullptr) {
// Copy a pre-made null descriptor since it's faster than to create an SRV.
device->CopyDescriptorsSimple(
1, handle,
provider->OffsetViewDescriptor(null_srv_descriptor_heap_start_,
uint32_t(null_descriptor_index)),
D3D12_DESCRIPTOR_HEAP_TYPE_CBV_SRV_UAV);
return;
}
MarkTextureUsed(texture);
// Take the descriptor from the cache if it's cached, or create a new one in
// the cache, or directly if this texture was already used with a different
// swizzle. Profiling results say that CreateShaderResourceView takes the
// longest time of draw call processing, and it's very noticeable in many
// games.
bool cached_handle_available = false;
D3D12_CPU_DESCRIPTOR_HANDLE cached_handle = {};
assert_not_null(texture);
if (texture->cached_srv_descriptor_swizzle !=
Texture::kCachedSRVDescriptorSwizzleMissing) {
// Use an existing cached descriptor if it has the needed swizzle.
if (binding.swizzle == texture->cached_srv_descriptor_swizzle) {
cached_handle_available = true;
cached_handle = texture->cached_srv_descriptor;
}
} else {
// Try to create a new cached descriptor if it doesn't exist yet.
if (!srv_descriptor_cache_free_.empty()) {
cached_handle_available = true;
cached_handle = srv_descriptor_cache_free_.back();
srv_descriptor_cache_free_.pop_back();
} else if (srv_descriptor_cache_.empty() ||
srv_descriptor_cache_.back().current_usage >=
SRVDescriptorCachePage::kHeapSize) {
D3D12_DESCRIPTOR_HEAP_DESC new_heap_desc;
new_heap_desc.Type = D3D12_DESCRIPTOR_HEAP_TYPE_CBV_SRV_UAV;
new_heap_desc.NumDescriptors = SRVDescriptorCachePage::kHeapSize;
new_heap_desc.Flags = D3D12_DESCRIPTOR_HEAP_FLAG_NONE;
new_heap_desc.NodeMask = 0;
ID3D12DescriptorHeap* new_heap;
if (SUCCEEDED(device->CreateDescriptorHeap(&new_heap_desc,
IID_PPV_ARGS(&new_heap)))) {
SRVDescriptorCachePage new_page;
new_page.heap = new_heap;
new_page.heap_start = new_heap->GetCPUDescriptorHandleForHeapStart();
new_page.current_usage = 1;
cached_handle_available = true;
cached_handle = new_page.heap_start;
srv_descriptor_cache_.push_back(new_page);
}
} else {
SRVDescriptorCachePage& page = srv_descriptor_cache_.back();
cached_handle_available = true;
cached_handle =
provider->OffsetViewDescriptor(page.heap_start, page.current_usage);
++page.current_usage;
}
if (cached_handle_available) {
device->CreateShaderResourceView(resource, &desc, cached_handle);
texture->cached_srv_descriptor = cached_handle;
texture->cached_srv_descriptor_swizzle = binding.swizzle;
}
}
if (cached_handle_available) {
device->CopyDescriptorsSimple(1, handle, cached_handle,
D3D12_DESCRIPTOR_HEAP_TYPE_CBV_SRV_UAV);
} else {
device->CreateShaderResourceView(resource, &desc, handle);
}
}
TextureCache::SamplerParameters TextureCache::GetSamplerParameters(
const D3D12Shader::SamplerBinding& binding) const {
auto& regs = *register_file_;
const auto& fetch = regs.Get<xenos::xe_gpu_texture_fetch_t>(
XE_GPU_REG_SHADER_CONSTANT_FETCH_00_0 + binding.fetch_constant * 6);
SamplerParameters parameters;
parameters.clamp_x = fetch.clamp_x;
parameters.clamp_y = fetch.clamp_y;
parameters.clamp_z = fetch.clamp_z;
parameters.border_color = fetch.border_color;
uint32_t mip_min_level, mip_max_level;
texture_util::GetSubresourcesFromFetchConstant(
fetch, nullptr, nullptr, nullptr, nullptr, nullptr, &mip_min_level,
&mip_max_level, binding.mip_filter);
parameters.mip_min_level = mip_min_level;
parameters.mip_max_level = std::max(mip_max_level, mip_min_level);
parameters.lod_bias = fetch.lod_bias;
AnisoFilter aniso_filter = binding.aniso_filter == AnisoFilter::kUseFetchConst
? fetch.aniso_filter
: binding.aniso_filter;
aniso_filter = std::min(aniso_filter, AnisoFilter::kMax_16_1);
parameters.aniso_filter = aniso_filter;
if (aniso_filter != AnisoFilter::kDisabled) {
parameters.mag_linear = 1;
parameters.min_linear = 1;
parameters.mip_linear = 1;
} else {
TextureFilter mag_filter =
binding.mag_filter == TextureFilter::kUseFetchConst
? fetch.mag_filter
: binding.mag_filter;
parameters.mag_linear = mag_filter == TextureFilter::kLinear;
TextureFilter min_filter =
binding.min_filter == TextureFilter::kUseFetchConst
? fetch.min_filter
: binding.min_filter;
parameters.min_linear = min_filter == TextureFilter::kLinear;
TextureFilter mip_filter =
binding.mip_filter == TextureFilter::kUseFetchConst
? fetch.mip_filter
: binding.mip_filter;
parameters.mip_linear = mip_filter == TextureFilter::kLinear;
}
return parameters;
}
void TextureCache::WriteSampler(SamplerParameters parameters,
D3D12_CPU_DESCRIPTOR_HANDLE handle) const {
D3D12_SAMPLER_DESC desc;
if (parameters.aniso_filter != AnisoFilter::kDisabled) {
desc.Filter = D3D12_FILTER_ANISOTROPIC;
desc.MaxAnisotropy = 1u << (uint32_t(parameters.aniso_filter) - 1);
} else {
D3D12_FILTER_TYPE d3d_filter_min = parameters.min_linear
? D3D12_FILTER_TYPE_LINEAR
: D3D12_FILTER_TYPE_POINT;
D3D12_FILTER_TYPE d3d_filter_mag = parameters.mag_linear
? D3D12_FILTER_TYPE_LINEAR
: D3D12_FILTER_TYPE_POINT;
D3D12_FILTER_TYPE d3d_filter_mip = parameters.mip_linear
? D3D12_FILTER_TYPE_LINEAR
: D3D12_FILTER_TYPE_POINT;
desc.Filter = D3D12_ENCODE_BASIC_FILTER(
d3d_filter_min, d3d_filter_mag, d3d_filter_mip,
D3D12_FILTER_REDUCTION_TYPE_STANDARD);
desc.MaxAnisotropy = 1;
}
// FIXME(Triang3l): Halfway and mirror clamp to border aren't mapped properly.
static const D3D12_TEXTURE_ADDRESS_MODE kAddressModeMap[] = {
/* kRepeat */ D3D12_TEXTURE_ADDRESS_MODE_WRAP,
/* kMirroredRepeat */ D3D12_TEXTURE_ADDRESS_MODE_MIRROR,
/* kClampToEdge */ D3D12_TEXTURE_ADDRESS_MODE_CLAMP,
/* kMirrorClampToEdge */ D3D12_TEXTURE_ADDRESS_MODE_MIRROR_ONCE,
/* kClampToHalfway */ D3D12_TEXTURE_ADDRESS_MODE_CLAMP,
/* kMirrorClampToHalfway */ D3D12_TEXTURE_ADDRESS_MODE_MIRROR_ONCE,
/* kClampToBorder */ D3D12_TEXTURE_ADDRESS_MODE_BORDER,
/* kMirrorClampToBorder */ D3D12_TEXTURE_ADDRESS_MODE_MIRROR_ONCE,
};
desc.AddressU = kAddressModeMap[uint32_t(parameters.clamp_x)];
desc.AddressV = kAddressModeMap[uint32_t(parameters.clamp_y)];
desc.AddressW = kAddressModeMap[uint32_t(parameters.clamp_z)];
desc.MipLODBias = parameters.lod_bias * (1.0f / 32.0f);
desc.ComparisonFunc = D3D12_COMPARISON_FUNC_NEVER;
// TODO(Triang3l): Border colors k_ACBYCR_BLACK and k_ACBCRY_BLACK.
if (parameters.border_color == BorderColor::k_AGBR_White) {
desc.BorderColor[0] = 1.0f;
desc.BorderColor[1] = 1.0f;
desc.BorderColor[2] = 1.0f;
desc.BorderColor[3] = 1.0f;
} else {
desc.BorderColor[0] = 0.0f;
desc.BorderColor[1] = 0.0f;
desc.BorderColor[2] = 0.0f;
desc.BorderColor[3] = 0.0f;
}
desc.MinLOD = float(parameters.mip_min_level);
desc.MaxLOD = float(parameters.mip_max_level);
auto device =
command_processor_->GetD3D12Context()->GetD3D12Provider()->GetDevice();
device->CreateSampler(&desc, handle);
}
void TextureCache::MarkRangeAsResolved(uint32_t start_unscaled,
uint32_t length_unscaled) {
if (length_unscaled == 0) {
return;
}
start_unscaled &= 0x1FFFFFFF;
length_unscaled = std::min(length_unscaled, 0x20000000 - start_unscaled);
if (IsResolutionScale2X()) {
uint32_t page_first = start_unscaled >> 12;
uint32_t page_last = (start_unscaled + length_unscaled - 1) >> 12;
uint32_t block_first = page_first >> 5;
uint32_t block_last = page_last >> 5;
auto global_lock = global_critical_region_.Acquire();
for (uint32_t i = block_first; i <= block_last; ++i) {
uint32_t add_bits = UINT32_MAX;
if (i == block_first) {
add_bits &= ~((1u << (page_first & 31)) - 1);
}
if (i == block_last && (page_last & 31) != 31) {
add_bits &= (1u << ((page_last & 31) + 1)) - 1;
}
scaled_resolve_pages_[i] |= add_bits;
scaled_resolve_pages_l2_[i >> 6] |= 1ull << (i & 63);
}
}
// Invalidate textures. Toggling individual textures between scaled and
// unscaled also relies on invalidation through shared memory.
shared_memory_->RangeWrittenByGPU(start_unscaled, length_unscaled);
}
bool TextureCache::TileResolvedTexture(
TextureFormat format, uint32_t texture_base, uint32_t texture_pitch,
uint32_t texture_height, bool is_3d, uint32_t offset_x, uint32_t offset_y,
uint32_t offset_z, uint32_t resolve_width, uint32_t resolve_height,
Endian128 endian, ID3D12Resource* buffer, uint32_t buffer_size,
const D3D12_PLACED_SUBRESOURCE_FOOTPRINT& footprint,
uint32_t* written_address_out, uint32_t* written_length_out) {
if (written_address_out) {
*written_address_out = 0;
}
if (written_length_out) {
*written_length_out = 0;
}
ResolveTileMode resolve_tile_mode =
host_formats_[uint32_t(format)].resolve_tile_mode;
if (resolve_tile_mode == ResolveTileMode::kUnknown) {
assert_always();
return false;
}
const ResolveTileModeInfo& resolve_tile_mode_info =
resolve_tile_mode_info_[uint32_t(resolve_tile_mode)];
auto command_list = command_processor_->GetDeferredCommandList();
auto provider = command_processor_->GetD3D12Context()->GetD3D12Provider();
auto device = provider->GetDevice();
uint32_t resolution_scale_log2 = IsResolutionScale2X() ? 1 : 0;
texture_base &= 0x1FFFFFFF;
if (resolve_tile_mode_info.typed_uav_format == DXGI_FORMAT_UNKNOWN) {
assert_false(texture_base & (sizeof(uint32_t) - 1));
texture_base &= ~(uint32_t(sizeof(uint32_t) - 1));
} else {
assert_false(texture_base &
((1u << resolve_tile_mode_info.uav_texel_size_log2) - 1));
texture_base &= ~((1u << resolve_tile_mode_info.uav_texel_size_log2) - 1);
}
texture_pitch = xe::align(texture_pitch, 32u);
texture_height = xe::align(texture_height, 32u);
// Calculate the address and the size of the region that specifically
// is being resolved. Can't just use the texture height for size calculation
// because it's sometimes bigger than needed (in Red Dead Redemption, an UI
// texture used for the letterbox bars alpha is located within a 1280x720
// resolve target, but only 1280x208 is being resolved, and with scaled
// resolution the UI texture gets ignored). This doesn't apply to 3D resolves,
// however, because their tiling is more complex - some excess data will even
// be marked as resolved for them if resolving not to (0,0).
uint32_t bpb_log2 =
xe::log2_floor(FormatInfo::Get(format)->bits_per_pixel >> 3);
if (is_3d) {
texture_base += texture_util::GetTiledOffset3D(
offset_x & ~31u, offset_y & ~31u, offset_z & ~7u, texture_pitch,
texture_height, bpb_log2);
offset_z &= 7;
} else {
texture_base += texture_util::GetTiledOffset2D(
offset_x & ~31u, offset_y & ~31u, texture_pitch, bpb_log2);
offset_z = 0;
}
offset_x &= 31;
offset_y &= 31;
uint32_t texture_size;
uint32_t texture_modified_start = texture_base;
uint32_t texture_modified_length;
if (is_3d) {
// Depth granularity is 4 (though TiledAddress chaining is possible with 8
// granularity).
texture_size = texture_util::GetGuestMipSliceStorageSize(
texture_pitch, texture_height, 4, true, format, nullptr, false);
if (offset_z >= 4) {
texture_modified_start += texture_size;
}
texture_modified_length = texture_size;
texture_size *= 2;
} else {
texture_size = texture_util::GetGuestMipSliceStorageSize(
texture_pitch, xe::align(offset_y + resolve_height, 32u), 1, true,
format, nullptr, false);
texture_modified_length = texture_size;
}
if (texture_size == 0) {
return true;
}
if (resolution_scale_log2) {
if (!EnsureScaledResolveBufferResident(texture_modified_start,
texture_modified_length)) {
return false;
}
} else {
if (!shared_memory_->EnsureTilesResident(texture_modified_start,
texture_modified_length)) {
return false;
}
}
// Tile the texture.
D3D12_CPU_DESCRIPTOR_HANDLE descriptor_cpu_start;
D3D12_GPU_DESCRIPTOR_HANDLE descriptor_gpu_start;
if (command_processor_->RequestViewDescriptors(
ui::d3d12::DescriptorHeapPool::kHeapIndexInvalid, 2, 2,
descriptor_cpu_start, descriptor_gpu_start) ==
ui::d3d12::DescriptorHeapPool::kHeapIndexInvalid) {
return false;
}
if (resolution_scale_log2) {
UseScaledResolveBufferForWriting();
} else {
shared_memory_->UseForWriting();
}
command_processor_->SubmitBarriers();
command_list->D3DSetComputeRootSignature(resolve_tile_root_signature_);
ResolveTileConstants resolve_tile_constants;
resolve_tile_constants.info = uint32_t(endian) | (uint32_t(format) << 3) |
(resolution_scale_log2 << 9) |
((texture_pitch >> 5) << 10) |
(is_3d ? ((texture_height >> 5) << 19) : 0);
resolve_tile_constants.offset = offset_x | (offset_y << 5) | (offset_z << 10);
resolve_tile_constants.size = resolve_width | (resolve_height << 16);
resolve_tile_constants.host_base = uint32_t(footprint.Offset);
resolve_tile_constants.host_pitch = uint32_t(footprint.Footprint.RowPitch);
ui::d3d12::util::CreateRawBufferSRV(device, descriptor_cpu_start, buffer,
buffer_size);
D3D12_CPU_DESCRIPTOR_HANDLE descriptor_cpu_uav =
provider->OffsetViewDescriptor(descriptor_cpu_start, 1);
if (resolve_tile_mode_info.typed_uav_format != DXGI_FORMAT_UNKNOWN) {
// Not sure if this alignment is actually needed in Direct3D 12, but for
// safety. Also not using the full 512 MB buffer as a typed UAV because
// there can't be more than 128M texels in one
// (D3D12_REQ_BUFFER_RESOURCE_TEXEL_COUNT_2_TO_EXP).
resolve_tile_constants.guest_base =
(texture_base & 0xFFFu) >> resolve_tile_mode_info.uav_texel_size_log2;
D3D12_UNORDERED_ACCESS_VIEW_DESC uav_desc;
uav_desc.Format = resolve_tile_mode_info.typed_uav_format;
uav_desc.ViewDimension = D3D12_UAV_DIMENSION_BUFFER;
uav_desc.Buffer.FirstElement =
(texture_base & ~0xFFFu) >> resolve_tile_mode_info.uav_texel_size_log2
<< (resolution_scale_log2 * 2);
uav_desc.Buffer.NumElements =
xe::align(texture_size + (texture_base & 0xFFFu), 0x1000u) >>
resolve_tile_mode_info.uav_texel_size_log2
<< (resolution_scale_log2 * 2);
uav_desc.Buffer.StructureByteStride = 0;
uav_desc.Buffer.CounterOffsetInBytes = 0;
uav_desc.Buffer.Flags = D3D12_BUFFER_UAV_FLAG_NONE;
device->CreateUnorderedAccessView(resolution_scale_log2
? scaled_resolve_buffer_
: shared_memory_->GetBuffer(),
nullptr, &uav_desc, descriptor_cpu_uav);
} else {
if (resolution_scale_log2) {
resolve_tile_constants.guest_base = texture_base & 0xFFF;
CreateScaledResolveBufferRawUAV(
descriptor_cpu_uav, texture_base >> 12,
((texture_base + texture_size - 1) >> 12) - (texture_base >> 12) + 1);
} else {
resolve_tile_constants.guest_base = texture_base;
shared_memory_->WriteRawUAVDescriptor(descriptor_cpu_uav);
}
}
command_list->D3DSetComputeRootDescriptorTable(1, descriptor_gpu_start);
command_list->D3DSetComputeRoot32BitConstants(
0, sizeof(resolve_tile_constants) / sizeof(uint32_t),
&resolve_tile_constants, 0);
command_processor_->SetComputePipeline(
resolve_tile_pipelines_[uint32_t(resolve_tile_mode)]);
// Each group processes 32x32 texels after resolution scaling has been
// applied.
command_list->D3DDispatch(
((resolve_width << resolution_scale_log2) + 31) >> 5,
((resolve_height << resolution_scale_log2) + 31) >> 5, 1);
// Commit the write.
command_processor_->PushUAVBarrier(resolution_scale_log2
? scaled_resolve_buffer_
: shared_memory_->GetBuffer());
// Invalidate textures and mark the range as scaled if needed.
MarkRangeAsResolved(texture_modified_start, texture_modified_length);
if (written_address_out) {
*written_address_out = texture_modified_start;
}
if (written_length_out) {
*written_length_out = texture_modified_length;
}
return true;
}
bool TextureCache::EnsureScaledResolveBufferResident(uint32_t start_unscaled,
uint32_t length_unscaled) {
assert_true(IsResolutionScale2X());
if (length_unscaled == 0) {
return true;
}
start_unscaled &= 0x1FFFFFFF;
if ((0x20000000 - start_unscaled) < length_unscaled) {
// Exceeds the physical address space.
return false;
}
uint32_t heap_first = (start_unscaled << 2) >> kScaledResolveHeapSizeLog2;
uint32_t heap_last = ((start_unscaled + length_unscaled - 1) << 2) >>
kScaledResolveHeapSizeLog2;
for (uint32_t i = heap_first; i <= heap_last; ++i) {
if (scaled_resolve_heaps_[i] != nullptr) {
continue;
}
auto provider = command_processor_->GetD3D12Context()->GetD3D12Provider();
auto device = provider->GetDevice();
auto direct_queue = provider->GetDirectQueue();
D3D12_HEAP_DESC heap_desc = {};
heap_desc.SizeInBytes = kScaledResolveHeapSize;
heap_desc.Properties.Type = D3D12_HEAP_TYPE_DEFAULT;
heap_desc.Flags = D3D12_HEAP_FLAG_ALLOW_ONLY_BUFFERS;
if (FAILED(device->CreateHeap(&heap_desc,
IID_PPV_ARGS(&scaled_resolve_heaps_[i])))) {
XELOGE("Texture cache: Failed to create a scaled resolve tile heap");
return false;
}
++scaled_resolve_heap_count_;
COUNT_profile_set(
"gpu/texture_cache/scaled_resolve_buffer_used_mb",
scaled_resolve_heap_count_ << (kScaledResolveHeapSizeLog2 - 20));
D3D12_TILED_RESOURCE_COORDINATE region_start_coordinates;
region_start_coordinates.X = (i << kScaledResolveHeapSizeLog2) /
D3D12_TILED_RESOURCE_TILE_SIZE_IN_BYTES;
region_start_coordinates.Y = 0;
region_start_coordinates.Z = 0;
region_start_coordinates.Subresource = 0;
D3D12_TILE_REGION_SIZE region_size;
region_size.NumTiles =
kScaledResolveHeapSize / D3D12_TILED_RESOURCE_TILE_SIZE_IN_BYTES;
region_size.UseBox = FALSE;
D3D12_TILE_RANGE_FLAGS range_flags = D3D12_TILE_RANGE_FLAG_NONE;
UINT heap_range_start_offset = 0;
UINT range_tile_count =
kScaledResolveHeapSize / D3D12_TILED_RESOURCE_TILE_SIZE_IN_BYTES;
// FIXME(Triang3l): This may cause issues if the emulator is shut down
// mid-frame and the heaps are destroyed before tile mappings are updated
// (awaiting the fence won't catch this then). Defer this until the actual
// command list submission.
direct_queue->UpdateTileMappings(
scaled_resolve_buffer_, 1, &region_start_coordinates, &region_size,
scaled_resolve_heaps_[i], 1, &range_flags, &heap_range_start_offset,
&range_tile_count, D3D12_TILE_MAPPING_FLAG_NONE);
}
return true;
}
void TextureCache::UseScaledResolveBufferForReading() {
assert_true(IsResolutionScale2X());
command_processor_->PushTransitionBarrier(
scaled_resolve_buffer_, scaled_resolve_buffer_state_,
D3D12_RESOURCE_STATE_NON_PIXEL_SHADER_RESOURCE);
scaled_resolve_buffer_state_ = D3D12_RESOURCE_STATE_NON_PIXEL_SHADER_RESOURCE;
}
void TextureCache::UseScaledResolveBufferForWriting() {
assert_true(IsResolutionScale2X());
command_processor_->PushTransitionBarrier(
scaled_resolve_buffer_, scaled_resolve_buffer_state_,
D3D12_RESOURCE_STATE_UNORDERED_ACCESS);
scaled_resolve_buffer_state_ = D3D12_RESOURCE_STATE_UNORDERED_ACCESS;
}
void TextureCache::CreateScaledResolveBufferRawSRV(
D3D12_CPU_DESCRIPTOR_HANDLE handle, uint32_t first_unscaled_4kb_page,
uint32_t unscaled_4kb_page_count) {
assert_true(IsResolutionScale2X());
first_unscaled_4kb_page = std::min(first_unscaled_4kb_page, 0x1FFFFu);
unscaled_4kb_page_count = std::max(
std::min(unscaled_4kb_page_count, 0x20000u - first_unscaled_4kb_page),
1u);
ui::d3d12::util::CreateRawBufferSRV(
command_processor_->GetD3D12Context()->GetD3D12Provider()->GetDevice(),
handle, scaled_resolve_buffer_, unscaled_4kb_page_count << 14,
first_unscaled_4kb_page << 14);
}
void TextureCache::CreateScaledResolveBufferRawUAV(
D3D12_CPU_DESCRIPTOR_HANDLE handle, uint32_t first_unscaled_4kb_page,
uint32_t unscaled_4kb_page_count) {
assert_true(IsResolutionScale2X());
first_unscaled_4kb_page = std::min(first_unscaled_4kb_page, 0x1FFFFu);
unscaled_4kb_page_count = std::max(
std::min(unscaled_4kb_page_count, 0x20000u - first_unscaled_4kb_page),
1u);
ui::d3d12::util::CreateRawBufferUAV(
command_processor_->GetD3D12Context()->GetD3D12Provider()->GetDevice(),
handle, scaled_resolve_buffer_, unscaled_4kb_page_count << 14,
first_unscaled_4kb_page << 14);
}
bool TextureCache::RequestSwapTexture(D3D12_CPU_DESCRIPTOR_HANDLE handle,
TextureFormat& format_out) {
auto& regs = *register_file_;
const auto& fetch = regs.Get<xenos::xe_gpu_texture_fetch_t>(
XE_GPU_REG_SHADER_CONSTANT_FETCH_00_0);
TextureKey key;
uint32_t swizzle;
BindingInfoFromFetchConstant(fetch, key, &swizzle, nullptr, nullptr);
if (key.base_page == 0 || key.dimension != Dimension::k2D) {
return false;
}
Texture* texture = FindOrCreateTexture(key);
if (texture == nullptr || !LoadTextureData(texture)) {
return false;
}
MarkTextureUsed(texture);
command_processor_->PushTransitionBarrier(
texture->resource, texture->state,
D3D12_RESOURCE_STATE_PIXEL_SHADER_RESOURCE);
texture->state = D3D12_RESOURCE_STATE_PIXEL_SHADER_RESOURCE;
D3D12_SHADER_RESOURCE_VIEW_DESC srv_desc;
srv_desc.Format = GetDXGIUnormFormat(key);
srv_desc.ViewDimension = D3D12_SRV_DIMENSION_TEXTURE2D;
srv_desc.Shader4ComponentMapping =
swizzle |
D3D12_SHADER_COMPONENT_MAPPING_ALWAYS_SET_BIT_AVOIDING_ZEROMEM_MISTAKES;
srv_desc.Texture2D.MostDetailedMip = 0;
srv_desc.Texture2D.MipLevels = 1;
srv_desc.Texture2D.PlaneSlice = 0;
srv_desc.Texture2D.ResourceMinLODClamp = 0.0f;
auto device =
command_processor_->GetD3D12Context()->GetD3D12Provider()->GetDevice();
device->CreateShaderResourceView(texture->resource, &srv_desc, handle);
format_out = key.format;
return true;
}
bool TextureCache::IsDecompressionNeeded(TextureFormat format, uint32_t width,
uint32_t height) {
DXGI_FORMAT dxgi_format_uncompressed =
host_formats_[uint32_t(format)].dxgi_format_uncompressed;
if (dxgi_format_uncompressed == DXGI_FORMAT_UNKNOWN) {
return false;
}
const FormatInfo* format_info = FormatInfo::Get(format);
return (width & (format_info->block_width - 1)) != 0 ||
(height & (format_info->block_height - 1)) != 0;
}
TextureCache::LoadMode TextureCache::GetLoadMode(TextureKey key) {
const HostFormat& host_format = host_formats_[uint32_t(key.format)];
if (key.signed_separate) {
return host_format.load_mode_snorm;
}
if (IsDecompressionNeeded(key.format, key.width, key.height)) {
return host_format.decompress_mode;
}
return host_format.load_mode;
}
void TextureCache::BindingInfoFromFetchConstant(
const xenos::xe_gpu_texture_fetch_t& fetch, TextureKey& key_out,
uint32_t* swizzle_out, bool* has_unsigned_out, bool* has_signed_out) {
// Reset the key and the swizzle.
key_out.MakeInvalid();
if (swizzle_out != nullptr) {
*swizzle_out = xenos::XE_GPU_SWIZZLE_0 | (xenos::XE_GPU_SWIZZLE_0 << 3) |
(xenos::XE_GPU_SWIZZLE_0 << 6) |
(xenos::XE_GPU_SWIZZLE_0 << 9);
}
if (has_unsigned_out != nullptr) {
*has_unsigned_out = false;
}
if (has_signed_out != nullptr) {
*has_signed_out = false;
}
switch (fetch.type) {
case xenos::FetchConstantType::kTexture:
break;
case xenos::FetchConstantType::kInvalidTexture:
if (cvars::gpu_allow_invalid_fetch_constants) {
break;
}
XELOGW(
"Texture fetch constant (%.8X %.8X %.8X %.8X %.8X %.8X) has "
"\"invalid\" type! This is incorrect behavior, but you can try "
"bypassing this by launching Xenia with "
"--gpu_allow_invalid_fetch_constants=true.",
fetch.dword_0, fetch.dword_1, fetch.dword_2, fetch.dword_3,
fetch.dword_4, fetch.dword_5);
return;
default:
XELOGW(
"Texture fetch constant (%.8X %.8X %.8X %.8X %.8X %.8X) is "
"completely invalid!",
fetch.dword_0, fetch.dword_1, fetch.dword_2, fetch.dword_3,
fetch.dword_4, fetch.dword_5);
return;
}
uint32_t width, height, depth_or_faces;
uint32_t base_page, mip_page, mip_max_level;
texture_util::GetSubresourcesFromFetchConstant(
fetch, &width, &height, &depth_or_faces, &base_page, &mip_page, nullptr,
&mip_max_level);
if (base_page == 0 && mip_page == 0) {
// No texture data at all.
return;
}
if (fetch.dimension == Dimension::k1D && width > 8192) {
XELOGE(
"1D texture is too wide (%u) - ignoring! "
"Report the game to Xenia developers",
width);
return;
}
TextureFormat format = GetBaseFormat(fetch.format);
key_out.base_page = base_page;
key_out.mip_page = mip_page;
key_out.dimension = fetch.dimension;
key_out.width = width;
key_out.height = height;
key_out.depth = depth_or_faces;
key_out.mip_max_level = mip_max_level;
key_out.tiled = fetch.tiled;
key_out.packed_mips = fetch.packed_mips;
key_out.format = format;
key_out.endianness = fetch.endianness;
if (swizzle_out != nullptr) {
uint32_t swizzle = 0;
for (uint32_t i = 0; i < 4; ++i) {
uint32_t swizzle_component = (fetch.swizzle >> (i * 3)) & 0b111;
if (swizzle_component >= 4) {
// Get rid of 6 and 7 values (to prevent device losses if the game has
// something broken) the quick and dirty way - by changing them to 4 (0)
// and 5 (1).
swizzle_component &= 0b101;
} else {
swizzle_component =
host_formats_[uint32_t(format)].swizzle[swizzle_component];
}
swizzle |= swizzle_component << (i * 3);
}
*swizzle_out = swizzle;
}
if (has_unsigned_out != nullptr) {
*has_unsigned_out = fetch.sign_x != TextureSign::kSigned ||
fetch.sign_y != TextureSign::kSigned ||
fetch.sign_z != TextureSign::kSigned ||
fetch.sign_w != TextureSign::kSigned;
}
if (has_signed_out != nullptr) {
*has_signed_out = fetch.sign_x == TextureSign::kSigned ||
fetch.sign_y == TextureSign::kSigned ||
fetch.sign_z == TextureSign::kSigned ||
fetch.sign_w == TextureSign::kSigned;
}
}
void TextureCache::LogTextureKeyAction(TextureKey key, const char* action) {
XELOGGPU(
"%s %s %s%ux%ux%u %s %s texture with %u %spacked mip level%s, "
"base at 0x%.8X, mips at 0x%.8X",
action, key.tiled ? "tiled" : "linear",
key.scaled_resolve ? "2x-scaled " : "", key.width, key.height, key.depth,
dimension_names_[uint32_t(key.dimension)],
FormatInfo::Get(key.format)->name, key.mip_max_level + 1,
key.packed_mips ? "" : "un", key.mip_max_level != 0 ? "s" : "",
key.base_page << 12, key.mip_page << 12);
}
void TextureCache::LogTextureAction(const Texture* texture,
const char* action) {
XELOGGPU(
"%s %s %s%ux%ux%u %s %s texture with %u %spacked mip level%s, "
"base at 0x%.8X (size %u), mips at 0x%.8X (size %u)",
action, texture->key.tiled ? "tiled" : "linear",
texture->key.scaled_resolve ? "2x-scaled " : "", texture->key.width,
texture->key.height, texture->key.depth,
dimension_names_[uint32_t(texture->key.dimension)],
FormatInfo::Get(texture->key.format)->name,
texture->key.mip_max_level + 1, texture->key.packed_mips ? "" : "un",
texture->key.mip_max_level != 0 ? "s" : "", texture->key.base_page << 12,
texture->base_size, texture->key.mip_page << 12, texture->mip_size);
}
TextureCache::Texture* TextureCache::FindOrCreateTexture(TextureKey key) {
// Check if the texture is a 2x-scaled resolve texture.
if (IsResolutionScale2X() && key.tiled) {
LoadMode load_mode = GetLoadMode(key);
if (load_mode != LoadMode::kUnknown &&
load_pipelines_2x_[uint32_t(load_mode)] != nullptr) {
uint32_t base_size = 0, mip_size = 0;
texture_util::GetTextureTotalSize(
key.dimension, key.width, key.height, key.depth, key.format,
key.tiled, key.packed_mips, key.mip_max_level,
key.base_page != 0 ? &base_size : nullptr,
key.mip_page != 0 ? &mip_size : nullptr);
if ((base_size != 0 &&
IsRangeScaledResolved(key.base_page << 12, base_size)) ||
(mip_size != 0 &&
IsRangeScaledResolved(key.mip_page << 12, mip_size))) {
key.scaled_resolve = 1;
}
}
}
uint64_t map_key = key.GetMapKey();
// Try to find an existing texture.
// TODO(Triang3l): Reuse a texture with mip_page unchanged, but base_page
// previously 0, now not 0, to save memory - common case in streaming.
auto found_range = textures_.equal_range(map_key);
for (auto iter = found_range.first; iter != found_range.second; ++iter) {
Texture* found_texture = iter->second;
if (found_texture->key.bucket_key == key.bucket_key) {
return found_texture;
}
}
// Create the resource. If failed to create one, don't create a texture object
// at all so it won't be in indeterminate state.
D3D12_RESOURCE_DESC desc;
desc.Format = GetDXGIResourceFormat(key);
if (desc.Format == DXGI_FORMAT_UNKNOWN) {
unsupported_format_features_used_[uint32_t(key.format)] |=
kUnsupportedResourceBit;
return nullptr;
}
if (key.dimension == Dimension::k3D) {
desc.Dimension = D3D12_RESOURCE_DIMENSION_TEXTURE3D;
} else {
// 1D textures are treated as 2D for simplicity.
desc.Dimension = D3D12_RESOURCE_DIMENSION_TEXTURE2D;
}
desc.Alignment = 0;
desc.Width = key.width;
desc.Height = key.height;
if (key.scaled_resolve) {
desc.Width *= 2;
desc.Height *= 2;
}
desc.DepthOrArraySize = key.depth;
desc.MipLevels = key.mip_max_level + 1;
desc.SampleDesc.Count = 1;
desc.SampleDesc.Quality = 0;
desc.Layout = D3D12_TEXTURE_LAYOUT_UNKNOWN;
// Untiling through a buffer instead of using unordered access because copying
// is not done that often.
desc.Flags = D3D12_RESOURCE_FLAG_NONE;
auto device =
command_processor_->GetD3D12Context()->GetD3D12Provider()->GetDevice();
// Assuming untiling will be the next operation.
D3D12_RESOURCE_STATES state = D3D12_RESOURCE_STATE_COPY_DEST;
ID3D12Resource* resource;
if (FAILED(device->CreateCommittedResource(
&ui::d3d12::util::kHeapPropertiesDefault, D3D12_HEAP_FLAG_NONE, &desc,
state, nullptr, IID_PPV_ARGS(&resource)))) {
LogTextureKeyAction(key, "Failed to create");
return nullptr;
}
// Create the texture object and add it to the map.
Texture* texture = new Texture;
texture->key = key;
texture->resource = resource;
texture->resource_size =
device->GetResourceAllocationInfo(0, 1, &desc).SizeInBytes;
texture->state = state;
texture->last_usage_frame = command_processor_->GetCurrentFrame();
texture->last_usage_time = texture_current_usage_time_;
texture->used_previous = texture_used_last_;
texture->used_next = nullptr;
if (texture_used_last_ != nullptr) {
texture_used_last_->used_next = texture;
} else {
texture_used_first_ = texture;
}
texture_used_last_ = texture;
texture->mip_offsets[0] = 0;
uint32_t width_blocks, height_blocks, depth_blocks;
uint32_t array_size = key.dimension != Dimension::k3D ? key.depth : 1;
if (key.base_page != 0) {
texture_util::GetGuestMipBlocks(key.dimension, key.width, key.height,
key.depth, key.format, 0, width_blocks,
height_blocks, depth_blocks);
uint32_t slice_size = texture_util::GetGuestMipSliceStorageSize(
width_blocks, height_blocks, depth_blocks, key.tiled, key.format,
&texture->pitches[0]);
texture->slice_sizes[0] = slice_size;
texture->base_size = slice_size * array_size;
texture->base_in_sync = false;
} else {
texture->base_size = 0;
texture->slice_sizes[0] = 0;
texture->pitches[0] = 0;
// Never try to upload the base level if there is none.
texture->base_in_sync = true;
}
texture->mip_size = 0;
if (key.mip_page != 0) {
uint32_t mip_max_storage_level = key.mip_max_level;
if (key.packed_mips) {
mip_max_storage_level =
std::min(mip_max_storage_level,
texture_util::GetPackedMipLevel(key.width, key.height));
}
for (uint32_t i = 1; i <= mip_max_storage_level; ++i) {
texture_util::GetGuestMipBlocks(key.dimension, key.width, key.height,
key.depth, key.format, i, width_blocks,
height_blocks, depth_blocks);
texture->mip_offsets[i] = texture->mip_size;
uint32_t slice_size = texture_util::GetGuestMipSliceStorageSize(
width_blocks, height_blocks, depth_blocks, key.tiled, key.format,
&texture->pitches[i]);
texture->slice_sizes[i] = slice_size;
texture->mip_size += slice_size * array_size;
}
// The rest are either packed levels or don't exist at all.
for (uint32_t i = mip_max_storage_level + 1;
i < xe::countof(texture->mip_offsets); ++i) {
texture->mip_offsets[i] = texture->mip_offsets[mip_max_storage_level];
texture->slice_sizes[i] = texture->slice_sizes[mip_max_storage_level];
texture->pitches[i] = texture->pitches[mip_max_storage_level];
}
texture->mips_in_sync = false;
} else {
std::memset(&texture->mip_offsets[1], 0,
(xe::countof(texture->mip_offsets) - 1) * sizeof(uint32_t));
std::memset(&texture->slice_sizes[1], 0,
(xe::countof(texture->slice_sizes) - 1) * sizeof(uint32_t));
std::memset(&texture->pitches[1], 0,
(xe::countof(texture->pitches) - 1) * sizeof(uint32_t));
// Never try to upload the mipmaps if there are none.
texture->mips_in_sync = true;
}
texture->base_watch_handle = nullptr;
texture->mip_watch_handle = nullptr;
texture->cached_srv_descriptor_swizzle =
Texture::kCachedSRVDescriptorSwizzleMissing;
textures_.insert(std::make_pair(map_key, texture));
COUNT_profile_set("gpu/texture_cache/textures", textures_.size());
textures_total_size_ += texture->resource_size;
COUNT_profile_set("gpu/texture_cache/total_size_mb",
uint32_t(textures_total_size_ >> 20));
LogTextureAction(texture, "Created");
return texture;
}
bool TextureCache::LoadTextureData(Texture* texture) {
// See what we need to upload.
bool base_in_sync, mips_in_sync;
{
auto global_lock = global_critical_region_.Acquire();
base_in_sync = texture->base_in_sync;
mips_in_sync = texture->mips_in_sync;
}
if (base_in_sync && mips_in_sync) {
return true;
}
auto command_list = command_processor_->GetDeferredCommandList();
auto provider = command_processor_->GetD3D12Context()->GetD3D12Provider();
auto device = provider->GetDevice();
// Get the pipeline.
LoadMode load_mode = GetLoadMode(texture->key);
if (load_mode == LoadMode::kUnknown) {
return false;
}
bool scaled_resolve = texture->key.scaled_resolve ? true : false;
ID3D12PipelineState* pipeline = scaled_resolve
? load_pipelines_2x_[uint32_t(load_mode)]
: load_pipelines_[uint32_t(load_mode)];
if (pipeline == nullptr) {
return false;
}
// Request uploading of the texture data to the shared memory.
// This is also necessary when resolution scale is used - the texture cache
// relies on shared memory for invalidation of both unscaled and scaled
// textures! Plus a texture may be unscaled partially, when only a portion of
// its pages is invalidated, in this case we'll need the texture from the
// shared memory to load the unscaled parts.
if (!base_in_sync) {
if (!shared_memory_->RequestRange(texture->key.base_page << 12,
texture->base_size)) {
return false;
}
}
if (!mips_in_sync) {
if (!shared_memory_->RequestRange(texture->key.mip_page << 12,
texture->mip_size)) {
return false;
}
}
if (scaled_resolve) {
// Make sure all heaps are created.
if (!EnsureScaledResolveBufferResident(texture->key.base_page << 12,
texture->base_size)) {
return false;
}
if (!EnsureScaledResolveBufferResident(texture->key.mip_page << 12,
texture->mip_size)) {
return false;
}
}
// Update LRU caching because the texture will be used by the command list.
MarkTextureUsed(texture);
// Get the guest layout.
bool is_3d = texture->key.dimension == Dimension::k3D;
uint32_t width = texture->key.width;
uint32_t height = texture->key.height;
uint32_t depth = is_3d ? texture->key.depth : 1;
uint32_t slice_count = is_3d ? 1 : texture->key.depth;
TextureFormat guest_format = texture->key.format;
const FormatInfo* guest_format_info = FormatInfo::Get(guest_format);
uint32_t block_width = guest_format_info->block_width;
uint32_t block_height = guest_format_info->block_height;
// Get the host layout and the buffer.
D3D12_RESOURCE_DESC resource_desc = texture->resource->GetDesc();
D3D12_PLACED_SUBRESOURCE_FOOTPRINT host_layouts[D3D12_REQ_MIP_LEVELS];
UINT64 host_slice_size;
device->GetCopyableFootprints(&resource_desc, 0, resource_desc.MipLevels, 0,
host_layouts, nullptr, nullptr,
&host_slice_size);
// The shaders deliberately overflow for simplicity, and GetCopyableFootprints
// doesn't align the size of the last row (or the size if there's only one
// row, not really sure) to row pitch, so add some excess bytes for safety.
// 1x1 8-bit and 16-bit textures even give a device loss because the raw UAV
// has a size of 0.
host_slice_size =
xe::align(host_slice_size, UINT64(D3D12_TEXTURE_DATA_PITCH_ALIGNMENT));
D3D12_RESOURCE_STATES copy_buffer_state =
D3D12_RESOURCE_STATE_UNORDERED_ACCESS;
ID3D12Resource* copy_buffer = command_processor_->RequestScratchGPUBuffer(
uint32_t(host_slice_size), copy_buffer_state);
if (copy_buffer == nullptr) {
return false;
}
// Begin loading.
uint32_t mip_first = base_in_sync ? 1 : 0;
uint32_t mip_last = mips_in_sync ? 0 : resource_desc.MipLevels - 1;
// Can't address more than 512 MB directly on Nvidia - need two separate UAV
// descriptors for base and mips.
bool separate_base_and_mips_descriptors =
scaled_resolve && mip_first == 0 && mip_last != 0;
uint32_t descriptor_count = separate_base_and_mips_descriptors ? 4 : 2;
D3D12_CPU_DESCRIPTOR_HANDLE descriptor_cpu_start;
D3D12_GPU_DESCRIPTOR_HANDLE descriptor_gpu_start;
if (command_processor_->RequestViewDescriptors(
ui::d3d12::DescriptorHeapPool::kHeapIndexInvalid, descriptor_count,
descriptor_count, descriptor_cpu_start, descriptor_gpu_start) ==
ui::d3d12::DescriptorHeapPool::kHeapIndexInvalid) {
command_processor_->ReleaseScratchGPUBuffer(copy_buffer, copy_buffer_state);
return false;
}
if (scaled_resolve) {
// TODO(Triang3l): Allow partial invalidation of scaled textures - send a
// part of scaled_resolve_pages_ to the shader and choose the source
// according to whether a specific page contains scaled texture data. If
// it's not, duplicate the texels from the unscaled version - will be
// blocky with filtering, but better than nothing.
UseScaledResolveBufferForReading();
uint32_t srv_descriptor_offset = 0;
if (mip_first == 0) {
CreateScaledResolveBufferRawSRV(
provider->OffsetViewDescriptor(descriptor_cpu_start,
srv_descriptor_offset),
texture->key.base_page, (texture->base_size + 0xFFF) >> 12);
srv_descriptor_offset += 2;
}
if (mip_last != 0) {
CreateScaledResolveBufferRawSRV(
provider->OffsetViewDescriptor(descriptor_cpu_start,
srv_descriptor_offset),
texture->key.mip_page, (texture->mip_size + 0xFFF) >> 12);
}
} else {
shared_memory_->UseForReading();
shared_memory_->WriteRawSRVDescriptor(descriptor_cpu_start);
}
// Create two destination descriptors since the table has both.
for (uint32_t i = 1; i < descriptor_count; i += 2) {
ui::d3d12::util::CreateRawBufferUAV(
device, provider->OffsetViewDescriptor(descriptor_cpu_start, i),
copy_buffer, uint32_t(host_slice_size));
}
command_processor_->SetComputePipeline(pipeline);
command_list->D3DSetComputeRootSignature(load_root_signature_);
if (!separate_base_and_mips_descriptors) {
// Will be bound later.
command_list->D3DSetComputeRootDescriptorTable(1, descriptor_gpu_start);
}
// Submit commands.
command_processor_->PushTransitionBarrier(texture->resource, texture->state,
D3D12_RESOURCE_STATE_COPY_DEST);
texture->state = D3D12_RESOURCE_STATE_COPY_DEST;
auto cbuffer_pool = command_processor_->GetConstantBufferPool();
LoadConstants load_constants;
load_constants.is_3d = is_3d ? 1 : 0;
load_constants.endianness = uint32_t(texture->key.endianness);
load_constants.guest_format = uint32_t(guest_format);
if (!texture->key.packed_mips) {
load_constants.guest_mip_offset[0] = 0;
load_constants.guest_mip_offset[1] = 0;
load_constants.guest_mip_offset[2] = 0;
}
for (uint32_t i = 0; i < slice_count; ++i) {
command_processor_->PushTransitionBarrier(
copy_buffer, copy_buffer_state, D3D12_RESOURCE_STATE_UNORDERED_ACCESS);
copy_buffer_state = D3D12_RESOURCE_STATE_UNORDERED_ACCESS;
for (uint32_t j = mip_first; j <= mip_last; ++j) {
if (scaled_resolve) {
// Offset already applied in the buffer because more than 512 MB can't
// be directly addresses on Nvidia.
load_constants.guest_base = 0;
} else {
if (j == 0) {
load_constants.guest_base = texture->key.base_page << 12;
} else {
load_constants.guest_base = texture->key.mip_page << 12;
}
}
load_constants.guest_base +=
texture->mip_offsets[j] + i * texture->slice_sizes[j];
load_constants.guest_pitch = texture->key.tiled
? LoadConstants::kGuestPitchTiled
: texture->pitches[j];
load_constants.host_base = uint32_t(host_layouts[j].Offset);
load_constants.host_pitch = host_layouts[j].Footprint.RowPitch;
load_constants.size_texels[0] = std::max(width >> j, 1u);
load_constants.size_texels[1] = std::max(height >> j, 1u);
load_constants.size_texels[2] = std::max(depth >> j, 1u);
load_constants.size_blocks[0] =
(load_constants.size_texels[0] + (block_width - 1)) / block_width;
load_constants.size_blocks[1] =
(load_constants.size_texels[1] + (block_height - 1)) / block_height;
load_constants.size_blocks[2] = load_constants.size_texels[2];
if (j == 0) {
load_constants.guest_storage_width_height[0] =
xe::align(load_constants.size_blocks[0], 32u);
load_constants.guest_storage_width_height[1] =
xe::align(load_constants.size_blocks[1], 32u);
} else {
load_constants.guest_storage_width_height[0] =
xe::align(xe::next_pow2(load_constants.size_blocks[0]), 32u);
load_constants.guest_storage_width_height[1] =
xe::align(xe::next_pow2(load_constants.size_blocks[1]), 32u);
}
if (texture->key.packed_mips) {
texture_util::GetPackedMipOffset(width, height, depth, guest_format, j,
load_constants.guest_mip_offset[0],
load_constants.guest_mip_offset[1],
load_constants.guest_mip_offset[2]);
}
D3D12_GPU_VIRTUAL_ADDRESS cbuffer_gpu_address;
uint8_t* cbuffer_mapping = cbuffer_pool->Request(
command_processor_->GetCurrentFrame(),
xe::align(uint32_t(sizeof(load_constants)), 256u), nullptr, nullptr,
&cbuffer_gpu_address);
if (cbuffer_mapping == nullptr) {
command_processor_->ReleaseScratchGPUBuffer(copy_buffer,
copy_buffer_state);
return false;
}
std::memcpy(cbuffer_mapping, &load_constants, sizeof(load_constants));
command_list->D3DSetComputeRootConstantBufferView(0, cbuffer_gpu_address);
if (separate_base_and_mips_descriptors) {
if (j == 0) {
command_list->D3DSetComputeRootDescriptorTable(1,
descriptor_gpu_start);
} else if (j == 1) {
command_list->D3DSetComputeRootDescriptorTable(
1, provider->OffsetViewDescriptor(descriptor_gpu_start, 2));
}
}
command_processor_->SubmitBarriers();
// Each thread group processes 32x32x1 blocks after resolution scaling has
// been applied.
uint32_t group_count_x = load_constants.size_blocks[0];
uint32_t group_count_y = load_constants.size_blocks[1];
if (texture->key.scaled_resolve) {
group_count_x *= 2;
group_count_y *= 2;
}
group_count_x = (group_count_x + 31) >> 5;
group_count_y = (group_count_y + 31) >> 5;
command_list->D3DDispatch(group_count_x, group_count_y,
load_constants.size_blocks[2]);
}
command_processor_->PushUAVBarrier(copy_buffer);
command_processor_->PushTransitionBarrier(copy_buffer, copy_buffer_state,
D3D12_RESOURCE_STATE_COPY_SOURCE);
copy_buffer_state = D3D12_RESOURCE_STATE_COPY_SOURCE;
command_processor_->SubmitBarriers();
UINT slice_first_subresource = i * resource_desc.MipLevels;
for (uint32_t j = mip_first; j <= mip_last; ++j) {
D3D12_TEXTURE_COPY_LOCATION location_source, location_dest;
location_source.pResource = copy_buffer;
location_source.Type = D3D12_TEXTURE_COPY_TYPE_PLACED_FOOTPRINT;
location_source.PlacedFootprint = host_layouts[j];
location_dest.pResource = texture->resource;
location_dest.Type = D3D12_TEXTURE_COPY_TYPE_SUBRESOURCE_INDEX;
location_dest.SubresourceIndex = slice_first_subresource + j;
command_list->CopyTexture(location_dest, location_source);
}
}
command_processor_->ReleaseScratchGPUBuffer(copy_buffer, copy_buffer_state);
// Mark the ranges as uploaded and watch them. This is needed for scaled
// resolves as well to detect when the CPU wants to reuse the memory for a
// regular texture or a vertex buffer, and thus the scaled resolve version is
// not up to date anymore.
{
auto global_lock = global_critical_region_.Acquire();
texture->base_in_sync = true;
texture->mips_in_sync = true;
if (!base_in_sync) {
texture->base_watch_handle = shared_memory_->WatchMemoryRange(
texture->key.base_page << 12, texture->base_size, WatchCallbackThunk,
this, texture, 0);
}
if (!mips_in_sync) {
texture->mip_watch_handle = shared_memory_->WatchMemoryRange(
texture->key.mip_page << 12, texture->mip_size, WatchCallbackThunk,
this, texture, 1);
}
}
LogTextureAction(texture, "Loaded");
return true;
}
void TextureCache::MarkTextureUsed(Texture* texture) {
uint64_t current_frame = command_processor_->GetCurrentFrame();
// This is called very frequently, don't relink unless needed for caching.
if (texture->last_usage_frame != current_frame) {
texture->last_usage_frame = current_frame;
texture->last_usage_time = texture_current_usage_time_;
if (texture->used_next == nullptr) {
// Simplify the code a bit - already in the end of the list.
return;
}
if (texture->used_previous != nullptr) {
texture->used_previous->used_next = texture->used_next;
} else {
texture_used_first_ = texture->used_next;
}
texture->used_next->used_previous = texture->used_previous;
texture->used_previous = texture_used_last_;
texture->used_next = nullptr;
if (texture_used_last_ != nullptr) {
texture_used_last_->used_next = texture;
}
texture_used_last_ = texture;
}
}
void TextureCache::WatchCallbackThunk(void* context, void* data,
uint64_t argument,
bool invalidated_by_gpu) {
TextureCache* texture_cache = reinterpret_cast<TextureCache*>(context);
texture_cache->WatchCallback(reinterpret_cast<Texture*>(data), argument != 0);
}
void TextureCache::WatchCallback(Texture* texture, bool is_mip) {
// Mutex already locked here.
if (is_mip) {
texture->mips_in_sync = false;
texture->mip_watch_handle = nullptr;
} else {
texture->base_in_sync = false;
texture->base_watch_handle = nullptr;
}
texture_invalidated_.store(true, std::memory_order_release);
}
void TextureCache::ClearBindings() {
std::memset(texture_bindings_, 0, sizeof(texture_bindings_));
texture_keys_in_sync_ = 0;
// Already reset everything.
texture_invalidated_.store(false, std::memory_order_relaxed);
}
bool TextureCache::IsRangeScaledResolved(uint32_t start_unscaled,
uint32_t length_unscaled) {
if (!IsResolutionScale2X() || length_unscaled == 0) {
return false;
}
start_unscaled &= 0x1FFFFFFF;
length_unscaled = std::min(length_unscaled, 0x20000000 - start_unscaled);
// Two-level check for faster rejection since resolve targets are usually
// placed in relatively small and localized memory portions (confirmed by
// testing - pretty much all times the deeper level was entered, the texture
// was a resolve target).
uint32_t page_first = start_unscaled >> 12;
uint32_t page_last = (start_unscaled + length_unscaled - 1) >> 12;
uint32_t block_first = page_first >> 5;
uint32_t block_last = page_last >> 5;
uint32_t l2_block_first = block_first >> 6;
uint32_t l2_block_last = block_last >> 6;
auto global_lock = global_critical_region_.Acquire();
for (uint32_t i = l2_block_first; i <= l2_block_last; ++i) {
uint64_t l2_block = scaled_resolve_pages_l2_[i];
if (i == l2_block_first) {
l2_block &= ~((1ull << (block_first & 63)) - 1);
}
if (i == l2_block_last && (block_last & 63) != 63) {
l2_block &= (1ull << ((block_last & 63) + 1)) - 1;
}
uint32_t block_relative_index;
while (xe::bit_scan_forward(l2_block, &block_relative_index)) {
l2_block &= ~(1ull << block_relative_index);
uint32_t block_index = (i << 6) + block_relative_index;
uint32_t check_bits = UINT32_MAX;
if (block_index == block_first) {
check_bits &= ~((1u << (page_first & 31)) - 1);
}
if (block_index == block_last && (page_last & 31) != 31) {
check_bits &= (1u << ((page_last & 31) + 1)) - 1;
}
if (scaled_resolve_pages_[block_index] & check_bits) {
return true;
}
}
}
return false;
}
void TextureCache::ScaledResolveGlobalWatchCallbackThunk(
void* context, uint32_t address_first, uint32_t address_last,
bool invalidated_by_gpu) {
TextureCache* texture_cache = reinterpret_cast<TextureCache*>(context);
texture_cache->ScaledResolveGlobalWatchCallback(address_first, address_last,
invalidated_by_gpu);
}
void TextureCache::ScaledResolveGlobalWatchCallback(uint32_t address_first,
uint32_t address_last,
bool invalidated_by_gpu) {
assert_true(IsResolutionScale2X());
if (invalidated_by_gpu) {
// Resolves themselves do exactly the opposite of what this should do.
return;
}
// Mark scaled resolve ranges as non-scaled. Textures themselves will be
// invalidated by their own per-range watches.
uint32_t resolve_page_first = address_first >> 12;
uint32_t resolve_page_last = address_last >> 12;
uint32_t resolve_block_first = resolve_page_first >> 5;
uint32_t resolve_block_last = resolve_page_last >> 5;
uint32_t resolve_l2_block_first = resolve_block_first >> 6;
uint32_t resolve_l2_block_last = resolve_block_last >> 6;
for (uint32_t i = resolve_l2_block_first; i <= resolve_l2_block_last; ++i) {
uint64_t resolve_l2_block = scaled_resolve_pages_l2_[i];
uint32_t resolve_block_relative_index;
while (
xe::bit_scan_forward(resolve_l2_block, &resolve_block_relative_index)) {
resolve_l2_block &= ~(1ull << resolve_block_relative_index);
uint32_t resolve_block_index = (i << 6) + resolve_block_relative_index;
uint32_t resolve_keep_bits = 0;
if (resolve_block_index == resolve_block_first) {
resolve_keep_bits |= (1u << (resolve_page_first & 31)) - 1;
}
if (resolve_block_index == resolve_block_last &&
(resolve_page_last & 31) != 31) {
resolve_keep_bits |= ~((1u << ((resolve_page_last & 31) + 1)) - 1);
}
scaled_resolve_pages_[resolve_block_index] &= resolve_keep_bits;
if (scaled_resolve_pages_[resolve_block_index] == 0) {
scaled_resolve_pages_l2_[i] &= ~(1ull << resolve_block_relative_index);
}
}
}
}
} // namespace d3d12
} // namespace gpu
} // namespace xe