From 0d395ce9abf0374ae895d5474bc942dcfbfe7542 Mon Sep 17 00:00:00 2001 From: goldislead <69987043+goldislead@users.noreply.github.com> Date: Sat, 15 Aug 2026 17:25:07 -0700 Subject: [PATCH] [GPU] Move debug cvars to a new TOML block, misc cleanup native_2x_msaa is now debug_msaa_4x_as_2x, hopefully clearing up any user ambiguity. --- src/xenia/gpu/command_processor.cc | 6 ++--- .../gpu/d3d12/d3d12_render_target_cache.cc | 4 ++-- src/xenia/gpu/draw_extent_estimator.cc | 4 ++-- src/xenia/gpu/dxbc_shader_translator.cc | 4 ++-- src/xenia/gpu/gpu_flags.cc | 12 +++++----- src/xenia/gpu/graphics_system.cc | 2 +- src/xenia/gpu/primitive_processor.cc | 10 ++++---- src/xenia/gpu/render_target_cache.cc | 23 +++++++++---------- src/xenia/gpu/render_target_cache.h | 2 +- src/xenia/gpu/texture_cache.cc | 10 ++++---- src/xenia/gpu/trace_dump.cc | 5 ++-- .../gpu/vulkan/vulkan_render_target_cache.cc | 2 +- 12 files changed, 42 insertions(+), 42 deletions(-) diff --git a/src/xenia/gpu/command_processor.cc b/src/xenia/gpu/command_processor.cc index 21008b50e..5865d9036 100644 --- a/src/xenia/gpu/command_processor.cc +++ b/src/xenia/gpu/command_processor.cc @@ -32,18 +32,18 @@ DEFINE_bool( "Only does anything in debug builds, if set will log every write to a gpu " "register done by a guest. Does not log writes that are done by the CP on " "its own, just ones the guest makes or instructs it to make.", - "GPU"); + "Logging"); DEFINE_bool(disassemble_pm4, false, "Only does anything in debug builds, if set will disassemble and " "log all PM4 packets sent to the CP.", - "GPU"); + "Logging"); DEFINE_bool( log_ringbuffer_kickoff_initiator_bts, false, "Only does anything in debug builds, if set will log the pseudo-stacktrace " "of the guest thread that wrote the new read position.", - "GPU"); + "Logging"); DEFINE_bool(clear_memory_page_state, false, "Refresh state of memory pages to enable gpu written data. (Use " diff --git a/src/xenia/gpu/d3d12/d3d12_render_target_cache.cc b/src/xenia/gpu/d3d12/d3d12_render_target_cache.cc index e3dfbda71..14b12747b 100644 --- a/src/xenia/gpu/d3d12/d3d12_render_target_cache.cc +++ b/src/xenia/gpu/d3d12/d3d12_render_target_cache.cc @@ -36,7 +36,7 @@ DEFINE_bool( native_stencil_value_output_d3d12_intel, false, "Allow stencil reference output usage on Direct3D 12 on Intel GPUs - not " "working on UHD Graphics 630 as of March 2021 (driver 27.20.0100.8336).", - "GPU"); + "GPU.Debug"); // TODO(Triang3l): Make ROV the default when it's optimized better (for // instance, using static shader modifications to pass render target // parameters). @@ -483,7 +483,7 @@ bool D3D12RenderTargetCache::Initialize() { // Check if 2x MSAA is supported or needs to be emulated with 4x MSAA // instead. - if (cvars::native_2x_msaa) { + if (!cvars::debug_msaa_4x_as_2x) { msaa_2x_supported_ = true; static constexpr DXGI_FORMAT kRenderTargetDXGIFormats[] = { DXGI_FORMAT_R16G16B16A16_FLOAT, diff --git a/src/xenia/gpu/draw_extent_estimator.cc b/src/xenia/gpu/draw_extent_estimator.cc index 9e2fa92be..7162a322f 100644 --- a/src/xenia/gpu/draw_extent_estimator.cc +++ b/src/xenia/gpu/draw_extent_estimator.cc @@ -34,7 +34,7 @@ DEFINE_bool( "this prevents corruption of other render targets located after the " "current ones in the EDRAM by lossy range ownership transfers done for " "those draws.", - "GPU"); + "GPU.Debug"); DEFINE_bool( execute_unclipped_draw_vs_on_cpu_with_scissor, false, "Don't restrict the usage of execute_unclipped_draw_vs_on_cpu to only " @@ -45,7 +45,7 @@ DEFINE_bool( "Enabling this may cause excessive processing of vertices on the CPU, as " "some games draw rectangles (for their UI, for instance) without clipping, " "but with a proper scissor rectangle.", - "GPU"); + "GPU.Debug"); namespace xe { namespace gpu { diff --git a/src/xenia/gpu/dxbc_shader_translator.cc b/src/xenia/gpu/dxbc_shader_translator.cc index be7627327..5eee7f8df 100644 --- a/src/xenia/gpu/dxbc_shader_translator.cc +++ b/src/xenia/gpu/dxbc_shader_translator.cc @@ -29,11 +29,11 @@ DEFINE_bool(dxbc_switch, true, "(possibly the shader compiler tries to flatten them). On Intel " "HD Graphics, this is ignored because of a crash with the switch " "instruction.", - "GPU"); + "GPU.Debug"); DEFINE_bool(dxbc_source_map, false, "Disassemble Xenos instructions as comments in the resulting DXBC " "for debugging.", - "GPU"); + "GPU.Debug"); namespace xe { namespace gpu { diff --git a/src/xenia/gpu/gpu_flags.cc b/src/xenia/gpu/gpu_flags.cc index 806df0de2..e9ee535b5 100644 --- a/src/xenia/gpu/gpu_flags.cc +++ b/src/xenia/gpu/gpu_flags.cc @@ -10,13 +10,13 @@ #include "xenia/gpu/gpu_flags.h" DEFINE_path(trace_gpu_prefix, "scratch/gpu/", - "Prefix path for GPU trace files.", "GPU"); -DEFINE_bool(trace_gpu_stream, false, "Trace all GPU packets.", "GPU"); + "Prefix path for GPU trace files.", "GPU.Debug"); +DEFINE_bool(trace_gpu_stream, false, "Trace all GPU packets.", "GPU.Debug"); DEFINE_path( dump_shaders, "", "For shader debugging, path to dump GPU shaders to as they are compiled.", - "GPU"); + "GPU.Debug"); DEFINE_bool(vsync, true, "Enable VSYNC.", "GPU"); @@ -43,7 +43,7 @@ DEFINE_bool( "Disable filtering between cube map faces near edges where possible " "(Vulkan with VK_EXT_non_seamless_cube_map) to reproduce the Direct3D 9 " "behavior.", - "GPU"); + "GPU.Debug"); // Extremely bright screen borders in 4D5307E6. // Reading between texels with half-pixel offset in 58410954. @@ -57,7 +57,7 @@ DEFINE_bool( "textures, for instance, when they are read between texels rather than " "at texel centers, or the leftmost/topmost pixels may not be fully covered " "when MSAA is used with fullscreen passes.", - "GPU"); + "GPU.Debug"); DEFINE_int32(occlusion_query_fake_lower_threshold, 80, "Lower end of the fake sample count value written on " @@ -104,7 +104,7 @@ DEFINE_int32(anisotropic_override, -1, DEFINE_bool(no_discard_stencil_in_transfer_pipelines, false, "Skip stencil bit discard in render target transfer pipelines. " "May improve performance on some GPUs.", - "GPU"); + "GPU.Debug"); DEFINE_bool(gpu_3d_to_2d_texture, true, "Handle shaders that sample 3D textures as 2D by creating a 2D " diff --git a/src/xenia/gpu/graphics_system.cc b/src/xenia/gpu/graphics_system.cc index 50956d463..851513945 100644 --- a/src/xenia/gpu/graphics_system.cc +++ b/src/xenia/gpu/graphics_system.cc @@ -34,7 +34,7 @@ DEFINE_bool( store_shaders, true, "Store shaders persistently and load them when loading games to avoid " "runtime spikes and freezes when playing the game not for the first time.", - "GPU"); + "GPU.Debug"); namespace xe { namespace gpu { diff --git a/src/xenia/gpu/primitive_processor.cc b/src/xenia/gpu/primitive_processor.cc index a39f5a9cf..7364c36d8 100644 --- a/src/xenia/gpu/primitive_processor.cc +++ b/src/xenia/gpu/primitive_processor.cc @@ -30,13 +30,13 @@ DEFINE_bool( "For host graphics API downlevel support testing only, force CPU " "conversion of triangle fans to triangle lists even if the host supports " "triangle fan primitives natively.", - "GPU"); + "GPU.Debug"); DEFINE_bool( force_convert_line_loops_to_strips, false, "For host graphics API downlevel support testing only, force CPU " "conversion of line loops to line strips even if the host supports line " "loop primitives natively.", - "GPU"); + "GPU.Debug"); DEFINE_bool( force_convert_quad_lists_to_triangle_lists, false, "For host graphics API downlevel support testing only, force CPU " @@ -44,14 +44,14 @@ DEFINE_bool( "list primitives natively or via geometry shader emulation.\n" "May also be useful for graphics debugging when the debugger doesn't " "display the geometry generated by geometry shaders properly.", - "GPU"); + "GPU.Debug"); DEFINE_bool( ignore_32bit_vertex_index_support, false, "For host graphics API downlevel testing only (useful only for Qualcomm " "Adreno 4xx-level host GPU testing), force indirection or pre-masking and " "pre-swapping of 32-bit vertex indices as if the host only supports 24-bit " "indices.", - "GPU"); + "GPU.Debug"); // TODO(Triang3l): More investigation of the cache threshold as cache lookups // and insertions require global critical region locking, and insertions also // require protecting pages. At 1024, the cache only made the performance worse @@ -64,7 +64,7 @@ DEFINE_int32( "Setting this to a very high value may result in excessive CPU processing, " "while a very low value may result in excessive locking and lookups.\n" "Negative values disable caching.", - "GPU"); + "GPU.Debug"); namespace xe { namespace gpu { diff --git a/src/xenia/gpu/render_target_cache.cc b/src/xenia/gpu/render_target_cache.cc index 80723f08c..28c68668e 100644 --- a/src/xenia/gpu/render_target_cache.cc +++ b/src/xenia/gpu/render_target_cache.cc @@ -21,6 +21,12 @@ #include "xenia/gpu/registers.h" #include "xenia/gpu/xenos.h" +DEFINE_bool( + debug_msaa_4x_as_2x, false, + "Use 4x MSAA with 2 samples instead of native 2x MSAA when available. " + "For scalability testing on host GPU APIs where 2x is not mandatory. MSAA " + "will be of a similar or worse quality and use more memory.", + "GPU.Debug"); DEFINE_bool( depth_transfer_not_equal_test, true, "When transferring data between depth render targets, use the \"not " @@ -33,7 +39,7 @@ DEFINE_bool( "beneficial to subsequent rendering, while setting this to false may " "reduce bandwidth usage during transfers as the previous depth won't need " "to be read.", - "GPU"); + "GPU.Debug"); // Lossless round trip: 545407F2. // Lossy round trip with the "greater or equal" test afterwards: 4D530919. // Lossy round trip with the "equal" test afterwards: 535107F5, 565507EF. @@ -163,7 +169,7 @@ DEFINE_bool( "Greatly increases accuracy for this format, but may result in render " "target copying costs if the game switches between 8_8_8_8_GAMMA and " "8_8_8_8 views for the same EDRAM render target.", - "GPU"); + "GPU.Debug"); DEFINE_bool( mrt_edram_used_range_clamp_to_min, true, "With host render targets, if multiple render targets are bound, estimate " @@ -173,14 +179,7 @@ DEFINE_bool( "Has effect primarily on draws without viewport clipping.\n" "Setting this to false results in higher accuracy in rare cases, but may " "increase the amount of copying that needs to be done sometimes.", - "GPU"); -DEFINE_bool( - native_2x_msaa, true, - "Use host 2x MSAA when available. Can be disabled for scalability testing " - "on host GPU APIs where 2x is not mandatory, in this case, 2 samples of 4x " - "MSAA will be used instead (with similar or worse quality and higher " - "memory usage).", - "GPU"); + "GPU.Debug"); DEFINE_bool( native_stencil_value_output, true, "Use pixel shader stencil reference output where available for purposes " @@ -193,7 +192,7 @@ DEFINE_bool( "When the host can only support 16_16 and 16_16_16_16 render targets as " "-1...1, remap -32...32 to -1...1 to use the full possible range of " "values, at the expense of multiplicative blending correctness.", - "GPU"); + "GPU.Debug"); // Enabled by default as the GPU is overall usually the bottleneck when the // pixel shader interlock render backend implementation is used, anything that // may improve GPU performance is favorable. @@ -206,7 +205,7 @@ DEFINE_bool( "needed when the ownership of a EDRAM range is changed.\n" "If this is enabled, excessive barriers may be eliminated when switching " "between different render targets in separate EDRAM locations.", - "GPU"); + "GPU.Debug"); namespace xe { namespace gpu { diff --git a/src/xenia/gpu/render_target_cache.h b/src/xenia/gpu/render_target_cache.h index 962b28cd4..018eff7ac 100644 --- a/src/xenia/gpu/render_target_cache.h +++ b/src/xenia/gpu/render_target_cache.h @@ -28,12 +28,12 @@ #include "xenia/gpu/shader.h" #include "xenia/gpu/xenos.h" +DECLARE_bool(debug_msaa_4x_as_2x); DECLARE_bool(depth_transfer_not_equal_test); DECLARE_bool(depth_float24_round); DECLARE_bool(depth_float24_convert_in_pixel_shader); DECLARE_bool(draw_resolution_scaled_texture_offsets); DECLARE_bool(gamma_render_target_as_unorm16); -DECLARE_bool(native_2x_msaa); DECLARE_bool(native_stencil_value_output); DECLARE_bool(snorm16_render_target_full_range); diff --git a/src/xenia/gpu/texture_cache.cc b/src/xenia/gpu/texture_cache.cc index f6558e23f..534d5a0de 100644 --- a/src/xenia/gpu/texture_cache.cc +++ b/src/xenia/gpu/texture_cache.cc @@ -40,17 +40,17 @@ DEFINE_uint32( texture_cache_memory_limit_soft, 384, "Maximum host texture memory usage (in megabytes) above which old textures " "will be destroyed.", - "GPU"); + "GPU.Debug"); DEFINE_uint32( texture_cache_memory_limit_soft_lifetime, 30, "Seconds a texture should be unused to be considered old enough to be " "deleted if texture memory usage exceeds texture_cache_memory_limit_soft.", - "GPU"); + "GPU.Debug"); DEFINE_uint32( texture_cache_memory_limit_hard, 768, "Maximum host texture memory usage (in megabytes) above which textures " "will be destroyed as soon as possible.", - "GPU"); + "GPU.Debug"); DEFINE_uint32( texture_cache_memory_limit_render_to_texture, 24, "Part of the host texture memory budget (in megabytes) that will be scaled " @@ -60,11 +60,11 @@ DEFINE_uint32( "render-to-texture (resolve) targets and 384 - 24 = 360 MB of regular " "textures - so with 2x2 resolution scaling, the soft limit will be 360 + " "96 MB, and with 3x3, it will be 360 + 216 MB.", - "GPU"); + "GPU.Debug"); DEFINE_bool(tiled_shared_memory, true, "Enable tiled/sparse resources for efficient large address space " "support. Disable for graphics debugger compatibility.", - "GPU"); + "GPU.Debug"); namespace xe { namespace gpu { diff --git a/src/xenia/gpu/trace_dump.cc b/src/xenia/gpu/trace_dump.cc index 6a27325c7..6009ee806 100644 --- a/src/xenia/gpu/trace_dump.cc +++ b/src/xenia/gpu/trace_dump.cc @@ -35,8 +35,9 @@ #pragma clang diagnostic pop #endif -DEFINE_path(target_trace_file, "", "Specifies the trace file to load.", "GPU"); -DEFINE_path(trace_dump_path, "", "Output path for dumped files.", "GPU"); +DEFINE_path(target_trace_file, "", "Specifies the trace file to load.", + "GPU.Debug"); +DEFINE_path(trace_dump_path, "", "Output path for dumped files.", "GPU.Debug"); namespace xe { namespace gpu { diff --git a/src/xenia/gpu/vulkan/vulkan_render_target_cache.cc b/src/xenia/gpu/vulkan/vulkan_render_target_cache.cc index 24b778c8b..613f8fa17 100644 --- a/src/xenia/gpu/vulkan/vulkan_render_target_cache.cc +++ b/src/xenia/gpu/vulkan/vulkan_render_target_cache.cc @@ -265,7 +265,7 @@ bool VulkanRenderTargetCache::Initialize(uint32_t shared_memory_binding_count) { // 2x MSAA support. // TODO(Triang3l): Handle sampledImageIntegerSampleCounts 4 not supported in // transfers. - if (cvars::native_2x_msaa) { + if (!cvars::debug_msaa_4x_as_2x) { // Multisampled integer sampled images are optional in Vulkan and in Xenia. msaa_2x_attachments_supported_ = (device_properties.framebufferColorSampleCounts &