[GPU/UI] XeSL readability improvements + float suffix

Use the _xe suffix instead of the xesl_ prefix for quicker visual
recognition of identifiers, also switch to snake_case for consistency.

Also add the f suffix to float32 literals because the Metal Shading
Language is based on C++.
This commit is contained in:
Triang3l
2025-08-19 21:36:06 +03:00
parent 3b4b04c371
commit 04d5c40d0d
201 changed files with 90552 additions and 92369 deletions

View File

@@ -10,31 +10,31 @@
#include "pixel_formats.xesli"
#include "texture_load.xesli"
xesl_writeTypedStorageBuffer_declare(xesl_uint4, xe_texture_load_dest, set=0,
binding=0, u0, space0)
xesl_typedStorageBuffer_declare(xesl_uint4, xe_texture_load_source, set=1,
binding=0, t0, space0)
xesl_entry_bindings_begin_compute
array_buffer_wo_declare_xe(uint4_xe, xe_texture_load_dest, set=0, binding=0, u0,
space0)
array_buffer_declare_xe(uint4_xe, xe_texture_load_source, set=1, binding=0, t0,
space0)
entry_bindings_begin_compute_xe
XE_TEXTURE_LOAD_CONSTANT_BUFFER_BINDING
xesl_entry_binding_next
xesl_writeTypedStorageBuffer_binding(xesl_uint4, xe_texture_load_dest,
buffer(1))
xesl_entry_binding_next
xesl_typedStorageBuffer_binding(xesl_uint4, xe_texture_load_source, buffer(2))
xesl_entry_bindings_end_inputs_begin_compute
xesl_entry_input_globalInvocationID
xesl_entry_inputs_end_code_begin_compute
entry_binding_next_xe
array_buffer_wo_binding_xe(uint4_xe, xe_texture_load_dest, buffer(1))
entry_binding_next_xe
array_buffer_binding_xe(uint4_xe, xe_texture_load_source, buffer(2))
entry_bindings_end_inputs_begin_compute_xe
entry_in_global_thread_id_xe
entry_inputs_end_code_begin_compute_xe
{
// 1 thread = 2 DXT3 blocks to 8x4 R8G8B8A8 texels.
XeTextureLoadInfo load_info = XeTextureLoadGetInfo(
xesl_function_call_constantBuffer(xe_texture_load_constants));
xesl_uint3 block_index = xesl_GlobalInvocationID << xesl_uint3(1u, 0u, 0u);
xesl_dont_flatten
if (any(xesl_greaterThanEqual(block_index.xy, load_info.size_blocks.xy))) {
pass_const_buffer_xe(xe_texture_load_constants));
uint3_xe block_index = in_global_thread_id_xe << uint3_xe(1u, 0u, 0u);
dont_flatten_xe
if (any(greater_than_equal_xe(block_index.xy, load_info.size_blocks.xy))) {
return;
}
xesl_uint3 texel_index_host = block_index << xesl_uint3(2u, 2u, 0u);
uint3_xe texel_index_host = block_index << uint3_xe(2u, 2u, 0u);
uint block_offset_host = uint(
(XeTextureHostLinearOffset(xesl_int3(texel_index_host),
(XeTextureHostLinearOffset(int3_xe(texel_index_host),
load_info.host_pitch, load_info.height_texels,
4u) +
load_info.host_offset) >> 4u);
@@ -42,45 +42,46 @@ xesl_entry_inputs_end_code_begin_compute
uint block_offset_guest =
XeTextureLoadGuestBlockOffset(load_info, block_index, 16u, 4u) >> 4u;
uint i;
xesl_unroll for (i = 0u; i < 2u; ++i) {
unroll_xe for (i = 0u; i < 2u; ++i) {
if (i != 0u) {
++block_offset_host;
// Odd block = even block + 32 guest bytes when tiled.
block_offset_guest += load_info.is_tiled ? 2u : 1u;
}
xesl_uint4 block = XeEndianSwap32(
xesl_typedStorageBufferLoad(xe_texture_load_source, block_offset_guest),
uint4_xe block = XeEndianSwap32(
array_buffer_load_xe(xe_texture_load_source, block_offset_guest),
load_info.endian_32);
xesl_uint2 bgr_end_8in10 = XeDXTColorEndpointsToBGR8In10(block.z);
uint2_xe bgr_end_8in10 = XeDXTColorEndpointsToBGR8In10(block.z);
// Sort the color indices so they can be used as weights for the second
// endpoint.
uint bgr_weights = XeDXTHighColorWeights(block.w);
xesl_typedStorageBufferStore(
array_buffer_store_xe(
xe_texture_load_dest, block_offset_host,
XeDXTOpaqueRowToRGB8(bgr_end_8in10, bgr_weights) +
((block.xxxx >> xesl_uint4(0u, 4u, 8u, 12u)) & 0xFu) * 0x11000000u);
xesl_dont_flatten if (texel_index_host.y + 1u < load_info.height_texels) {
xesl_typedStorageBufferStore(
((block.xxxx >> uint4_xe(0u, 4u, 8u, 12u)) & 0xFu) * 0x11000000u);
dont_flatten_xe if (texel_index_host.y + 1u < load_info.height_texels) {
array_buffer_store_xe(
xe_texture_load_dest, block_offset_host + elements_pitch_host,
XeDXTOpaqueRowToRGB8(bgr_end_8in10, bgr_weights >> 8u) +
((block.xxxx >> xesl_uint4(16u, 20u, 24u, 28u)) & 0xFu) *
((block.xxxx >> uint4_xe(16u, 20u, 24u, 28u)) & 0xFu) *
0x11000000u);
xesl_dont_flatten if (texel_index_host.y + 2u < load_info.height_texels) {
xesl_typedStorageBufferStore(
dont_flatten_xe if (texel_index_host.y + 2u < load_info.height_texels) {
array_buffer_store_xe(
xe_texture_load_dest, block_offset_host + 2u * elements_pitch_host,
XeDXTOpaqueRowToRGB8(bgr_end_8in10, bgr_weights >> 16u) +
((block.yyyy >> xesl_uint4(0u, 4u, 8u, 12u)) & 0xFu) *
((block.yyyy >> uint4_xe(0u, 4u, 8u, 12u)) & 0xFu) *
0x11000000u);
xesl_dont_flatten
dont_flatten_xe
if (texel_index_host.y + 3u < load_info.height_texels) {
xesl_typedStorageBufferStore(
array_buffer_store_xe(
xe_texture_load_dest,
block_offset_host + 3u * elements_pitch_host,
XeDXTOpaqueRowToRGB8(bgr_end_8in10, bgr_weights >> 24u) +
((block.yyyy >> xesl_uint4(16u, 20u, 24u, 28u)) & 0xFu) *
((block.yyyy >> uint4_xe(16u, 20u, 24u, 28u)) & 0xFu) *
0x11000000u);
}
}
}
}
xesl_entry_code_end_compute
}
entry_code_end_compute_xe