[Vulkan] Drop vertex buffer residency cache, hoist global lock like D3D12
This commit is contained in:
@@ -1272,13 +1272,6 @@ void VulkanCommandProcessor::WriteRegister(uint32_t index, uint32_t value) {
|
|||||||
texture_cache_->TextureFetchConstantWritten(
|
texture_cache_->TextureFetchConstantWritten(
|
||||||
(index - XE_GPU_REG_SHADER_CONSTANT_FETCH_00_0) / 6);
|
(index - XE_GPU_REG_SHADER_CONSTANT_FETCH_00_0) / 6);
|
||||||
}
|
}
|
||||||
// Invalidate vertex buffer cache for this fetch constant.
|
|
||||||
// Each vertex fetch constant is 2 DWORDs.
|
|
||||||
uint32_t vfetch_index = (index - XE_GPU_REG_SHADER_CONSTANT_FETCH_00_0) / 2;
|
|
||||||
if (vfetch_index < 96) {
|
|
||||||
vertex_buffers_in_sync_[vfetch_index >> 6] &=
|
|
||||||
~(uint64_t(1) << (vfetch_index & 63));
|
|
||||||
}
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
void VulkanCommandProcessor::WriteRegistersFromMem(uint32_t start_index,
|
void VulkanCommandProcessor::WriteRegistersFromMem(uint32_t start_index,
|
||||||
@@ -1318,13 +1311,6 @@ void VulkanCommandProcessor::IssueSwap(uint32_t frontbuffer_ptr,
|
|||||||
uint32_t frontbuffer_height) {
|
uint32_t frontbuffer_height) {
|
||||||
SCOPE_profile_cpu_f("gpu");
|
SCOPE_profile_cpu_f("gpu");
|
||||||
|
|
||||||
// Reset vertex buffer cache at frame boundaries to pick up any memory
|
|
||||||
// changes. This still provides significant savings by avoiding redundant
|
|
||||||
// RequestRange calls within the same frame (potentially 100k+ calls reduced
|
|
||||||
// to a few hundred).
|
|
||||||
vertex_buffers_in_sync_[0] = 0;
|
|
||||||
vertex_buffers_in_sync_[1] = 0;
|
|
||||||
|
|
||||||
ui::Presenter* presenter = graphics_system_->presenter();
|
ui::Presenter* presenter = graphics_system_->presenter();
|
||||||
if (!presenter) {
|
if (!presenter) {
|
||||||
return;
|
return;
|
||||||
@@ -2602,8 +2588,6 @@ bool VulkanCommandProcessor::IssueDraw(xenos::PrimitiveType prim_type,
|
|||||||
}
|
}
|
||||||
|
|
||||||
// Ensure vertex buffers are resident.
|
// Ensure vertex buffers are resident.
|
||||||
// Uses caching to avoid redundant RequestRange calls - only re-validates
|
|
||||||
// when fetch constants are written (detected in WriteRegister).
|
|
||||||
//
|
//
|
||||||
// Use the vertex_fetch_bitmap instead of vertex_bindings() to avoid using
|
// Use the vertex_fetch_bitmap instead of vertex_bindings() to avoid using
|
||||||
// cached/stale vertex binding indices. The bitmap is populated during shader
|
// cached/stale vertex binding indices. The bitmap is populated during shader
|
||||||
@@ -2611,86 +2595,82 @@ bool VulkanCommandProcessor::IssueDraw(xenos::PrimitiveType prim_type,
|
|||||||
// references, allowing us to check the current register values at draw time.
|
// references, allowing us to check the current register values at draw time.
|
||||||
const Shader::ConstantRegisterMap& constant_map_vertex =
|
const Shader::ConstantRegisterMap& constant_map_vertex =
|
||||||
vertex_shader->constant_register_map();
|
vertex_shader->constant_register_map();
|
||||||
for (uint32_t i = 0; i < xe::countof(constant_map_vertex.vertex_fetch_bitmap);
|
{
|
||||||
++i) {
|
uint32_t vfetch_addresses[96];
|
||||||
uint32_t vfetch_bits_remaining = constant_map_vertex.vertex_fetch_bitmap[i];
|
uint32_t vfetch_sizes[96];
|
||||||
uint32_t j;
|
uint32_t vfetch_current_queued = 0;
|
||||||
while (xe::bit_scan_forward(vfetch_bits_remaining, &j)) {
|
for (uint32_t i = 0;
|
||||||
vfetch_bits_remaining = xe::clear_lowest_bit(vfetch_bits_remaining);
|
i < xe::countof(constant_map_vertex.vertex_fetch_bitmap); ++i) {
|
||||||
uint32_t vfetch_index = i * 32 + j;
|
uint32_t vfetch_bits_remaining =
|
||||||
|
constant_map_vertex.vertex_fetch_bitmap[i];
|
||||||
// Check if already in sync (validated this frame with same address/size)
|
uint32_t j;
|
||||||
uint64_t vfetch_bit = uint64_t(1) << (vfetch_index & 63);
|
while (xe::bit_scan_forward(vfetch_bits_remaining, &j)) {
|
||||||
if (vertex_buffers_in_sync_[vfetch_index >> 6] & vfetch_bit) {
|
vfetch_bits_remaining = xe::clear_lowest_bit(vfetch_bits_remaining);
|
||||||
continue;
|
uint32_t vfetch_index = i * 32 + j;
|
||||||
}
|
xenos::xe_gpu_vertex_fetch_t vfetch_constant =
|
||||||
|
regs.GetVertexFetch(vfetch_index);
|
||||||
xenos::xe_gpu_vertex_fetch_t vfetch_constant =
|
switch (vfetch_constant.type) {
|
||||||
regs.GetVertexFetch(vfetch_index);
|
case xenos::FetchConstantType::kVertex:
|
||||||
switch (vfetch_constant.type) {
|
|
||||||
case xenos::FetchConstantType::kVertex:
|
|
||||||
break;
|
|
||||||
case xenos::FetchConstantType::kInvalidVertex:
|
|
||||||
if (cvars::gpu_allow_invalid_fetch_constants) {
|
|
||||||
break;
|
break;
|
||||||
}
|
case xenos::FetchConstantType::kInvalidVertex:
|
||||||
XELOGW(
|
if (cvars::gpu_allow_invalid_fetch_constants) {
|
||||||
"Vertex fetch constant {} ({:08X} {:08X}) has \"invalid\" type! "
|
break;
|
||||||
"This "
|
}
|
||||||
"is incorrect behavior, but you can try bypassing this by "
|
|
||||||
"launching Xenia with --gpu_allow_invalid_fetch_constants=true.",
|
|
||||||
vfetch_index, vfetch_constant.dword_0, vfetch_constant.dword_1);
|
|
||||||
return false;
|
|
||||||
default:
|
|
||||||
// Type is kTexture (2) or kInvalidTexture (3) - completely wrong for
|
|
||||||
// vertex data
|
|
||||||
if (cvars::gpu_allow_invalid_fetch_constants) {
|
|
||||||
XELOGW(
|
XELOGW(
|
||||||
"Vertex fetch constant {} ({:08X} {:08X}) has wrong type {} "
|
"Vertex fetch constant {} ({:08X} {:08X}) has \"invalid\" "
|
||||||
"(texture fetch constant in vertex slot) - allowing due to "
|
"type! "
|
||||||
"--gpu_allow_invalid_fetch_constants=true. This will likely "
|
"This is incorrect behavior, but you can try bypassing this by "
|
||||||
"crash "
|
"launching Xenia with "
|
||||||
"or produce garbage!",
|
"--gpu_allow_invalid_fetch_constants=true.",
|
||||||
|
vfetch_index, vfetch_constant.dword_0, vfetch_constant.dword_1);
|
||||||
|
return false;
|
||||||
|
default:
|
||||||
|
// Type is kTexture (2) or kInvalidTexture (3) - completely wrong
|
||||||
|
// for vertex data
|
||||||
|
if (cvars::gpu_allow_invalid_fetch_constants) {
|
||||||
|
XELOGW(
|
||||||
|
"Vertex fetch constant {} ({:08X} {:08X}) has wrong type {} "
|
||||||
|
"(texture fetch constant in vertex slot) - allowing due to "
|
||||||
|
"--gpu_allow_invalid_fetch_constants=true. This will likely "
|
||||||
|
"crash or produce garbage!",
|
||||||
|
vfetch_index, vfetch_constant.dword_0,
|
||||||
|
vfetch_constant.dword_1,
|
||||||
|
static_cast<uint32_t>(vfetch_constant.type));
|
||||||
|
break;
|
||||||
|
}
|
||||||
|
XELOGW(
|
||||||
|
"Vertex fetch constant {} ({:08X} {:08X}) is completely "
|
||||||
|
"invalid! Type={} - this slot contains a texture fetch "
|
||||||
|
"constant (type 2), not a vertex fetch constant (type 0). "
|
||||||
|
"This may indicate the shader is reading from the wrong fetch "
|
||||||
|
"constant index, or the game has a bug.",
|
||||||
vfetch_index, vfetch_constant.dword_0, vfetch_constant.dword_1,
|
vfetch_index, vfetch_constant.dword_0, vfetch_constant.dword_1,
|
||||||
static_cast<uint32_t>(vfetch_constant.type));
|
static_cast<uint32_t>(vfetch_constant.type));
|
||||||
break;
|
return false;
|
||||||
}
|
}
|
||||||
XELOGW(
|
vfetch_addresses[vfetch_current_queued] = vfetch_constant.address;
|
||||||
"Vertex fetch constant {} ({:08X} {:08X}) is completely invalid! "
|
vfetch_sizes[vfetch_current_queued++] = vfetch_constant.size;
|
||||||
"Type={} - this slot contains a texture fetch constant (type 2), "
|
}
|
||||||
"not a "
|
}
|
||||||
"vertex fetch constant (type 0). This may indicate the shader is "
|
|
||||||
"reading "
|
if (vfetch_current_queued) {
|
||||||
"from the wrong fetch constant index, or the game has a bug.",
|
// Pre-acquire the critical region so we're not repeatedly re-acquiring
|
||||||
vfetch_index, vfetch_constant.dword_0, vfetch_constant.dword_1,
|
// it in RequestRange - SharedMemory tracks dirty pages and only uploads
|
||||||
static_cast<uint32_t>(vfetch_constant.type));
|
// what actually changed, making redundant calls cheap under a hoisted
|
||||||
|
// lock.
|
||||||
|
auto shared_memory_request_range_hoisted =
|
||||||
|
global_critical_region::Acquire();
|
||||||
|
|
||||||
|
for (uint32_t i = 0; i < vfetch_current_queued; ++i) {
|
||||||
|
if (!shared_memory_->RequestRange(vfetch_addresses[i] << 2,
|
||||||
|
vfetch_sizes[i] << 2)) {
|
||||||
|
XELOGE(
|
||||||
|
"Failed to request vertex buffer at 0x{:08X} (size {}) in the "
|
||||||
|
"shared memory",
|
||||||
|
vfetch_addresses[i] << 2, vfetch_sizes[i] << 2);
|
||||||
return false;
|
return false;
|
||||||
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
// Check if address/size changed - if same as cached, just mark in sync
|
|
||||||
uint32_t address = vfetch_constant.address;
|
|
||||||
uint32_t size = vfetch_constant.size;
|
|
||||||
VertexBufferState& state = vertex_buffer_states_[vfetch_index];
|
|
||||||
if (state.address == address && state.size == size) {
|
|
||||||
// Same buffer, already resident - just mark in sync
|
|
||||||
vertex_buffers_in_sync_[vfetch_index >> 6] |= vfetch_bit;
|
|
||||||
continue;
|
|
||||||
}
|
|
||||||
|
|
||||||
// New or changed buffer - need to request range
|
|
||||||
if (!shared_memory_->RequestRange(address << 2, size << 2)) {
|
|
||||||
XELOGE(
|
|
||||||
"Failed to request vertex buffer at 0x{:08X} (size {}) in the "
|
|
||||||
"shared "
|
|
||||||
"memory",
|
|
||||||
address << 2, size << 2);
|
|
||||||
return false;
|
|
||||||
}
|
|
||||||
|
|
||||||
// Update cache
|
|
||||||
state.address = address;
|
|
||||||
state.size = size;
|
|
||||||
vertex_buffers_in_sync_[vfetch_index >> 6] |= vfetch_bit;
|
|
||||||
}
|
}
|
||||||
}
|
}
|
||||||
|
|
||||||
|
|||||||
@@ -742,18 +742,6 @@ class VulkanCommandProcessor final : public CommandProcessor {
|
|||||||
uint64_t current_float_constant_map_vertex_[4];
|
uint64_t current_float_constant_map_vertex_[4];
|
||||||
uint64_t current_float_constant_map_pixel_[4];
|
uint64_t current_float_constant_map_pixel_[4];
|
||||||
|
|
||||||
// Vertex buffer residency caching - avoids redundant RequestRange calls.
|
|
||||||
// Bit is set when the vertex buffer at that index has been validated and
|
|
||||||
// is resident in shared memory with the current address/size.
|
|
||||||
uint64_t vertex_buffers_in_sync_[2] =
|
|
||||||
{}; // 96 bits for 96 vertex fetch constants
|
|
||||||
// Cached address and size for each vertex fetch constant to detect changes.
|
|
||||||
struct VertexBufferState {
|
|
||||||
uint32_t address = 0;
|
|
||||||
uint32_t size = 0;
|
|
||||||
};
|
|
||||||
std::array<VertexBufferState, 96> vertex_buffer_states_ = {};
|
|
||||||
|
|
||||||
// System shader constants.
|
// System shader constants.
|
||||||
SpirvShaderTranslator::SystemConstants system_constants_;
|
SpirvShaderTranslator::SystemConstants system_constants_;
|
||||||
|
|
||||||
|
|||||||
Reference in New Issue
Block a user