[GPU] Split SharedMemory into common and D3D12
This commit is contained in:
255
src/xenia/gpu/shared_memory.h
Normal file
255
src/xenia/gpu/shared_memory.h
Normal file
@@ -0,0 +1,255 @@
|
||||
/**
|
||||
******************************************************************************
|
||||
* Xenia : Xbox 360 Emulator Research Project *
|
||||
******************************************************************************
|
||||
* Copyright 2020 Ben Vanik. All rights reserved. *
|
||||
* Released under the BSD license - see LICENSE in the root for more details. *
|
||||
******************************************************************************
|
||||
*/
|
||||
|
||||
#ifndef XENIA_GPU_SHARED_MEMORY_H_
|
||||
#define XENIA_GPU_SHARED_MEMORY_H_
|
||||
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <vector>
|
||||
|
||||
#include "xenia/base/mutex.h"
|
||||
#include "xenia/memory.h"
|
||||
|
||||
namespace xe {
|
||||
namespace gpu {
|
||||
|
||||
// Manages memory for unconverted textures, resolve targets, vertex and index
|
||||
// buffers that can be accessed from shaders with Xenon physical addresses, with
|
||||
// system page size granularity.
|
||||
class SharedMemory {
|
||||
public:
|
||||
virtual ~SharedMemory();
|
||||
// Call in the implementation-specific ClearCache.
|
||||
virtual void ClearCache();
|
||||
|
||||
typedef void (*GlobalWatchCallback)(void* context, uint32_t address_first,
|
||||
uint32_t address_last,
|
||||
bool invalidated_by_gpu);
|
||||
typedef void* GlobalWatchHandle;
|
||||
// Registers a callback invoked when something is invalidated in the GPU
|
||||
// memory copy by the CPU or (if triggered explicitly - such as by a resolve)
|
||||
// by the GPU. It will be fired for writes to pages previously requested, but
|
||||
// may also be fired regardless of whether it was used by GPU emulation - for
|
||||
// example, if the game changes protection level of a memory range containing
|
||||
// the watched range.
|
||||
//
|
||||
// The callback is called within the global critical region.
|
||||
GlobalWatchHandle RegisterGlobalWatch(GlobalWatchCallback callback,
|
||||
void* callback_context);
|
||||
void UnregisterGlobalWatch(GlobalWatchHandle handle);
|
||||
typedef void (*WatchCallback)(void* context, void* data, uint64_t argument,
|
||||
bool invalidated_by_gpu);
|
||||
typedef void* WatchHandle;
|
||||
// Registers a callback invoked when the specified memory range is invalidated
|
||||
// in the GPU memory copy by the CPU or (if triggered explicitly - such as by
|
||||
// a resolve) by the GPU. It will be fired for writes to pages previously
|
||||
// requested, but may also be fired regardless of whether it was used by GPU
|
||||
// emulation - for example, if the game changes protection level of a memory
|
||||
// range containing the watched range.
|
||||
//
|
||||
// Generally the context is the subsystem pointer (for example, the texture
|
||||
// cache), the data is the object (such as a texture), and the argument is
|
||||
// additional subsystem/object-specific data (such as whether the range
|
||||
// belongs to the base mip level or to the rest of the mips).
|
||||
//
|
||||
// Called with the global critical region locked. Do NOT watch or unwatch
|
||||
// ranges from within it! The watch for the callback is cancelled after the
|
||||
// callback - the handle becomes invalid.
|
||||
WatchHandle WatchMemoryRange(uint32_t start, uint32_t length,
|
||||
WatchCallback callback, void* callback_context,
|
||||
void* callback_data, uint64_t callback_argument);
|
||||
// Unregisters previously registered watched memory range.
|
||||
void UnwatchMemoryRange(WatchHandle handle);
|
||||
|
||||
// Checks if the range has been updated, uploads new data if needed and
|
||||
// ensures the host GPU memory backing the range are resident. Returns true if
|
||||
// the range has been fully updated and is usable.
|
||||
bool RequestRange(uint32_t start, uint32_t length);
|
||||
|
||||
// Marks the range and, if not exact_range, potentially its surroundings
|
||||
// (to up to the first GPU-written page, as an access violation exception
|
||||
// count optimization) as modified by the CPU, also invalidating GPU-written
|
||||
// pages directly in the range.
|
||||
std::pair<uint32_t, uint32_t> MemoryInvalidationCallback(
|
||||
uint32_t physical_address_start, uint32_t length, bool exact_range);
|
||||
|
||||
// Marks the range as containing GPU-generated data (such as resolves),
|
||||
// triggering modification callbacks, making it valid (so pages are not
|
||||
// copied from the main memory until they're modified by the CPU) and
|
||||
// protecting it. Before writing anything from the GPU side, RequestRange must
|
||||
// be called, to make sure, if the GPU writes don't overwrite *everything* in
|
||||
// the pages they touch, the CPU data is properly loaded to the unmodified
|
||||
// regions in those pages.
|
||||
void RangeWrittenByGpu(uint32_t start, uint32_t length);
|
||||
|
||||
protected:
|
||||
SharedMemory(Memory& memory);
|
||||
// Call in implementation-specific initialization.
|
||||
void InitializeCommon();
|
||||
// Call last in implementation-specific shutdown, also callable from the
|
||||
// destructor.
|
||||
void ShutdownCommon();
|
||||
|
||||
static constexpr uint32_t kBufferSizeLog2 = 29;
|
||||
static constexpr uint32_t kBufferSize = 1 << kBufferSizeLog2;
|
||||
|
||||
// Sparse allocations are 4 MB, so not too many of them are allocated, but
|
||||
// also not to waste too much memory for padding (with 16 MB there's too
|
||||
// much).
|
||||
static constexpr uint32_t kOptimalAllocationLog2 = 22;
|
||||
static_assert(kOptimalAllocationLog2 <= kBufferSizeLog2);
|
||||
|
||||
Memory& memory() const { return memory_; }
|
||||
|
||||
uint32_t page_size_log2() const { return page_size_log2_; }
|
||||
|
||||
// Mark the memory range as updated and protect it.
|
||||
void MakeRangeValid(uint32_t start, uint32_t length, bool written_by_gpu);
|
||||
|
||||
// Ensures the host GPU memory backing the range is accessible by host GPU
|
||||
// drawing / computations / copying, but doesn't upload anything.
|
||||
virtual bool EnsureHostGpuMemoryAllocated(uint32_t start,
|
||||
uint32_t length) = 0;
|
||||
|
||||
// Uploads a range of host pages - only called if EnsureHostGpuMemoryAllocated
|
||||
// succeeded. While uploading, MarkRangeValid must be called for each
|
||||
// successfully uploaded range as early as possible, before the memcpy, to
|
||||
// make sure invalidation that happened during the CPU -> GPU memcpy isn't
|
||||
// missed (upload_page_ranges is in pages because of this - MarkRangeValid has
|
||||
// page granularity).
|
||||
virtual bool UploadRanges(
|
||||
const std::vector<std::pair<uint32_t, uint32_t>>& upload_page_ranges) = 0;
|
||||
|
||||
// Mutable so the implementation can skip ranges by setting their "second"
|
||||
// value to 0 if needed.
|
||||
std::vector<std::pair<uint32_t, uint32_t>>& trace_download_ranges() {
|
||||
return trace_download_ranges_;
|
||||
}
|
||||
uint32_t trace_download_page_count() const {
|
||||
return trace_download_page_count_;
|
||||
}
|
||||
// Fills trace_download_ranges() and trace_download_page_count() with
|
||||
// GPU-written ranges that need to be downloaded, and also invalidates
|
||||
// non-GPU-written ranges so only the needed data - not the all the collected
|
||||
// data - will be written in the trace. trace_download_page_count() will be 0
|
||||
// if nothing to download.
|
||||
void PrepareForTraceDownload();
|
||||
// Release memory used for trace download ranges, to be called after
|
||||
// downloading or in cases when download is dropped.
|
||||
void ReleaseTraceDownloadRanges();
|
||||
|
||||
private:
|
||||
Memory& memory_;
|
||||
|
||||
// Log2 of invalidation granularity (the system page size, but the dependency
|
||||
// on it is not hard - the access callback takes a range as an argument, and
|
||||
// touched pages of the buffer of this size will be invalidated).
|
||||
uint32_t page_size_log2_;
|
||||
|
||||
void* memory_invalidation_callback_handle_ = nullptr;
|
||||
void* memory_data_provider_handle_ = nullptr;
|
||||
|
||||
// Ranges that need to be uploaded, generated by GetRangesToUpload (a
|
||||
// persistently allocated vector).
|
||||
std::vector<std::pair<uint32_t, uint32_t>> upload_ranges_;
|
||||
|
||||
// GPU-written memory downloading for traces. <Start address, length>.
|
||||
std::vector<std::pair<uint32_t, uint32_t>> trace_download_ranges_;
|
||||
uint32_t trace_download_page_count_ = 0;
|
||||
|
||||
// Mutex between the guest memory subsystem and the command processor, to be
|
||||
// locked when checking or updating validity of pages/ranges and when firing
|
||||
// watches.
|
||||
xe::global_critical_region global_critical_region_;
|
||||
|
||||
// ***************************************************************************
|
||||
// Things below should be fully protected by global_critical_region.
|
||||
// ***************************************************************************
|
||||
|
||||
struct SystemPageFlagsBlock {
|
||||
// Whether each page is up to date in the GPU buffer.
|
||||
uint64_t valid;
|
||||
// Subset of valid pages - whether each page in the GPU buffer contains data
|
||||
// that was written on the GPU, thus should not be invalidated spuriously.
|
||||
uint64_t valid_and_gpu_written;
|
||||
};
|
||||
// Flags for each 64 system pages, interleaved as blocks, so bit scan can be
|
||||
// used to quickly extract ranges.
|
||||
std::vector<SystemPageFlagsBlock> system_page_flags_;
|
||||
|
||||
static std::pair<uint32_t, uint32_t> MemoryInvalidationCallbackThunk(
|
||||
void* context_ptr, uint32_t physical_address_start, uint32_t length,
|
||||
bool exact_range);
|
||||
|
||||
struct GlobalWatch {
|
||||
GlobalWatchCallback callback;
|
||||
void* callback_context;
|
||||
};
|
||||
std::vector<GlobalWatch*> global_watches_;
|
||||
struct WatchNode;
|
||||
// Watched range placed by other GPU subsystems.
|
||||
struct WatchRange {
|
||||
union {
|
||||
struct {
|
||||
WatchCallback callback;
|
||||
void* callback_context;
|
||||
void* callback_data;
|
||||
uint64_t callback_argument;
|
||||
WatchNode* node_first;
|
||||
uint32_t page_first;
|
||||
uint32_t page_last;
|
||||
};
|
||||
WatchRange* next_free;
|
||||
};
|
||||
};
|
||||
// Node for faster checking of watches when pages have been written to - all
|
||||
// 512 MB are split into smaller equally sized buckets, and then ranges are
|
||||
// linearly checked.
|
||||
struct WatchNode {
|
||||
union {
|
||||
struct {
|
||||
WatchRange* range;
|
||||
// Link to another node of this watched range in the next bucket.
|
||||
WatchNode* range_node_next;
|
||||
// Links to nodes belonging to other watched ranges in the bucket.
|
||||
WatchNode* bucket_node_previous;
|
||||
WatchNode* bucket_node_next;
|
||||
};
|
||||
WatchNode* next_free;
|
||||
};
|
||||
};
|
||||
static constexpr uint32_t kWatchBucketSizeLog2 = 22;
|
||||
static constexpr uint32_t kWatchBucketCount =
|
||||
1 << (kBufferSizeLog2 - kWatchBucketSizeLog2);
|
||||
WatchNode* watch_buckets_[kWatchBucketCount] = {};
|
||||
// Allocation from pools - taking new WatchRanges and WatchNodes from the free
|
||||
// list, and if there are none, creating a pool if the current one is fully
|
||||
// used, and linearly allocating from the current pool.
|
||||
static constexpr uint32_t kWatchRangePoolSize = 8192;
|
||||
static constexpr uint32_t kWatchNodePoolSize = 8192;
|
||||
std::vector<WatchRange*> watch_range_pools_;
|
||||
std::vector<WatchNode*> watch_node_pools_;
|
||||
uint32_t watch_range_current_pool_allocated_ = 0;
|
||||
uint32_t watch_node_current_pool_allocated_ = 0;
|
||||
WatchRange* watch_range_first_free_ = nullptr;
|
||||
WatchNode* watch_node_first_free_ = nullptr;
|
||||
// Triggers the watches (global and per-range), removing triggered range
|
||||
// watches.
|
||||
void FireWatches(uint32_t page_first, uint32_t page_last,
|
||||
bool invalidated_by_gpu);
|
||||
// Unlinks and frees the range and its nodes. Call this in the global critical
|
||||
// region.
|
||||
void UnlinkWatchRange(WatchRange* range);
|
||||
};
|
||||
|
||||
} // namespace gpu
|
||||
} // namespace xe
|
||||
|
||||
#endif // XENIA_GPU_SHARED_MEMORY_H_
|
||||
Reference in New Issue
Block a user