/** ****************************************************************************** * Xenia : Xbox 360 Emulator Research Project * ****************************************************************************** * Copyright 2020 Ben Vanik. All rights reserved. * * Released under the BSD license - see LICENSE in the root for more details. * ****************************************************************************** */ #ifndef XENIA_GPU_SHARED_MEMORY_H_ #define XENIA_GPU_SHARED_MEMORY_H_ #include #include #include #include "xenia/base/mutex.h" #include "xenia/memory.h" namespace xe { namespace gpu { // Manages memory for unconverted textures, resolve targets, vertex and index // buffers that can be accessed from shaders with Xenon physical addresses, with // system page size granularity. class SharedMemory { public: virtual ~SharedMemory(); // Call in the implementation-specific ClearCache. virtual void ClearCache(); typedef void (*GlobalWatchCallback)(void* context, uint32_t address_first, uint32_t address_last, bool invalidated_by_gpu); typedef void* GlobalWatchHandle; // Registers a callback invoked when something is invalidated in the GPU // memory copy by the CPU or (if triggered explicitly - such as by a resolve) // by the GPU. It will be fired for writes to pages previously requested, but // may also be fired regardless of whether it was used by GPU emulation - for // example, if the game changes protection level of a memory range containing // the watched range. // // The callback is called within the global critical region. GlobalWatchHandle RegisterGlobalWatch(GlobalWatchCallback callback, void* callback_context); void UnregisterGlobalWatch(GlobalWatchHandle handle); typedef void (*WatchCallback)(void* context, void* data, uint64_t argument, bool invalidated_by_gpu); typedef void* WatchHandle; // Registers a callback invoked when the specified memory range is invalidated // in the GPU memory copy by the CPU or (if triggered explicitly - such as by // a resolve) by the GPU. It will be fired for writes to pages previously // requested, but may also be fired regardless of whether it was used by GPU // emulation - for example, if the game changes protection level of a memory // range containing the watched range. // // Generally the context is the subsystem pointer (for example, the texture // cache), the data is the object (such as a texture), and the argument is // additional subsystem/object-specific data (such as whether the range // belongs to the base mip level or to the rest of the mips). // // Called with the global critical region locked. Do NOT watch or unwatch // ranges from within it! The watch for the callback is cancelled after the // callback - the handle becomes invalid. WatchHandle WatchMemoryRange(uint32_t start, uint32_t length, WatchCallback callback, void* callback_context, void* callback_data, uint64_t callback_argument); // Unregisters previously registered watched memory range. void UnwatchMemoryRange(WatchHandle handle); // Checks if the range has been updated, uploads new data if needed and // ensures the host GPU memory backing the range are resident. Returns true if // the range has been fully updated and is usable. bool RequestRange(uint32_t start, uint32_t length); // Marks the range and, if not exact_range, potentially its surroundings // (to up to the first GPU-written page, as an access violation exception // count optimization) as modified by the CPU, also invalidating GPU-written // pages directly in the range. std::pair MemoryInvalidationCallback( uint32_t physical_address_start, uint32_t length, bool exact_range); // Marks the range as containing GPU-generated data (such as resolves), // triggering modification callbacks, making it valid (so pages are not // copied from the main memory until they're modified by the CPU) and // protecting it. Before writing anything from the GPU side, RequestRange must // be called, to make sure, if the GPU writes don't overwrite *everything* in // the pages they touch, the CPU data is properly loaded to the unmodified // regions in those pages. void RangeWrittenByGpu(uint32_t start, uint32_t length); protected: SharedMemory(Memory& memory); // Call in implementation-specific initialization. void InitializeCommon(); // Call last in implementation-specific shutdown, also callable from the // destructor. void ShutdownCommon(); static constexpr uint32_t kBufferSizeLog2 = 29; static constexpr uint32_t kBufferSize = 1 << kBufferSizeLog2; // Sparse allocations are 4 MB, so not too many of them are allocated, but // also not to waste too much memory for padding (with 16 MB there's too // much). static constexpr uint32_t kOptimalAllocationLog2 = 22; static_assert(kOptimalAllocationLog2 <= kBufferSizeLog2); Memory& memory() const { return memory_; } uint32_t page_size_log2() const { return page_size_log2_; } // Mark the memory range as updated and protect it. void MakeRangeValid(uint32_t start, uint32_t length, bool written_by_gpu); // Ensures the host GPU memory backing the range is accessible by host GPU // drawing / computations / copying, but doesn't upload anything. virtual bool EnsureHostGpuMemoryAllocated(uint32_t start, uint32_t length) = 0; // Uploads a range of host pages - only called if EnsureHostGpuMemoryAllocated // succeeded. While uploading, MarkRangeValid must be called for each // successfully uploaded range as early as possible, before the memcpy, to // make sure invalidation that happened during the CPU -> GPU memcpy isn't // missed (upload_page_ranges is in pages because of this - MarkRangeValid has // page granularity). virtual bool UploadRanges( const std::vector>& upload_page_ranges) = 0; // Mutable so the implementation can skip ranges by setting their "second" // value to 0 if needed. std::vector>& trace_download_ranges() { return trace_download_ranges_; } uint32_t trace_download_page_count() const { return trace_download_page_count_; } // Fills trace_download_ranges() and trace_download_page_count() with // GPU-written ranges that need to be downloaded, and also invalidates // non-GPU-written ranges so only the needed data - not the all the collected // data - will be written in the trace. trace_download_page_count() will be 0 // if nothing to download. void PrepareForTraceDownload(); // Release memory used for trace download ranges, to be called after // downloading or in cases when download is dropped. void ReleaseTraceDownloadRanges(); private: Memory& memory_; // Log2 of invalidation granularity (the system page size, but the dependency // on it is not hard - the access callback takes a range as an argument, and // touched pages of the buffer of this size will be invalidated). uint32_t page_size_log2_; void* memory_invalidation_callback_handle_ = nullptr; void* memory_data_provider_handle_ = nullptr; // Ranges that need to be uploaded, generated by GetRangesToUpload (a // persistently allocated vector). std::vector> upload_ranges_; // GPU-written memory downloading for traces. . std::vector> trace_download_ranges_; uint32_t trace_download_page_count_ = 0; // Mutex between the guest memory subsystem and the command processor, to be // locked when checking or updating validity of pages/ranges and when firing // watches. xe::global_critical_region global_critical_region_; // *************************************************************************** // Things below should be fully protected by global_critical_region. // *************************************************************************** struct SystemPageFlagsBlock { // Whether each page is up to date in the GPU buffer. uint64_t valid; // Subset of valid pages - whether each page in the GPU buffer contains data // that was written on the GPU, thus should not be invalidated spuriously. uint64_t valid_and_gpu_written; }; // Flags for each 64 system pages, interleaved as blocks, so bit scan can be // used to quickly extract ranges. std::vector system_page_flags_; static std::pair MemoryInvalidationCallbackThunk( void* context_ptr, uint32_t physical_address_start, uint32_t length, bool exact_range); struct GlobalWatch { GlobalWatchCallback callback; void* callback_context; }; std::vector global_watches_; struct WatchNode; // Watched range placed by other GPU subsystems. struct WatchRange { union { struct { WatchCallback callback; void* callback_context; void* callback_data; uint64_t callback_argument; WatchNode* node_first; uint32_t page_first; uint32_t page_last; }; WatchRange* next_free; }; }; // Node for faster checking of watches when pages have been written to - all // 512 MB are split into smaller equally sized buckets, and then ranges are // linearly checked. struct WatchNode { union { struct { WatchRange* range; // Link to another node of this watched range in the next bucket. WatchNode* range_node_next; // Links to nodes belonging to other watched ranges in the bucket. WatchNode* bucket_node_previous; WatchNode* bucket_node_next; }; WatchNode* next_free; }; }; static constexpr uint32_t kWatchBucketSizeLog2 = 22; static constexpr uint32_t kWatchBucketCount = 1 << (kBufferSizeLog2 - kWatchBucketSizeLog2); WatchNode* watch_buckets_[kWatchBucketCount] = {}; // Allocation from pools - taking new WatchRanges and WatchNodes from the free // list, and if there are none, creating a pool if the current one is fully // used, and linearly allocating from the current pool. static constexpr uint32_t kWatchRangePoolSize = 8192; static constexpr uint32_t kWatchNodePoolSize = 8192; std::vector watch_range_pools_; std::vector watch_node_pools_; uint32_t watch_range_current_pool_allocated_ = 0; uint32_t watch_node_current_pool_allocated_ = 0; WatchRange* watch_range_first_free_ = nullptr; WatchNode* watch_node_first_free_ = nullptr; // Triggers the watches (global and per-range), removing triggered range // watches. void FireWatches(uint32_t page_first, uint32_t page_last, bool invalidated_by_gpu); // Unlinks and frees the range and its nodes. Call this in the global critical // region. void UnlinkWatchRange(WatchRange* range); }; } // namespace gpu } // namespace xe #endif // XENIA_GPU_SHARED_MEMORY_H_