Files
xenia-rs/crates/xenia-gpu/src/prof.rs
MechaCat02 2c883e9d5e [iterate-4A] diagnostics: XENIA_PROFILE wall-time profiler + probe/tooling snapshot
Handoff snapshot of the env-gated diagnostic scaffolding used across the
intro-video RE. Kept out of the milestone commits (645feb8..5573ac1) to keep
those clean; committed here so nothing is lost on handoff.

New — XENIA_PROFILE wall-time profiler (crates/xenia-gpu/src/prof.rs):
  Coarse buckets attributing playback wall time to interpreter (step_block),
  kernel HLE (call_export), block decode/cache (lookup_or_build), texture
  decode, host draw, and present; prints periodic snapshots (every 500M guest
  instr, or every 500 presents) + a clean-exit report. Hot path is gated on a
  cached is_on() (one relaxed load) so it is zero-cost when XENIA_PROFILE is
  unset. Call sites: main.rs run_superblock / parallel worker (step_block,
  lookup_or_build, call_export), texture_cache ensure_cached, render.rs present
  + dispatch_xenos_draws.

  First profile (movie playback, headless single-thread lockstep): effective
  ~35 MIPS; interpreter body ~40% @ ~95-102 MIPS; texture decode 0.3% (cache
  works); present ~0%; the rest is per-block dispatch + scheduler plumbing
  (~13 instr/block over 229M blocks). Overhead-bound, not interpreter-body
  bound; the levers are coarser execution units (superblock chaining) and
  ultimately a JIT.

Pre-existing read-only probe knobs (were uncommitted; env-gated, observe-only):
  XENIA_RET_CAPTURE_PC/_REG/_MEM, LOG_RESUMES, LOG_WAITS, LOG_SIGNAL,
  FORCE_TID, STARVE_LIMIT, INCUMBENT_PICK, INSTR_PER_MS, DUMP_FRAME,
  DUMP_WGSL, BIND_LOG, CONST_LOG, DISPATCH_REC, AUDIT_PC_TRACE.

Tooling: sylph-run.sh (movie oracle loop, 180s default timeout),
  zq.py (DuckDB disasm/xref helper).

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
2026-07-03 07:43:53 +02:00

194 lines
6.2 KiB
Rust

//! Lightweight env-gated wall-time profiler (probe-patch, UNCOMMITTED).
//!
//! Attributes emulator wall time to coarse buckets so we can tell whether the
//! movie-playback slowdown is CPU-interpreter bound, texture-decode bound, or
//! GPU-present bound. Enabled only when `XENIA_PROFILE` is set; the hot-path
//! cost when disabled is a single relaxed atomic add of already-measured nanos
//! (callers still pay `Instant::now()` — acceptable at the coarse boundaries we
//! instrument: per basic-block, per texture upload, per present).
//!
//! Read the buckets with `xenia_gpu::prof::report(wall_ns)` at clean shutdown.
use std::sync::atomic::{AtomicU64, Ordering};
use std::sync::OnceLock;
static START: OnceLock<std::time::Instant> = OnceLock::new();
/// Lazily anchor the wall-time window (first call wins). Called from `add`.
#[inline]
fn mark_start() {
START.get_or_init(std::time::Instant::now);
}
/// Nanos since the profiler's first accounted event.
pub fn wall_ns() -> u64 {
START.get().map(|t| t.elapsed().as_nanos() as u64).unwrap_or(0)
}
pub static STEP_NS: AtomicU64 = AtomicU64::new(0); // guest interpreter (step_block)
pub static STEP_INSTR: AtomicU64 = AtomicU64::new(0); // guest instructions retired
pub static STEP_CALLS: AtomicU64 = AtomicU64::new(0);
pub static TEXDEC_NS: AtomicU64 = AtomicU64::new(0); // texture decode + host upload
pub static TEXDEC_CALLS: AtomicU64 = AtomicU64::new(0);
pub static TEXDEC_BYTES: AtomicU64 = AtomicU64::new(0);
pub static PRESENT_NS: AtomicU64 = AtomicU64::new(0); // frontbuffer present
pub static PRESENT_CALLS: AtomicU64 = AtomicU64::new(0);
pub static DRAW_NS: AtomicU64 = AtomicU64::new(0); // host draw submission
pub static DRAW_CALLS: AtomicU64 = AtomicU64::new(0);
pub static KERNEL_NS: AtomicU64 = AtomicU64::new(0); // kernel HLE export dispatch
pub static KERNEL_CALLS: AtomicU64 = AtomicU64::new(0);
pub static BUILD_NS: AtomicU64 = AtomicU64::new(0); // block decode / cache lookup
pub static BUILD_CALLS: AtomicU64 = AtomicU64::new(0);
/// Cached on/off state so the per-block hot path never touches the
/// environment. 0 = uninitialised, 1 = on, 2 = off.
static ENABLED: std::sync::atomic::AtomicU8 = std::sync::atomic::AtomicU8::new(0);
/// Cheap (one relaxed load + branch) enabled check for hot paths. Resolves
/// `XENIA_PROFILE` from the environment exactly once, then caches it.
#[inline]
pub fn is_on() -> bool {
match ENABLED.load(Ordering::Relaxed) {
1 => true,
2 => false,
_ => {
let on = std::env::var_os("XENIA_PROFILE").is_some();
ENABLED.store(if on { 1 } else { 2 }, Ordering::Relaxed);
on
}
}
}
#[inline]
pub fn enabled() -> bool {
is_on()
}
#[inline]
pub fn add(counter: &AtomicU64, v: u64) {
mark_start();
counter.fetch_add(v, Ordering::Relaxed);
}
/// RAII timer: adds elapsed nanos to `ns` and bumps `calls` on drop.
pub struct ScopeTimer {
t0: std::time::Instant,
ns: &'static AtomicU64,
calls: &'static AtomicU64,
}
impl ScopeTimer {
#[inline]
pub fn new(ns: &'static AtomicU64, calls: &'static AtomicU64) -> Self {
mark_start();
Self { t0: std::time::Instant::now(), ns, calls }
}
}
impl Drop for ScopeTimer {
#[inline]
fn drop(&mut self) {
self.ns.fetch_add(self.t0.elapsed().as_nanos() as u64, Ordering::Relaxed);
self.calls.fetch_add(1, Ordering::Relaxed);
}
}
static NEXT_REPORT_INSTR: AtomicU64 = AtomicU64::new(500_000_000);
/// Fire a snapshot every ~500M retired guest instructions (headless runs have
/// no present() to piggyback on and may never reach the clean-exit report).
#[inline]
pub fn maybe_report_by_instr() {
if !enabled() {
return;
}
let instr = STEP_INSTR.load(Ordering::Relaxed);
let thresh = NEXT_REPORT_INSTR.load(Ordering::Relaxed);
if instr >= thresh
&& NEXT_REPORT_INSTR
.compare_exchange(thresh, thresh + 500_000_000, Ordering::Relaxed, Ordering::Relaxed)
.is_ok()
{
report(0);
}
}
/// Print the accumulated buckets against the profiler's own wall window.
/// The argument is accepted for call-site convenience but ignored in favour
/// of the internally-anchored window (`wall_ns()`).
pub fn report(_ignored: u64) {
let wall_ns = wall_ns();
let g = |c: &AtomicU64| c.load(Ordering::Relaxed);
let ms = |ns: u64| ns as f64 / 1e6;
let pct = |ns: u64| {
if wall_ns == 0 {
0.0
} else {
100.0 * ns as f64 / wall_ns as f64
}
};
let step_ns = g(&STEP_NS);
let step_instr = g(&STEP_INSTR);
let tex_ns = g(&TEXDEC_NS);
let pres_ns = g(&PRESENT_NS);
let draw_ns = g(&DRAW_NS);
let mips = if step_ns == 0 {
0.0
} else {
step_instr as f64 / (step_ns as f64 / 1e3) // instr / us = MIPS
};
eprintln!("=== XENIA_PROFILE (wall {:.1} ms) ===", ms(wall_ns));
eprintln!(
" interp step_block : {:>10.1} ms {:>5.1}% ({} calls, {} instr, {:.1} MIPS)",
ms(step_ns),
pct(step_ns),
g(&STEP_CALLS),
step_instr,
mips
);
eprintln!(
" texture decode+up : {:>10.1} ms {:>5.1}% ({} calls, {} MiB)",
ms(tex_ns),
pct(tex_ns),
g(&TEXDEC_CALLS),
g(&TEXDEC_BYTES) / (1024 * 1024)
);
eprintln!(
" host draw submit : {:>10.1} ms {:>5.1}% ({} calls)",
ms(draw_ns),
pct(draw_ns),
g(&DRAW_CALLS)
);
let kern_ns = g(&KERNEL_NS);
let build_ns = g(&BUILD_NS);
eprintln!(
" kernel HLE export : {:>10.1} ms {:>5.1}% ({} calls)",
ms(kern_ns),
pct(kern_ns),
g(&KERNEL_CALLS)
);
eprintln!(
" block decode/cache: {:>10.1} ms {:>5.1}% ({} calls)",
ms(build_ns),
pct(build_ns),
g(&BUILD_CALLS)
);
eprintln!(
" frontbuffer present: {:>9.1} ms {:>5.1}% ({} calls)",
ms(pres_ns),
pct(pres_ns),
g(&PRESENT_CALLS)
);
let accounted = step_ns + tex_ns + draw_ns + pres_ns + kern_ns + build_ns;
eprintln!(
" ---- accounted {:.1}% ; remainder (locks/kernel/scheduler/idle) {:.1}%",
pct(accounted),
pct(wall_ns.saturating_sub(accounted))
);
}