diff --git a/crates/xenia-app/src/main.rs b/crates/xenia-app/src/main.rs index 8b53a3d..f672bb5 100644 --- a/crates/xenia-app/src/main.rs +++ b/crates/xenia-app/src/main.rs @@ -2479,6 +2479,10 @@ struct WorkerCtx { hw_id: u8, block_cache: xenia_cpu::block_cache::BlockCache, decode_cache: xenia_cpu::decoder::DecodeCache, + /// Compiled-block cache (Cranelift JIT). Keyed identically to `block_cache` + /// on `(start_pc, page_version)`; owns the JIT module and thus the native + /// code. Only consulted on the `XENIA_JIT` / `XENIA_JIT_DIFF` paths. + jit_cache: xenia_cpu::jit::JitCache, force_per_instr: bool, } @@ -2488,6 +2492,7 @@ impl WorkerCtx { hw_id, block_cache: xenia_cpu::block_cache::BlockCache::new(), decode_cache: xenia_cpu::decoder::DecodeCache::new(), + jit_cache: xenia_cpu::jit::JitCache::new(), force_per_instr, } } @@ -3145,9 +3150,9 @@ fn run_superblock( // - XENIA_JIT: JIT authoritative (M0 = interpreter fallback), // - else: interpreter directly. if xenia_cpu::recompiler::diff_enabled() { - xenia_cpu::recompiler::diff_step(ctx, mem, block) + xenia_cpu::recompiler::diff_step(ctx, mem, block, &mut wc.jit_cache) } else if xenia_cpu::recompiler::jit_enabled() { - xenia_cpu::recompiler::run_block(ctx, mem, block) + xenia_cpu::recompiler::run_block(ctx, mem, block, &mut wc.jit_cache) } else { step_block(ctx, mem, block) } @@ -4396,6 +4401,8 @@ fn dump_thread_diagnostic( } // JIT differential harness: checked/skipped/mismatch tally at clean exit. xenia_cpu::recompiler::report_diff_summary(); + // JIT native-vs-interpreted block coverage at clean exit. + xenia_cpu::recompiler::report_jit_summary(); // JIT opcode-frequency histogram (XENIA_JIT_HIST) — drives M1 coverage. xenia_cpu::recompiler::report_histogram(); diff --git a/crates/xenia-cpu/src/jit.rs b/crates/xenia-cpu/src/jit.rs index 81d78b2..a6ab1d9 100644 --- a/crates/xenia-cpu/src/jit.rs +++ b/crates/xenia-cpu/src/jit.rs @@ -21,6 +21,7 @@ use core::mem::offset_of; +use cranelift_codegen::ir::condcodes::IntCC; use cranelift_codegen::ir::{types, AbiParam, InstBuilder, MemFlags, Value}; use cranelift_frontend::{FunctionBuilder, FunctionBuilderContext}; use cranelift_jit::{JITBuilder, JITModule}; @@ -47,10 +48,87 @@ pub struct MemEnv<'a> { /// [`Jit`]'s module lives (the module owns the executable memory). pub type CompiledFn = extern "C" fn(*mut PpcContext, *const MemEnv) -> u32; +// ---- compiled-block cache ------------------------------------------------- + +/// Direct-mapped compiled-block cache — same slot geometry and `(start_pc, +/// page_version)` keying as [`crate::block_cache::BlockCache`], so a +/// self-modifying / DMA'd code page invalidates compiled code the same way it +/// invalidates the decoded block (both re-key on the bumped `page_version`). +const JIT_CACHE_SIZE: usize = 1 << 16; +const JIT_CACHE_MASK: u32 = (JIT_CACHE_SIZE - 1) as u32; + +/// One compiled-cache slot. `func == None` records "this block is **uncovered** +/// (or failed to compile) — don't retry", so an uncovered block is compile- +/// attempted at most once per `(pc, page_version)` instead of every visit. +struct CompiledSlot { + start_pc: u32, + page_version: u64, + func: Option, +} + +/// A [`Jit`] plus a direct-mapped cache of its compiled blocks. One instance +/// lives per worker slot (like `BlockCache`); the owned [`Jit`] module keeps +/// every [`CompiledFn`] it hands out valid for the cache's lifetime. +pub struct JitCache { + /// `None` only if Cranelift failed to initialise on this host — then every + /// lookup misses and the caller interprets (no JIT available). + jit: Option, + slots: Box<[Option]>, +} + +impl Default for JitCache { + fn default() -> Self { + Self::new() + } +} + +impl JitCache { + pub fn new() -> Self { + let mut slots: Vec> = Vec::with_capacity(JIT_CACHE_SIZE); + slots.resize_with(JIT_CACHE_SIZE, || None); + Self { jit: Jit::new().ok(), slots: slots.into_boxed_slice() } + } + + /// Return the compiled entry point for `block`, compiling it on the first + /// visit. `None` means the block is uncovered (or compilation failed) and + /// the caller must interpret it. The `None` verdict is cached per + /// `(start_pc, page_version)` so uncovered blocks aren't re-attempted. + pub fn get_or_compile(&mut self, block: &DecodedBlock) -> Option { + let idx = ((block.start_pc >> 2) & JIT_CACHE_MASK) as usize; + if let Some(slot) = &self.slots[idx] { + if slot.start_pc == block.start_pc && slot.page_version == block.page_version { + return slot.func; + } + } + let func = self.jit.as_mut().and_then(|j| j.compile(block)); + self.slots[idx] = Some(CompiledSlot { + start_pc: block.start_pc, + page_version: block.page_version, + func, + }); + func + } +} + /// Is this opcode currently lowerable to native code? Must mirror [`emit_op`]. /// A block is JIT-compiled only if `covered` is true for all its instructions. +/// +/// Covered so far: `addi`/`addis` (integer immediate add) and the three direct +/// branch terminators `bx`/`bcx`/`bclrx`. `bcctrx` is deliberately excluded — +/// it targets an indirect address and runs the `dispatch_rec` diagnostic hook, +/// which the native path would silently skip. Every branch computes its target +/// from immediates (or the live `lr` for `bclrx`) exactly as the interpreter. pub fn covered(instr: &DecodedInstr) -> bool { - matches!(instr.opcode, PpcOpcode::addi | PpcOpcode::addis) + matches!( + instr.opcode, + PpcOpcode::addi | PpcOpcode::addis | PpcOpcode::bx | PpcOpcode::bcx | PpcOpcode::bclrx + ) +} + +/// Does this opcode write `pc` itself (a branch), so the block epilogue must +/// **not** overwrite `pc` with `end_pc`? Only ever true for the terminator. +fn writes_pc(instr: &DecodedInstr) -> bool { + matches!(instr.opcode, PpcOpcode::bx | PpcOpcode::bcx | PpcOpcode::bclrx) } /// True if the whole block can be compiled (all opcodes covered). @@ -101,9 +179,14 @@ impl Jit { emit_op(&mut b, ctxp, instr); } - // Straight-line covered block: PC advances linearly to end_pc. - let end_pc = b.ins().iconst(types::I32, block.end_pc as i64); - b.ins().store(MemFlags::trusted(), end_pc, ctxp, off(offset_of!(PpcContext, pc))); + // A branch terminator writes `pc` itself (to its computed target); + // otherwise the block is straight-line and `pc` advances to + // `end_pc`. `writes_pc` is only ever true for the last instruction + // (branches terminate the block), so testing it is sufficient. + if !block.instrs.last().is_some_and(writes_pc) { + let end_pc = b.ins().iconst(types::I32, block.end_pc as i64); + b.ins().store(MemFlags::trusted(), end_pc, ctxp, off(offset_of!(PpcContext, pc))); + } // cycle_count += N ; timebase += N (one per executed instruction). let n = block.instrs.len() as i64; @@ -156,6 +239,65 @@ fn bump_u64(b: &mut FunctionBuilder, ctxp: Value, offset: usize, by: i64) { b.ins().store(MemFlags::trusted(), next, ctxp, off(offset)); } +#[inline] +fn store_pc(b: &mut FunctionBuilder, ctxp: Value, val: Value) { + b.ins().store(MemFlags::trusted(), val, ctxp, off(offset_of!(PpcContext, pc))); +} + +#[inline] +fn store_lr(b: &mut FunctionBuilder, ctxp: Value, val_u64: u64) { + let v = b.ins().iconst(types::I64, val_u64 as i64); + b.ins().store(MemFlags::trusted(), v, ctxp, off(offset_of!(PpcContext, lr))); +} + +/// Load the single CR bit `bi` (0-31) as an `I8` boolean (0/1). CR fields are +/// `#[repr] struct CrField { lt, gt, eq, so }` — one byte each — so bit +/// `bi = field*4 + sub` is the byte at `cr + field*4 + sub`. +#[inline] +fn emit_cr_bit(b: &mut FunctionBuilder, ctxp: Value, bi: u32) -> Value { + let field = (bi / 4) as usize; + let sub = (bi % 4) as usize; + let offset = (offset_of!(PpcContext, cr) + field * 4 + sub) as i32; + b.ins().load(types::I8, MemFlags::trusted(), ctxp, offset) +} + +/// Emit the shared `bcx`/`bclrx` "branch taken?" predicate, mirroring the +/// interpreter exactly: optionally decrement CTR, then `ctr_ok && cond_ok`. +/// Returns an `I8` (0/1). BO/BI are immediates, so the constant sub-cases +/// (`branch always`, `no CTR`, `ignore CR`) fold away at compile time. +fn emit_branch_taken(b: &mut FunctionBuilder, ctxp: Value, bo: u32, bi: u32) -> Value { + let ctr_off = off(offset_of!(PpcContext, ctr)); + + // ctr_ok. BO2 (bo & 0b00100): 1 = don't test CTR. Else decrement CTR and + // test (ctr!=0) possibly inverted by BO3 (bo & 0b00010). + let ctr_ok = if bo & 0b00100 != 0 { + b.ins().iconst(types::I8, 1) + } else { + let cur = b.ins().load(types::I64, MemFlags::trusted(), ctxp, ctr_off); + let dec = b.ins().iadd_imm(cur, -1); + b.ins().store(MemFlags::trusted(), dec, ctxp, ctr_off); + let lo = b.ins().ireduce(types::I32, dec); + // interpreter: (ctr as u32 != 0) ^ (bo&2 != 0) + let cc = if bo & 0b00010 != 0 { IntCC::Equal } else { IntCC::NotEqual }; + b.ins().icmp_imm(cc, lo, 0) + }; + + // cond_ok. BO0 (bo & 0b10000): 1 = don't test CR. Else CR bit BI must equal + // BO1 (bo & 0b01000). + let cond_ok = if bo & 0b10000 != 0 { + b.ins().iconst(types::I8, 1) + } else { + let crbit = emit_cr_bit(b, ctxp, bi); + if bo & 0b01000 != 0 { + crbit + } else { + b.ins().icmp_imm(IntCC::Equal, crbit, 0) + } + }; + + b.ins().band(ctr_ok, cond_ok) +} + /// Emit IR for one covered instruction. Must mirror [`covered`] and the exact /// semantics of the interpreter's `execute()` (validated by the diff harness). fn emit_op(b: &mut FunctionBuilder, ctxp: Value, instr: &DecodedInstr) { @@ -182,6 +324,59 @@ fn emit_op(b: &mut FunctionBuilder, ctxp: Value, instr: &DecodedInstr) { let res = b.ins().iadd_imm(rav, (instr.simm16() as i64) << 16); store_gpr(b, ctxp, instr.rd(), res); } + // Unconditional branch. Target is an immediate; `pc` (= this instr's + // address in the interpreter) is `instr.addr` at emit time. + PpcOpcode::bx => { + let target = if instr.aa() { + instr.li() as u32 + } else { + instr.addr.wrapping_add(instr.li() as u32) + }; + if instr.lk() { + store_lr(b, ctxp, instr.addr.wrapping_add(4) as u64); + } + let t = b.ins().iconst(types::I32, target as i64); + store_pc(b, ctxp, t); + } + // Conditional branch (relative). CTR decrement / CR test via + // `emit_branch_taken`; both targets are immediates. + PpcOpcode::bcx => { + let next = instr.addr.wrapping_add(4); + let target = if instr.aa() { + instr.bd() as u32 + } else { + instr.addr.wrapping_add(instr.bd() as u32) + }; + let taken = emit_branch_taken(b, ctxp, instr.bo(), instr.bi()); + let t = b.ins().iconst(types::I32, target as i64); + let n = b.ins().iconst(types::I32, next as i64); + let pc = b.ins().select(taken, t, n); + store_pc(b, ctxp, pc); + // interpreter sets LR = next unconditionally when LK, on both paths. + if instr.lk() { + store_lr(b, ctxp, next as u64); + } + } + // Branch-conditional to LR. Target is the live `lr & !3`, read BEFORE a + // LK link overwrites it. + PpcOpcode::bclrx => { + let next = instr.addr.wrapping_add(4); + let lr_cur = b.ins().load( + types::I64, + MemFlags::trusted(), + ctxp, + off(offset_of!(PpcContext, lr)), + ); + let lr32 = b.ins().ireduce(types::I32, lr_cur); + let target = b.ins().band_imm(lr32, !3i64); + let taken = emit_branch_taken(b, ctxp, instr.bo(), instr.bi()); + let n = b.ins().iconst(types::I32, next as i64); + let pc = b.ins().select(taken, target, n); + store_pc(b, ctxp, pc); + if instr.lk() { + store_lr(b, ctxp, next as u64); + } + } // covered() gates this — unreachable for uncovered opcodes. _ => unreachable!("emit_op called on uncovered opcode {:?}", instr.opcode), } diff --git a/crates/xenia-cpu/src/recompiler.rs b/crates/xenia-cpu/src/recompiler.rs index 8b45075..d6087af 100644 --- a/crates/xenia-cpu/src/recompiler.rs +++ b/crates/xenia-cpu/src/recompiler.rs @@ -43,6 +43,7 @@ use std::sync::atomic::{AtomicU64, AtomicU8, Ordering}; use crate::block_cache::DecodedBlock; use crate::context::PpcContext; use crate::interpreter::{execute, step_block, StepResult}; +use crate::jit::{JitCache, MemEnv, RET_CONTINUE}; use crate::opcode::PpcOpcode; use xenia_memory::MemoryAccess; @@ -127,16 +128,38 @@ pub fn report_histogram() { } } -/// Execute one decoded block. **M0: interpreter fallback for every opcode.** +// Coverage telemetry: blocks executed as native code vs interpreted (uncovered). +static JIT_COMPILED_RUN: AtomicU64 = AtomicU64::new(0); +static JIT_INTERP_RUN: AtomicU64 = AtomicU64::new(0); + +/// Execute one decoded block, natively if it is fully covered, else via the +/// interpreter fallback. /// -/// Byte-for-byte the same loop as [`crate::interpreter::step_block`]: bump -/// `cycle_count`/`timebase` per instruction, bail on the first non-`Continue` -/// result, and stop on a PC discontinuity (only the terminator may branch). +/// If [`JitCache::get_or_compile`] returns a compiled entry point, the whole +/// block runs as native code (which advances `pc`/`cycle_count`/`timebase` +/// exactly like the interpreter — validated by the diff harness). Otherwise +/// this is byte-for-byte the same loop as [`crate::interpreter::step_block`]: +/// bump `cycle_count`/`timebase` per instruction, bail on the first +/// non-`Continue` result, and stop on a PC discontinuity (only the terminator +/// may branch). pub fn run_block( ctx: &mut PpcContext, mem: &dyn MemoryAccess, block: &DecodedBlock, + jit: &mut JitCache, ) -> StepResult { + if let Some(f) = jit.get_or_compile(block) { + JIT_COMPILED_RUN.fetch_add(1, Ordering::Relaxed); + let env = MemEnv { mem }; + let raw = f(ctx as *mut PpcContext, &env as *const MemEnv); + // Covered blocks are straight-line (branch/sc/trap/db16cyc are all + // uncovered), so a compiled block always runs to completion and + // returns `Continue`. + debug_assert_eq!(raw, RET_CONTINUE, "covered block returned non-Continue"); + return StepResult::Continue; + } + JIT_INTERP_RUN.fetch_add(1, Ordering::Relaxed); + let hist = hist_enabled(); let mut result = StepResult::Continue; for instr in &block.instrs { @@ -174,6 +197,7 @@ pub fn diff_step( ctx: &mut PpcContext, mem: &dyn MemoryAccess, block: &DecodedBlock, + jit: &mut JitCache, ) -> StepResult { // Reservation/barrier blocks share cross-thread state; don't speculate. if block.sync_sensitive { @@ -186,7 +210,7 @@ pub fn diff_step( let mut cand = ctx.clone(); cand.reservation_table = None; // never touch the shared reservation table let overlay = OverlayMemory::new(mem); - let _ = run_block(&mut cand, &overlay, block); + let _ = run_block(&mut cand, &overlay, block, jit); let touched_mmio = overlay.touched_mmio.get(); // Authoritative interpreter run: commits real ctx + memory. @@ -271,6 +295,22 @@ pub fn report_diff_summary() { ); } +/// Print native-vs-interpreted block coverage (call at clean exit). Shown +/// whenever the JIT is engaged (`XENIA_JIT` or `XENIA_JIT_DIFF`) so we can see +/// how much of the real boot+movie workload the current covered set captures. +pub fn report_jit_summary() { + if !jit_enabled() && !diff_enabled() { + return; + } + let compiled = JIT_COMPILED_RUN.load(Ordering::Relaxed); + let interp = JIT_INTERP_RUN.load(Ordering::Relaxed); + let total = compiled + interp; + let pct = if total == 0 { 0.0 } else { 100.0 * compiled as f64 / total as f64 }; + eprintln!( + "=== JIT COVERAGE: native blocks={compiled} interpreted={interp} ({pct:.2}% of {total} block-runs native) ===" + ); +} + // ---- overlay memory (speculative-write buffer) --------------------------- /// A [`MemoryAccess`] wrapper for speculative JIT execution: reads fall through