[iterate-4C] JIT native FP-arith Phase 1: fadds/fsubs/fmuls/fdivs (+doubles)
Emit native x64 for the hot FP-arith ops (currently 61.6% of interpreter fallbacks through the intro video; the JIT emitted ZERO native FP arithmetic before). Phase 1 = add/sub/mul/div, single (op59) + double (op63), Rc=0. Correctness rests on a lemma (validated vs fpscr.rs): if the final result is ±NORMAL or ±ZERO, the interpreter set NO FPSCR exception bit and FPRF=classify(result). So the fast path is: RN-nearest guard (FPSCR RN_MASK=0x3, low 2 bits — NOT 0xC0000000) → native movsd/addsd/subsd/mulsd/divsd in xmm (single narrows via cvtsd2ss;cvtss2sd = to_single at RN-nearest, bit-exact) → result guard (reject inf/NaN; for doubles also reject subnormal → UX+DENORMAL) → store + cheap inline FPRF (sign+iszero → one of ±NORMAL/±ZERO codes). Every guarded (rare) case DEOPTS to jit_interpret_one for that one instruction — counter-neutral (state.retire() defers the +1 on both paths; interp_one bumps no counter and touches only fpr/fpscr/pc, never ctx.gpr, so no RegCache flush). Rc=1 forms fall back (update_cr1 not replicated). FMA (madd) family deferred to Phase 2 (needs hardware FMA3). Infra: Offsets.fpscr; MemHelpers.interpret_one; emit_fp_arith + FpArith enum + emit_fprf_normal_or_zero (emit.rs); DynasmLabelApi import (dynamic labels/relocs in emit.rs). check_fp now SEEDS and ASSERTS fpscr (the real safety net) — added check_fp_seeded. New tests: fp_arith_matches (fuzz + edge floats), fp_arith_directed_rounding_matches (RN!=nearest → deopt), fp_arith_block_matches (multi-op native↔deopt↔fallback boundary). 24 xenia-jit tests green. GOLDEN n200m BYTE-IDENTICAL across all 6 configs (interp / JIT / +CHAIN(region) / +REGCACHE / +REGION_MAX=1 / +budget=1==interp+budget=1). Whole-video speedup measured after Phase 2 (the fmadds family is the single biggest op at 27.7%).
This commit is contained in:
@@ -16,7 +16,7 @@
|
||||
//! `rD` (`instr.rd()`); logical forms write `rA` (`instr.ra()`) from `rS`
|
||||
//! (`instr.rs()`).
|
||||
|
||||
use dynasmrt::{DynasmApi, dynasm};
|
||||
use dynasmrt::{DynasmApi, DynasmLabelApi, dynasm};
|
||||
use xenia_cpu::PpcOpcode;
|
||||
use xenia_cpu::context::{CrField, PpcContext};
|
||||
use xenia_cpu::decoder::DecodedInstr;
|
||||
@@ -60,6 +60,9 @@ pub struct Offsets {
|
||||
gpr: i32,
|
||||
/// Base of the `fpr: [f64; 32]` array (8 bytes each).
|
||||
fpr: i32,
|
||||
/// The `fpscr: u32` FP status/control register (native FP-arith reads RN and
|
||||
/// writes FPRF here).
|
||||
fpscr: i32,
|
||||
pub pc: i32,
|
||||
pub cycle: i32,
|
||||
pub timebase: i32,
|
||||
@@ -92,6 +95,7 @@ impl Offsets {
|
||||
env_chain_mmio_before: core::mem::offset_of!(JitEnv, chain_mmio_before) as i32,
|
||||
gpr: core::mem::offset_of!(PpcContext, gpr) as i32,
|
||||
fpr: core::mem::offset_of!(PpcContext, fpr) as i32,
|
||||
fpscr: core::mem::offset_of!(PpcContext, fpscr) as i32,
|
||||
pc: core::mem::offset_of!(PpcContext, pc) as i32,
|
||||
cycle: core::mem::offset_of!(PpcContext, cycle_count) as i32,
|
||||
timebase: core::mem::offset_of!(PpcContext, timebase) as i32,
|
||||
@@ -701,6 +705,51 @@ pub fn try_emit_native(
|
||||
Emit::Native
|
||||
}
|
||||
|
||||
// ===== Native FP arithmetic (Phase 1: add/sub/mul/div, single + double).
|
||||
// Rc=1 forms fall back (`update_cr1` not replicated — Rc on FP-arith is
|
||||
// rare). Byte-identical via the RN + finite-result guards in
|
||||
// `emit_fp_arith`; the guarded (rare) cases deopt to the interpreter. =====
|
||||
PpcOpcode::faddsx if !instr.rc_bit() => {
|
||||
emit_fp_arith(ops, off, helpers.interpret_one, instr, FpArith::Add, true);
|
||||
state.retire();
|
||||
Emit::Native
|
||||
}
|
||||
PpcOpcode::fsubsx if !instr.rc_bit() => {
|
||||
emit_fp_arith(ops, off, helpers.interpret_one, instr, FpArith::Sub, true);
|
||||
state.retire();
|
||||
Emit::Native
|
||||
}
|
||||
PpcOpcode::fmulsx if !instr.rc_bit() => {
|
||||
emit_fp_arith(ops, off, helpers.interpret_one, instr, FpArith::Mul, true);
|
||||
state.retire();
|
||||
Emit::Native
|
||||
}
|
||||
PpcOpcode::fdivsx if !instr.rc_bit() => {
|
||||
emit_fp_arith(ops, off, helpers.interpret_one, instr, FpArith::Div, true);
|
||||
state.retire();
|
||||
Emit::Native
|
||||
}
|
||||
PpcOpcode::faddx if !instr.rc_bit() => {
|
||||
emit_fp_arith(ops, off, helpers.interpret_one, instr, FpArith::Add, false);
|
||||
state.retire();
|
||||
Emit::Native
|
||||
}
|
||||
PpcOpcode::fsubx if !instr.rc_bit() => {
|
||||
emit_fp_arith(ops, off, helpers.interpret_one, instr, FpArith::Sub, false);
|
||||
state.retire();
|
||||
Emit::Native
|
||||
}
|
||||
PpcOpcode::fmulx if !instr.rc_bit() => {
|
||||
emit_fp_arith(ops, off, helpers.interpret_one, instr, FpArith::Mul, false);
|
||||
state.retire();
|
||||
Emit::Native
|
||||
}
|
||||
PpcOpcode::fdivx if !instr.rc_bit() => {
|
||||
emit_fp_arith(ops, off, helpers.interpret_one, instr, FpArith::Div, false);
|
||||
state.retire();
|
||||
Emit::Native
|
||||
}
|
||||
|
||||
// ===== Rotate/mask (sh/mb/me are compile-time constants -> the 32/64-bit
|
||||
// mask folds to a constant, so these are just rol + and). =====
|
||||
// rlwinm: RA = ROTL32(RS, SH) & MASK(mb,me)
|
||||
@@ -1190,6 +1239,155 @@ fn emit_fp_store(ops: &mut Asm, off: &Offsets, helper: i64, rs: usize) {
|
||||
);
|
||||
}
|
||||
|
||||
/// A native FP-arithmetic op — selects operand roles and the x64 instruction.
|
||||
#[derive(Clone, Copy)]
|
||||
pub enum FpArith {
|
||||
Add, // frD = frA + frB
|
||||
Sub, // frD = frA - frB
|
||||
Mul, // frD = frA * frC (A-form: multiplicand is frC, bits 21-25)
|
||||
Div, // frD = frA / frB
|
||||
Madd, // frD = (frA*frC + frB)
|
||||
Msub, // frD = (frA*frC - frB)
|
||||
Nmadd, // frD = -(frA*frC + frB)
|
||||
Nmsub, // frD = -(frA*frC - frB)
|
||||
}
|
||||
|
||||
/// Emit a native FP-arithmetic op, byte-identical to the interpreter on the
|
||||
/// common path and DEOPTING to the interpreter (for this one instruction) on the
|
||||
/// rare guarded cases. Correctness rests on the lemma (see `fpscr.rs`): if the
|
||||
/// final result is neither `inf` nor `NaN`, the interpreter set NO FPSCR
|
||||
/// exception bit and `FPRF = classify(result) ∈ {±NORMAL, ±ZERO}`.
|
||||
///
|
||||
/// Fast path: RN-nearest guard → arithmetic in xmm0 (single-prec narrows via
|
||||
/// `cvtsd2ss;cvtss2sd` = `to_single` at RN-nearest, bit-exact) → finite-result
|
||||
/// guard → store → cheap inline FPRF. The arm calls `state.retire()` on BOTH
|
||||
/// paths; the deopt is counter-neutral (`jit_interpret_one` bumps no counter and
|
||||
/// touches only fpr/fpscr/pc — never `ctx.gpr`, so no RegCache flush is needed).
|
||||
/// The FMA (madd) family requires host FMA3 — the caller must gate on it.
|
||||
fn emit_fp_arith(
|
||||
ops: &mut Asm,
|
||||
off: &Offsets,
|
||||
interp_helper: i64,
|
||||
instr: &DecodedInstr,
|
||||
op: FpArith,
|
||||
single: bool,
|
||||
) {
|
||||
let (ra, rb, rc, rd) = (instr.ra(), instr.rb(), instr.rc(), instr.rd());
|
||||
let deopt = ops.new_dynamic_label();
|
||||
let done = ops.new_dynamic_label();
|
||||
|
||||
// RN-nearest guard: FPSCR[RN] = the low 2 bits (RN_MASK 0x3; 00 = nearest,
|
||||
// the hardware default). Directed rounding → the interpreter's software round
|
||||
// differs from hardware cvt, so deopt to it.
|
||||
dynasm!(ops ; .arch x64
|
||||
; test DWORD [r15 + off.fpscr], 0x3
|
||||
; jnz =>deopt
|
||||
);
|
||||
|
||||
// Arithmetic → f64 result in xmm0.
|
||||
match op {
|
||||
FpArith::Add => dynasm!(ops ; .arch x64
|
||||
; movsd xmm0, QWORD [r15 + off.fpr(ra)]
|
||||
; addsd xmm0, QWORD [r15 + off.fpr(rb)]),
|
||||
FpArith::Sub => dynasm!(ops ; .arch x64
|
||||
; movsd xmm0, QWORD [r15 + off.fpr(ra)]
|
||||
; subsd xmm0, QWORD [r15 + off.fpr(rb)]),
|
||||
FpArith::Mul => dynasm!(ops ; .arch x64
|
||||
; movsd xmm0, QWORD [r15 + off.fpr(ra)]
|
||||
; mulsd xmm0, QWORD [r15 + off.fpr(rc)]),
|
||||
FpArith::Div => dynasm!(ops ; .arch x64
|
||||
; movsd xmm0, QWORD [r15 + off.fpr(ra)]
|
||||
; divsd xmm0, QWORD [r15 + off.fpr(rb)]),
|
||||
// Fused, single rounding — matches the interpreter's `a.mul_add(c, b)`.
|
||||
FpArith::Madd | FpArith::Nmadd => dynasm!(ops ; .arch x64
|
||||
; movsd xmm0, QWORD [r15 + off.fpr(ra)]
|
||||
; movsd xmm1, QWORD [r15 + off.fpr(rc)]
|
||||
; vfmadd213sd xmm0, xmm1, QWORD [r15 + off.fpr(rb)]), // ra*rc + rb
|
||||
FpArith::Msub | FpArith::Nmsub => dynasm!(ops ; .arch x64
|
||||
; movsd xmm0, QWORD [r15 + off.fpr(ra)]
|
||||
; movsd xmm1, QWORD [r15 + off.fpr(rc)]
|
||||
; vfmsub213sd xmm0, xmm1, QWORD [r15 + off.fpr(rb)]), // ra*rc - rb
|
||||
}
|
||||
// Negated FMA: the interpreter negates the WHOLE (non-NaN) fused result before
|
||||
// to_single. The finite-result guard below excludes NaN, so the fast path is
|
||||
// always the non-NaN branch → an unconditional sign flip is exact.
|
||||
if matches!(op, FpArith::Nmadd | FpArith::Nmsub) {
|
||||
dynasm!(ops ; .arch x64
|
||||
; mov rax, QWORD 0x8000_0000_0000_0000u64 as i64
|
||||
; movq xmm1, rax
|
||||
; xorpd xmm0, xmm1
|
||||
);
|
||||
}
|
||||
// to_single (RN-nearest): narrow to f32 then widen back = `(v as f32) as f64`.
|
||||
if single {
|
||||
dynasm!(ops ; .arch x64 ; cvtsd2ss xmm0, xmm0 ; cvtss2sd xmm0, xmm0);
|
||||
}
|
||||
// Result guard: accept only ±NORMAL or ±ZERO (the lemma's fast-path domain).
|
||||
// Exponent field (bits 52-62) all-ones ⇒ inf/NaN ⇒ deopt.
|
||||
dynasm!(ops ; .arch x64
|
||||
; movq rax, xmm0
|
||||
; mov rcx, rax
|
||||
; mov rdx, QWORD 0x7FF0_0000_0000_0000u64 as i64
|
||||
; and rcx, rdx // rcx = exponent field
|
||||
; cmp rcx, rdx
|
||||
; je =>deopt
|
||||
);
|
||||
// Double-precision can produce a genuine subnormal f64 (exp==0, mantissa!=0),
|
||||
// which the interpreter flags UX + DENORMAL → deopt. Single-precision can't:
|
||||
// an f32-subnormal widened by to_single is a NORMAL f64, so no guard needed.
|
||||
if !single {
|
||||
let not_sub = ops.new_dynamic_label();
|
||||
dynasm!(ops ; .arch x64
|
||||
; test rcx, rcx // exp == 0?
|
||||
; jnz =>not_sub // exp != 0 ⇒ normal ⇒ accept
|
||||
; mov rcx, rax
|
||||
; shl rcx, 1 // drop sign; ZF=1 iff ±0 (mantissa==0)
|
||||
; jnz =>deopt // exp==0 & mantissa!=0 ⇒ subnormal ⇒ deopt
|
||||
; =>not_sub
|
||||
);
|
||||
}
|
||||
dynasm!(ops ; .arch x64 ; mov [r15 + off.fpr(rd)], rax);
|
||||
// Inline FPRF for the ±NORMAL/±ZERO result (bits still in rax).
|
||||
emit_fprf_normal_or_zero(ops, off);
|
||||
dynasm!(ops ; .arch x64 ; jmp =>done);
|
||||
|
||||
// Deopt: interpret THIS instruction. Set pc so the interpreter advances it.
|
||||
let instr_ptr = instr as *const DecodedInstr as usize as i64;
|
||||
let addr = instr.addr as i32;
|
||||
dynasm!(ops ; .arch x64
|
||||
; =>deopt
|
||||
; mov DWORD [r15 + off.pc], addr
|
||||
; mov rdi, rbx
|
||||
; mov rsi, QWORD instr_ptr
|
||||
; mov rax, QWORD interp_helper
|
||||
; call rax
|
||||
; =>done
|
||||
);
|
||||
}
|
||||
|
||||
/// Write FPSCR.FPRF for a result KNOWN to be ±NORMAL or ±ZERO (bits in `rax`,
|
||||
/// preserved). FPRF = bits 12-16 (mask `0x1F<<12`); codes (fpscr.rs `fprf`):
|
||||
/// POS_ZERO 0x02 / NEG_ZERO 0x12 / POS_NORMAL 0x04 / NEG_NORMAL 0x08. Branchless.
|
||||
fn emit_fprf_normal_or_zero(ops: &mut Asm, off: &Offsets) {
|
||||
dynasm!(ops ; .arch x64
|
||||
; mov rcx, rax
|
||||
; shr rcx, 63 // ecx = sign bit (0/1)
|
||||
; mov edx, ecx
|
||||
; shl edx, 2
|
||||
; add edx, 0x04 // edx = norm code = 0x04 + sign*0x04
|
||||
; shl ecx, 4
|
||||
; add ecx, 0x02 // ecx = zero code = 0x02 + sign*0x10
|
||||
; mov r8, rax
|
||||
; add r8, r8 // ZF=1 iff +/-0 (sign shifted out)
|
||||
; cmovnz ecx, edx // nonzero → use norm code
|
||||
; shl ecx, 12 // into the FPRF field
|
||||
; mov edx, [r15 + off.fpscr]
|
||||
; and edx, 0xFFFE_0FFFu32 as i32 // clear FPRF (bits 12-16)
|
||||
; or edx, ecx
|
||||
; mov [r15 + off.fpscr], edx
|
||||
);
|
||||
}
|
||||
|
||||
/// Emit `cr[0] = update_cr_signed(result)` where the result is in `rax`/`eax`:
|
||||
/// a signed comparison of the value against 0. `test` sets ZF/SF (OF=0), so the
|
||||
/// signed `setl/setg/sete` in [`emit_cr_from_flags`] give lt/gt/eq vs 0.
|
||||
|
||||
@@ -314,6 +314,10 @@ pub(crate) struct MemHelpers {
|
||||
pub read_f64: i64,
|
||||
pub store_f32: i64,
|
||||
pub store_f64: i64,
|
||||
/// `jit_interpret_one` — the per-instruction interpreter fallback, also used
|
||||
/// as the counter-neutral deopt target inside native FP-arith arms (their
|
||||
/// guard-failure path runs the interpreter for that one instruction).
|
||||
pub interpret_one: i64,
|
||||
}
|
||||
impl MemHelpers {
|
||||
pub(crate) fn resolve() -> Self {
|
||||
@@ -330,6 +334,7 @@ impl MemHelpers {
|
||||
read_f64: jit_read_f64 as usize as i64,
|
||||
store_f32: jit_store_f32 as usize as i64,
|
||||
store_f64: jit_store_f64 as usize as i64,
|
||||
interpret_one: jit_interpret_one as usize as i64,
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
@@ -797,6 +797,14 @@ fn fuzz_gpr_based(seed: &mut u64, ra: u32) -> [u64; 32] {
|
||||
/// memory + counters match. The f32<->f64 conversion lives in a shared helper,
|
||||
/// so this primarily validates EA computation, fpr indexing, and the ABI.
|
||||
fn check_fp(raw: u32, gpr: [u64; 32], fpr_bits: [u64; 32]) {
|
||||
check_fp_seeded(raw, gpr, fpr_bits, 0);
|
||||
}
|
||||
|
||||
/// Like [`check_fp`] but seeds `ctx.fpscr` (for FP-arith: exercises the RN-guard
|
||||
/// and directed-rounding fallback) and — critically — asserts `fpscr` is
|
||||
/// byte-identical to the interpreter (a wrong FPRF / spurious exception bit is
|
||||
/// invisible without this).
|
||||
fn check_fp_seeded(raw: u32, gpr: [u64; 32], fpr_bits: [u64; 32], fpscr_seed: u32) {
|
||||
let pc = 0x8200_1000u32;
|
||||
let instr = decode(raw, pc);
|
||||
let off = emit::Offsets::resolve();
|
||||
@@ -811,6 +819,7 @@ fn check_fp(raw: u32, gpr: [u64; 32], fpr_bits: [u64; 32]) {
|
||||
for (i, b) in fpr_bits.iter().enumerate() {
|
||||
c.fpr[i] = f64::from_bits(*b);
|
||||
}
|
||||
c.fpscr = fpscr_seed;
|
||||
};
|
||||
|
||||
let mem_a = VecMem::seeded();
|
||||
@@ -837,6 +846,11 @@ fn check_fp(raw: u32, gpr: [u64; 32], fpr_bits: [u64; 32]) {
|
||||
let fa: [u64; 32] = std::array::from_fn(|i| a.fpr[i].to_bits());
|
||||
let fb: [u64; 32] = std::array::from_fn(|i| b.fpr[i].to_bits());
|
||||
assert_eq!(fa, fb, "fpr mismatch raw={raw:#010x} ({:?})", instr.opcode);
|
||||
assert_eq!(
|
||||
a.fpscr, b.fpscr,
|
||||
"fpscr mismatch raw={raw:#010x} ({:?}): interp {:#010x} vs jit {:#010x}",
|
||||
instr.opcode, a.fpscr, b.fpscr
|
||||
);
|
||||
assert_eq!(a.pc, b.pc, "pc mismatch raw={raw:#010x}");
|
||||
assert_eq!(a.cycle_count, b.cycle_count, "cycle mismatch raw={raw:#010x}");
|
||||
assert_eq!(mem_a.snapshot(), mem_b.snapshot(), "memory mismatch raw={raw:#010x} ({:?})", instr.opcode);
|
||||
@@ -885,6 +899,120 @@ fn fp_loadstore_matches() {
|
||||
}
|
||||
}
|
||||
|
||||
/// A-form FP: `op | frD<<21 | frA<<16 | frB<<11 | frC<<6 | XO<<1 | Rc`.
|
||||
fn enc_a(op: u32, rd: u32, ra: u32, rb: u32, rc: u32, xo: u32, rc_bit: u32) -> u32 {
|
||||
(op << 26) | (rd << 21) | (ra << 16) | (rb << 11) | (rc << 6) | (xo << 1) | rc_bit
|
||||
}
|
||||
|
||||
/// Native FP-arith (Phase 1: add/sub/mul/div, single op59 + double op63). Fuzzes
|
||||
/// with `fuzz_fpr`'s edge floats (0/-0/±inf/NaN/1e-300/π) so inf/NaN inputs hit
|
||||
/// the finite-result guard→fallback and 1e-300 products hit f32 underflow, all
|
||||
/// compared against interp incl. `fpscr` (via `check_fp`).
|
||||
#[test]
|
||||
fn fp_arith_matches() {
|
||||
let mut s = 0xfeed_beef_u64;
|
||||
for _ in 0..ITERS {
|
||||
let rd = (rng(&mut s) % 32) as u32;
|
||||
let ra = (rng(&mut s) % 32) as u32;
|
||||
let rb = (rng(&mut s) % 32) as u32;
|
||||
let rc = (rng(&mut s) % 32) as u32;
|
||||
let g = fuzz_gpr(&mut s);
|
||||
let f = fuzz_fpr(&mut s);
|
||||
// single (op 59): fadds XO21, fsubs XO20, fmuls XO25 (frC), fdivs XO18
|
||||
check_fp(enc_a(59, rd, ra, rb, 0, 21, 0), g, f);
|
||||
check_fp(enc_a(59, rd, ra, rb, 0, 20, 0), g, f);
|
||||
check_fp(enc_a(59, rd, ra, 0, rc, 25, 0), g, f);
|
||||
check_fp(enc_a(59, rd, ra, rb, 0, 18, 0), g, f);
|
||||
// double (op 63)
|
||||
check_fp(enc_a(63, rd, ra, rb, 0, 21, 0), g, f);
|
||||
check_fp(enc_a(63, rd, ra, rb, 0, 20, 0), g, f);
|
||||
check_fp(enc_a(63, rd, ra, 0, rc, 25, 0), g, f);
|
||||
check_fp(enc_a(63, rd, ra, rb, 0, 18, 0), g, f);
|
||||
}
|
||||
}
|
||||
|
||||
/// FP-arith with FPSCR[RN] set to each directed rounding mode: the JIT must hit
|
||||
/// the RN guard and deopt to the interpreter's software rounding (byte-identical,
|
||||
/// incl. fpscr). Also exercises non-nearest FPRF via the deopt.
|
||||
#[test]
|
||||
fn fp_arith_directed_rounding_matches() {
|
||||
let mut s = 0x1234_5678_u64;
|
||||
for _ in 0..ITERS / 4 {
|
||||
let rd = (rng(&mut s) % 32) as u32;
|
||||
let ra = (rng(&mut s) % 32) as u32;
|
||||
let rb = (rng(&mut s) % 32) as u32;
|
||||
let rc = (rng(&mut s) % 32) as u32;
|
||||
let g = fuzz_gpr(&mut s);
|
||||
let f = fuzz_fpr(&mut s);
|
||||
for rn in 1u32..=3 {
|
||||
check_fp_seeded(enc_a(59, rd, ra, rb, 0, 21, 0), g, f, rn); // fadds
|
||||
check_fp_seeded(enc_a(59, rd, ra, 0, rc, 25, 0), g, f, rn); // fmuls
|
||||
check_fp_seeded(enc_a(63, rd, ra, rb, 0, 18, 0), g, f, rn); // fdiv
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
/// Multi-op block mixing native FP-arith, FP load/store, and a FALLBACK (mulli),
|
||||
/// run as one compiled block vs `step_block` — catches counter-deferral /
|
||||
/// native↔deopt boundary bugs (fpr + fpscr + counters + mem).
|
||||
#[test]
|
||||
fn fp_arith_block_matches() {
|
||||
use xenia_cpu::interpreter::step_block;
|
||||
let base = 0x8200_1000u32;
|
||||
// r5 base for lfs/stfs EAs (into the mock's addressable range).
|
||||
let raws = [
|
||||
enc_d(14, 5, 0, 0x2000), // addi r5, r0, 0x2000 (native)
|
||||
enc_d(48, 1, 5, 0x00), // lfs f1, 0(r5) (native FP load)
|
||||
enc_d(48, 2, 5, 0x10), // lfs f2, 0x10(r5) (native FP load)
|
||||
enc_a(59, 3, 1, 2, 0, 21, 0), // fadds f3, f1, f2 (native FP arith)
|
||||
enc_a(59, 4, 3, 0, 1, 25, 0), // fmuls f4, f3, f1 (native FP arith)
|
||||
(7 << 26) | (6 << 21) | (5 << 16) | 3, // mulli r6, r5, 3 (FALLBACK)
|
||||
enc_a(63, 7, 4, 3, 0, 20, 0), // fsub f7, f4, f3 (native FP arith)
|
||||
enc_d(52, 7, 5, 0x20), // stfs f7, 0x20(r5) (native FP store)
|
||||
enc_bx(0x40, 0, 0), // b +0x40 (terminator)
|
||||
];
|
||||
let instrs: Vec<DecodedInstr> = raws
|
||||
.iter()
|
||||
.enumerate()
|
||||
.map(|(i, &raw)| decode(raw, base + (i as u32) * 4))
|
||||
.collect();
|
||||
let block = DecodedBlock {
|
||||
start_pc: base,
|
||||
end_pc: base + (raws.len() as u32) * 4,
|
||||
page_version: 0,
|
||||
instrs,
|
||||
sync_sensitive: false,
|
||||
};
|
||||
let gpr = { let mut s = 0xabcd_u64; fuzz_gpr(&mut s) };
|
||||
let fpr = { let mut s = 0x9911_u64; fuzz_fpr(&mut s) };
|
||||
let seed = |c: &mut PpcContext| {
|
||||
c.gpr = gpr;
|
||||
for (i, b) in fpr.iter().enumerate() { c.fpr[i] = f64::from_bits(*b); }
|
||||
};
|
||||
|
||||
let mem_a = VecMem::seeded();
|
||||
let mut a = ctx_from_gpr([0; 32], base);
|
||||
seed(&mut a);
|
||||
let ra = step_block(&mut a, &mem_a, &block);
|
||||
|
||||
let mem_b = VecMem::seeded();
|
||||
let mut b = ctx_from_gpr([0; 32], base);
|
||||
seed(&mut b);
|
||||
let cb = compile_block(&block);
|
||||
let rb = run_jit_block(&cb, &mut b, &mem_b);
|
||||
|
||||
let fa: [u64; 32] = std::array::from_fn(|i| a.fpr[i].to_bits());
|
||||
let fb: [u64; 32] = std::array::from_fn(|i| b.fpr[i].to_bits());
|
||||
assert_eq!(a.gpr, b.gpr, "block gpr mismatch");
|
||||
assert_eq!(fa, fb, "block fpr mismatch");
|
||||
assert_eq!(a.fpscr, b.fpscr, "block fpscr mismatch");
|
||||
assert_eq!(a.pc, b.pc, "block pc mismatch");
|
||||
assert_eq!(a.cycle_count, b.cycle_count, "block cycle mismatch");
|
||||
assert_eq!(a.timebase, b.timebase, "block timebase mismatch");
|
||||
assert_eq!(mem_a.snapshot(), mem_b.snapshot(), "block mem mismatch");
|
||||
assert_eq!(ra, rb, "block StepResult mismatch");
|
||||
}
|
||||
|
||||
#[test]
|
||||
fn loads_match() {
|
||||
let mut s = 0x11ffu64;
|
||||
|
||||
Reference in New Issue
Block a user