[iterate-4C] JIT Phase 4c: per-block register caching (XENIA_JIT_REGCACHE)

Pin the 3 hottest guest GPRs (static ra/rb/rd frequency) into callee-saved
host regs r12/r13/r14 across a block. RegCache + gld64/gld32/gst64
accessors (dynasm Rq/Rd dynamic regs); ALL 75 GPR-access sites in the
emitters routed through the accessors. Prologue loads cached regs / epilogue
flushes (both l_cont+l_exit merge); fallback path flushes-before /
reloads-after (interpreter touches ctx.gpr; mem/FP helpers don't, and cached
regs are callee-saved, so no flush around those). RA=0 literal-0 rule
unaffected (emit_ea/load_ra_or_zero never read gpr[0]). Gated by
XENIA_JIT_REGCACHE (default off; forced on under cfg(test)); conditional
prologue keeps the cache-off codegen unchanged.

Golden n200m BYTE-IDENTICAL in all 3 configs (interp / JIT / JIT+RC); 17
differential tests pass with the cache FORCED ON. Measured (min of 6):
JIT+RC 4.16s vs JIT 4.14s vs interp 3.80s -> NEUTRAL on this benchmark:
per-block setup (5 push/pop + 3 load/flush) + fallback flush/reload (+6
mem ops each) cancels the intra-block savings (blocks ~13 instrs, fallbacks
frequent). Correct + default-off; may help more CPU-bound workloads
(gameplay, larger hot loops). Definitively: no JIT technique (coverage,
deferral, regcache) beats interp on this ~60%-overhead boot/render bench.

Co-Authored-By: Claude Opus 4.8 <noreply@anthropic.com>
This commit is contained in:
MechaCat02
2026-07-04 20:50:39 +02:00
parent 506b9554a5
commit 1ffa3fb56d
3 changed files with 336 additions and 167 deletions

View File

@@ -176,6 +176,106 @@ impl EmitState {
}
}
/// Host registers guest GPRs are cached into across a block. Callee-saved, so
/// they survive the `call`s to memory/fallback helpers (rbx=3 and r15=15 are
/// already pinned for env/ctx). Scratch regs used by emitters (rax/rcx/rdx/
/// rsi/rdi/r8) are disjoint from these, so an accessor `mov` is never a
/// self-move.
pub const CACHE_HOST_REGS: [u8; 3] = [12, 13, 14]; // r12, r13, r14
/// Per-block mapping of a few hot guest GPRs to host registers. Correctness is
/// independent of the choice: every native GPR access goes through the
/// accessors, cached regs are loaded at the prologue and flushed at every
/// block-exit edge, and the fallback path flushes-before / reloads-after (the
/// interpreter reads/writes `ctx.gpr` directly). Memory/FP helpers only touch
/// guest memory + the reservation table, not `ctx.gpr`, so cached regs (being
/// callee-saved) survive those calls with no flush.
pub struct RegCache {
host: [Option<u8>; 32],
active: Vec<(u8, usize)>, // (host reg code, guest gpr)
}
impl RegCache {
/// Empty cache: all GPR access goes to memory (unchanged codegen).
pub fn disabled() -> Self {
Self { host: [None; 32], active: Vec::new() }
}
/// Build a cache for `instrs` by caching the most-referenced GPRs. The
/// frequency count (ra/rb/rd fields) is a pure heuristic — any choice is
/// correct since all access is routed through the accessors.
pub fn build(instrs: &[DecodedInstr]) -> Self {
let mut freq = [0u32; 32];
for i in instrs {
freq[i.ra()] += 1;
freq[i.rb()] += 1;
freq[i.rd()] += 1; // == rs()
}
let mut order: Vec<usize> = (0..32).filter(|&g| freq[g] > 0).collect();
order.sort_by_key(|&g| core::cmp::Reverse((freq[g], g)));
let mut host = [None; 32];
let mut active = Vec::new();
for (slot, &g) in order.iter().take(CACHE_HOST_REGS.len()).enumerate() {
let h = CACHE_HOST_REGS[slot];
host[g] = Some(h);
active.push((h, g));
}
Self { host, active }
}
#[inline]
fn host(&self, g: usize) -> Option<u8> {
self.host[g]
}
pub fn active(&self) -> &[(u8, usize)] {
&self.active
}
}
/// Prologue: load each cached reg from `ctx.gpr` into its host register.
pub fn emit_cache_load(ops: &mut Asm, c: &RegCache, off: &Offsets) {
for &(h, g) in c.active() {
dynasm!(ops ; .arch x64 ; mov Rq(h), [r15 + off.gpr(g)]);
}
}
/// Flush each cached reg back to `ctx.gpr` (before a fallback call / at exit).
pub fn emit_cache_flush(ops: &mut Asm, c: &RegCache, off: &Offsets) {
for &(h, g) in c.active() {
dynasm!(ops ; .arch x64 ; mov [r15 + off.gpr(g)], Rq(h));
}
}
/// Emit `Rq(dst) <- gpr[g]` (64-bit) from the cached host reg or memory.
#[inline]
fn gld64(ops: &mut Asm, c: &RegCache, off: &Offsets, dst: u8, g: usize) {
match c.host(g) {
Some(h) => dynasm!(ops ; .arch x64 ; mov Rq(dst), Rq(h)),
None => dynasm!(ops ; .arch x64 ; mov Rq(dst), [r15 + off.gpr(g)]),
}
}
/// Emit `Rd(dst) <- gpr[g]` (low 32, zero-extends into `dst`).
#[inline]
fn gld32(ops: &mut Asm, c: &RegCache, off: &Offsets, dst: u8, g: usize) {
match c.host(g) {
Some(h) => dynasm!(ops ; .arch x64 ; mov Rd(dst), Rd(h)),
None => dynasm!(ops ; .arch x64 ; mov Rd(dst), [r15 + off.gpr(g)]),
}
}
/// Emit `gpr[g] <- Rq(src)` (64-bit) to the cached host reg or memory.
#[inline]
fn gst64(ops: &mut Asm, c: &RegCache, off: &Offsets, g: usize, src: u8) {
match c.host(g) {
Some(h) => dynasm!(ops ; .arch x64 ; mov Rq(h), Rq(src)),
None => dynasm!(ops ; .arch x64 ; mov [r15 + off.gpr(g)], Rq(src)),
}
}
// Scratch register codes used by the emitters.
const RAX: u8 = 0;
const RCX: u8 = 1;
const RDX: u8 = 2;
const RSI: u8 = 6;
/// Try to emit native x64 for `instr`. Returns [`Emit::Native`] if it fully
/// handled the instruction (computation + `advance_and_count`), [`Emit::Branch`]
/// if it handled a branch (caller must add the pc-discontinuity exit check), or
@@ -185,6 +285,7 @@ pub fn try_emit_native(
off: &Offsets,
helpers: &crate::MemHelpers,
state: &mut EmitState,
cache: &RegCache,
instr: &DecodedInstr,
) -> Emit {
let ra = instr.ra();
@@ -194,11 +295,11 @@ pub fn try_emit_native(
// rD = (rA==0 ? 0 : gpr[rA]) + EXTS(SIMM) [64-bit; never records]
PpcOpcode::addi => {
let simm = instr.simm16() as i32; // fits i32 (16-bit sign-extended)
load_ra_or_zero(ops, off, ra);
load_ra_or_zero(ops, off, cache, ra);
if simm != 0 {
dynasm!(ops ; .arch x64 ; add rax, simm);
}
dynasm!(ops ; .arch x64 ; mov [r15 + off.gpr(rd)], rax);
gst64(ops, cache, off, rd, RAX);
state.retire();
Emit::Native
}
@@ -207,57 +308,57 @@ pub fn try_emit_native(
// (i16 sign-extended) << 16 occupies bits 16..31 with the sign in
// bit 31 — exactly an i32, so it fits an `add r64, imm32`.
let simm = (instr.simm16() as i32) << 16;
load_ra_or_zero(ops, off, ra);
load_ra_or_zero(ops, off, cache, ra);
if simm != 0 {
dynasm!(ops ; .arch x64 ; add rax, simm);
}
dynasm!(ops ; .arch x64 ; mov [r15 + off.gpr(rd)], rax);
gst64(ops, cache, off, rd, RAX);
state.retire();
Emit::Native
}
// gpr[rA] = gpr[rS] | ZEXT(UIMM) [never records]
PpcOpcode::ori => {
let uimm = instr.uimm16() as i32; // 0..65535 -> positive i32 (zext == sext)
dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.gpr(rd)]);
gld64(ops, cache, off, RAX, rd);
if uimm != 0 {
dynasm!(ops ; .arch x64 ; or rax, uimm);
}
dynasm!(ops ; .arch x64 ; mov [r15 + off.gpr(ra)], rax);
gst64(ops, cache, off, ra, RAX);
state.retire();
Emit::Native
}
// gpr[rA] = gpr[rS] | (ZEXT(UIMM) << 16)
PpcOpcode::oris => {
let imm = (instr.uimm16() as u32) << 16;
gld64(ops, cache, off, RAX, rd);
dynasm!(ops ; .arch x64
; mov rax, [r15 + off.gpr(rd)]
; mov ecx, imm as i32 // zero-extends into rcx
; or rax, rcx
; mov [r15 + off.gpr(ra)], rax
);
gst64(ops, cache, off, ra, RAX);
state.retire();
Emit::Native
}
// gpr[rA] = gpr[rS] ^ ZEXT(UIMM)
PpcOpcode::xori => {
let uimm = instr.uimm16() as i32;
dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.gpr(rd)]);
gld64(ops, cache, off, RAX, rd);
if uimm != 0 {
dynasm!(ops ; .arch x64 ; xor rax, uimm);
}
dynasm!(ops ; .arch x64 ; mov [r15 + off.gpr(ra)], rax);
gst64(ops, cache, off, ra, RAX);
state.retire();
Emit::Native
}
// gpr[rA] = gpr[rS] ^ (ZEXT(UIMM) << 16)
PpcOpcode::xoris => {
let imm = (instr.uimm16() as u32) << 16;
gld64(ops, cache, off, RAX, rd);
dynasm!(ops ; .arch x64
; mov rax, [r15 + off.gpr(rd)]
; mov ecx, imm as i32
; xor rax, rcx
; mov [r15 + off.gpr(ra)], rax
);
gst64(ops, cache, off, ra, RAX);
state.retire();
Emit::Native
}
@@ -267,11 +368,10 @@ pub fn try_emit_native(
if instr.rc_bit() || instr.raw == 0x7FFF_FB78 {
return Emit::Fallback;
}
dynasm!(ops ; .arch x64
; mov rax, [r15 + off.gpr(rd)]
; or rax, [r15 + off.gpr(rb)]
; mov [r15 + off.gpr(ra)], rax
);
gld64(ops, cache, off, RAX, rd);
gld64(ops, cache, off, RCX, rb);
dynasm!(ops ; .arch x64 ; or rax, rcx);
gst64(ops, cache, off, ra, RAX);
state.retire();
Emit::Native
}
@@ -280,11 +380,10 @@ pub fn try_emit_native(
if instr.rc_bit() {
return Emit::Fallback;
}
dynasm!(ops ; .arch x64
; mov rax, [r15 + off.gpr(rd)]
; and rax, [r15 + off.gpr(rb)]
; mov [r15 + off.gpr(ra)], rax
);
gld64(ops, cache, off, RAX, rd);
gld64(ops, cache, off, RCX, rb);
dynasm!(ops ; .arch x64 ; and rax, rcx);
gst64(ops, cache, off, ra, RAX);
state.retire();
Emit::Native
}
@@ -293,11 +392,10 @@ pub fn try_emit_native(
if instr.rc_bit() {
return Emit::Fallback;
}
dynasm!(ops ; .arch x64
; mov rax, [r15 + off.gpr(rd)]
; xor rax, [r15 + off.gpr(rb)]
; mov [r15 + off.gpr(ra)], rax
);
gld64(ops, cache, off, RAX, rd);
gld64(ops, cache, off, RCX, rb);
dynasm!(ops ; .arch x64 ; xor rax, rcx);
gst64(ops, cache, off, ra, RAX);
state.retire();
Emit::Native
}
@@ -306,11 +404,10 @@ pub fn try_emit_native(
if instr.oe() || instr.rc_bit() {
return Emit::Fallback;
}
dynasm!(ops ; .arch x64
; mov rax, [r15 + off.gpr(ra)]
; add rax, [r15 + off.gpr(rb)]
; mov [r15 + off.gpr(rd)], rax
);
gld64(ops, cache, off, RAX, ra);
gld64(ops, cache, off, RCX, rb);
dynasm!(ops ; .arch x64 ; add rax, rcx);
gst64(ops, cache, off, rd, RAX);
state.retire();
Emit::Native
}
@@ -319,11 +416,10 @@ pub fn try_emit_native(
if instr.oe() || instr.rc_bit() {
return Emit::Fallback;
}
dynasm!(ops ; .arch x64
; mov rax, [r15 + off.gpr(rb)]
; sub rax, [r15 + off.gpr(ra)]
; mov [r15 + off.gpr(rd)], rax
);
gld64(ops, cache, off, RAX, rb);
gld64(ops, cache, off, RCX, ra);
dynasm!(ops ; .arch x64 ; sub rax, rcx);
gst64(ops, cache, off, rd, RAX);
state.retire();
Emit::Native
}
@@ -332,11 +428,9 @@ pub fn try_emit_native(
if instr.oe() || instr.rc_bit() {
return Emit::Fallback;
}
dynasm!(ops ; .arch x64
; mov rax, [r15 + off.gpr(ra)]
; neg rax
; mov [r15 + off.gpr(rd)], rax
);
gld64(ops, cache, off, RAX, ra);
dynasm!(ops ; .arch x64 ; neg rax);
gst64(ops, cache, off, rd, RAX);
state.retire();
Emit::Native
}
@@ -346,9 +440,11 @@ pub fn try_emit_native(
let bf = instr.crfd();
let imm = instr.simm16() as i32; // 16-bit sign-extended fits i32; cmp sign-extends to 64
if instr.l() {
dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.gpr(ra)] ; cmp rax, imm);
gld64(ops, cache, off, RAX, ra);
dynasm!(ops ; .arch x64 ; cmp rax, imm);
} else {
dynasm!(ops ; .arch x64 ; mov eax, [r15 + off.gpr(ra)] ; cmp eax, imm);
gld32(ops, cache, off, RAX, ra);
dynasm!(ops ; .arch x64 ; cmp eax, imm);
}
emit_cr_from_flags(ops, off, bf, /*signed=*/ true);
state.retire();
@@ -359,9 +455,11 @@ pub fn try_emit_native(
let bf = instr.crfd();
let imm = instr.uimm16() as i32; // 0..65535 -> positive i32 (zext == small positive)
if instr.l() {
dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.gpr(ra)] ; cmp rax, imm);
gld64(ops, cache, off, RAX, ra);
dynasm!(ops ; .arch x64 ; cmp rax, imm);
} else {
dynasm!(ops ; .arch x64 ; mov eax, [r15 + off.gpr(ra)] ; cmp eax, imm);
gld32(ops, cache, off, RAX, ra);
dynasm!(ops ; .arch x64 ; cmp eax, imm);
}
emit_cr_from_flags(ops, off, bf, /*signed=*/ false);
state.retire();
@@ -371,17 +469,13 @@ pub fn try_emit_native(
PpcOpcode::cmp => {
let bf = instr.crfd();
if instr.l() {
dynasm!(ops ; .arch x64
; mov rax, [r15 + off.gpr(ra)]
; mov rcx, [r15 + off.gpr(rb)]
; cmp rax, rcx
);
gld64(ops, cache, off, RAX, ra);
gld64(ops, cache, off, RCX, rb);
dynasm!(ops ; .arch x64 ; cmp rax, rcx);
} else {
dynasm!(ops ; .arch x64
; mov eax, [r15 + off.gpr(ra)]
; mov ecx, [r15 + off.gpr(rb)]
; cmp eax, ecx
);
gld32(ops, cache, off, RAX, ra);
gld32(ops, cache, off, RCX, rb);
dynasm!(ops ; .arch x64 ; cmp eax, ecx);
}
emit_cr_from_flags(ops, off, bf, /*signed=*/ true);
state.retire();
@@ -391,17 +485,13 @@ pub fn try_emit_native(
PpcOpcode::cmpl => {
let bf = instr.crfd();
if instr.l() {
dynasm!(ops ; .arch x64
; mov rax, [r15 + off.gpr(ra)]
; mov rcx, [r15 + off.gpr(rb)]
; cmp rax, rcx
);
gld64(ops, cache, off, RAX, ra);
gld64(ops, cache, off, RCX, rb);
dynasm!(ops ; .arch x64 ; cmp rax, rcx);
} else {
dynasm!(ops ; .arch x64
; mov eax, [r15 + off.gpr(ra)]
; mov ecx, [r15 + off.gpr(rb)]
; cmp eax, ecx
);
gld32(ops, cache, off, RAX, ra);
gld32(ops, cache, off, RCX, rb);
dynasm!(ops ; .arch x64 ; cmp eax, ecx);
}
emit_cr_from_flags(ops, off, bf, /*signed=*/ false);
state.retire();
@@ -410,88 +500,88 @@ pub fn try_emit_native(
// ===== Loads (D-form): rD = EXT(mem[(rA==0?0:gpr[rA]) + EXTS(D)]) =====
PpcOpcode::lbz => {
emit_load(ops, off, state, helpers.read_u8, ra, rd, instr.d(), Ext::Zx8);
emit_load(ops, cache, off, state, helpers.read_u8, ra, rd, instr.d(), Ext::Zx8);
Emit::Native
}
PpcOpcode::lhz => {
emit_load(ops, off, state, helpers.read_u16, ra, rd, instr.d(), Ext::Zx16);
emit_load(ops, cache, off, state, helpers.read_u16, ra, rd, instr.d(), Ext::Zx16);
Emit::Native
}
PpcOpcode::lha => {
emit_load(ops, off, state, helpers.read_u16, ra, rd, instr.d(), Ext::Sx16);
emit_load(ops, cache, off, state, helpers.read_u16, ra, rd, instr.d(), Ext::Sx16);
Emit::Native
}
PpcOpcode::lwz => {
emit_load(ops, off, state, helpers.read_u32, ra, rd, instr.d(), Ext::Zx32);
emit_load(ops, cache, off, state, helpers.read_u32, ra, rd, instr.d(), Ext::Zx32);
Emit::Native
}
// ===== Stores: mem[(rA==0?0:gpr[rA]) + EXTS(D/DS)] = gpr[rS] =====
PpcOpcode::stb => {
emit_store(ops, off, state, helpers.store_u8, ra, rd, instr.d());
emit_store(ops, cache, off, state, helpers.store_u8, ra, rd, instr.d());
Emit::Native
}
PpcOpcode::sth => {
emit_store(ops, off, state, helpers.store_u16, ra, rd, instr.d());
emit_store(ops, cache, off, state, helpers.store_u16, ra, rd, instr.d());
Emit::Native
}
PpcOpcode::stw => {
emit_store(ops, off, state, helpers.store_u32, ra, rd, instr.d());
emit_store(ops, cache, off, state, helpers.store_u32, ra, rd, instr.d());
Emit::Native
}
PpcOpcode::std => {
// DS-form displacement (14-bit signed << 2).
emit_store(ops, off, state, helpers.store_u64, ra, rd, instr.ds());
emit_store(ops, cache, off, state, helpers.store_u64, ra, rd, instr.ds());
Emit::Native
}
// ld/ldx: 64-bit integer load. rD = mem.read_u64(ea). (DS-form / indexed.)
PpcOpcode::ld => {
emit_ea(ops, off, ra, instr.ds());
emit_int_load64(ops, off, helpers.read_u64, rd);
emit_ea(ops, cache, off, ra, instr.ds());
emit_int_load64(ops, cache, off, helpers.read_u64, rd);
state.retire();
Emit::Native
}
PpcOpcode::ldx => {
emit_ea_x(ops, off, ra, rb);
emit_int_load64(ops, off, helpers.read_u64, rd);
emit_ea_x(ops, cache, off, ra, rb);
emit_int_load64(ops, cache, off, helpers.read_u64, rd);
state.retire();
Emit::Native
}
// Indexed integer loads/stores: EA = (rA==0?0:gpr[rA]) + gpr[rB].
// Same extension/value semantics as the D-forms.
PpcOpcode::lwzx => {
emit_ea_x(ops, off, ra, rb);
emit_load_tail(ops, off, state, helpers.read_u32, rd, Ext::Zx32);
emit_ea_x(ops, cache, off, ra, rb);
emit_load_tail(ops, cache, off, state, helpers.read_u32, rd, Ext::Zx32);
Emit::Native
}
PpcOpcode::lhzx => {
emit_ea_x(ops, off, ra, rb);
emit_load_tail(ops, off, state, helpers.read_u16, rd, Ext::Zx16);
emit_ea_x(ops, cache, off, ra, rb);
emit_load_tail(ops, cache, off, state, helpers.read_u16, rd, Ext::Zx16);
Emit::Native
}
PpcOpcode::lhax => {
emit_ea_x(ops, off, ra, rb);
emit_load_tail(ops, off, state, helpers.read_u16, rd, Ext::Sx16);
emit_ea_x(ops, cache, off, ra, rb);
emit_load_tail(ops, cache, off, state, helpers.read_u16, rd, Ext::Sx16);
Emit::Native
}
PpcOpcode::lbzx => {
emit_ea_x(ops, off, ra, rb);
emit_load_tail(ops, off, state, helpers.read_u8, rd, Ext::Zx8);
emit_ea_x(ops, cache, off, ra, rb);
emit_load_tail(ops, cache, off, state, helpers.read_u8, rd, Ext::Zx8);
Emit::Native
}
PpcOpcode::stwx => {
emit_ea_x(ops, off, ra, rb);
emit_store_tail(ops, off, state, helpers.store_u32, rd);
emit_ea_x(ops, cache, off, ra, rb);
emit_store_tail(ops, cache, off, state, helpers.store_u32, rd);
Emit::Native
}
PpcOpcode::sthx => {
emit_ea_x(ops, off, ra, rb);
emit_store_tail(ops, off, state, helpers.store_u16, rd);
emit_ea_x(ops, cache, off, ra, rb);
emit_store_tail(ops, cache, off, state, helpers.store_u16, rd);
Emit::Native
}
PpcOpcode::stbx => {
emit_ea_x(ops, off, ra, rb);
emit_store_tail(ops, off, state, helpers.store_u8, rd);
emit_ea_x(ops, cache, off, ra, rb);
emit_store_tail(ops, cache, off, state, helpers.store_u8, rd);
Emit::Native
}
@@ -501,52 +591,52 @@ pub fn try_emit_native(
// integer forms (RA=0 -> 0). =====
// lfs/lfsx: fpr[rd] = mem.read_f32(ea) as f64.
PpcOpcode::lfs => {
emit_ea(ops, off, ra, instr.d());
emit_ea(ops, cache, off, ra, instr.d());
emit_fp_load(ops, off, helpers.read_f32_as_f64, rd);
state.retire();
Emit::Native
}
PpcOpcode::lfsx => {
emit_ea_x(ops, off, ra, rb);
emit_ea_x(ops, cache, off, ra, rb);
emit_fp_load(ops, off, helpers.read_f32_as_f64, rd);
state.retire();
Emit::Native
}
// lfd/lfdx: fpr[rd] = mem.read_f64(ea).
PpcOpcode::lfd => {
emit_ea(ops, off, ra, instr.d());
emit_ea(ops, cache, off, ra, instr.d());
emit_fp_load(ops, off, helpers.read_f64, rd);
state.retire();
Emit::Native
}
PpcOpcode::lfdx => {
emit_ea_x(ops, off, ra, rb);
emit_ea_x(ops, cache, off, ra, rb);
emit_fp_load(ops, off, helpers.read_f64, rd);
state.retire();
Emit::Native
}
// stfs/stfsx: mem.write_f32(ea, fpr[rs] as f32).
PpcOpcode::stfs => {
emit_ea(ops, off, ra, instr.d());
emit_ea(ops, cache, off, ra, instr.d());
emit_fp_store(ops, off, helpers.store_f32, rd);
state.retire();
Emit::Native
}
PpcOpcode::stfsx => {
emit_ea_x(ops, off, ra, rb);
emit_ea_x(ops, cache, off, ra, rb);
emit_fp_store(ops, off, helpers.store_f32, rd);
state.retire();
Emit::Native
}
// stfd/stfdx: mem.write_f64(ea, fpr[rs]).
PpcOpcode::stfd => {
emit_ea(ops, off, ra, instr.d());
emit_ea(ops, cache, off, ra, instr.d());
emit_fp_store(ops, off, helpers.store_f64, rd);
state.retire();
Emit::Native
}
PpcOpcode::stfdx => {
emit_ea_x(ops, off, ra, rb);
emit_ea_x(ops, cache, off, ra, rb);
emit_fp_store(ops, off, helpers.store_f64, rd);
state.retire();
Emit::Native
@@ -558,11 +648,12 @@ pub fn try_emit_native(
PpcOpcode::rlwinmx => {
let sh = instr.sh();
let mask = rlw_mask(instr.mb(), instr.me()) as i32;
dynasm!(ops ; .arch x64 ; mov eax, [r15 + off.gpr(rd)]);
gld32(ops, cache, off, RAX, rd);
if sh != 0 {
dynasm!(ops ; .arch x64 ; rol eax, sh as i8);
}
dynasm!(ops ; .arch x64 ; and eax, mask ; mov [r15 + off.gpr(ra)], rax);
dynasm!(ops ; .arch x64 ; and eax, mask);
gst64(ops, cache, off, ra, RAX);
if instr.rc_bit() {
emit_cr0_from_reg(ops, off, false);
}
@@ -573,17 +664,17 @@ pub fn try_emit_native(
PpcOpcode::rlwimix => {
let sh = instr.sh();
let mask = rlw_mask(instr.mb(), instr.me());
dynasm!(ops ; .arch x64 ; mov eax, [r15 + off.gpr(rd)]);
gld32(ops, cache, off, RAX, rd);
if sh != 0 {
dynasm!(ops ; .arch x64 ; rol eax, sh as i8);
}
gld32(ops, cache, off, RCX, ra);
dynasm!(ops ; .arch x64
; and eax, mask as i32
; mov ecx, [r15 + off.gpr(ra)]
; and ecx, !mask as i32
; or eax, ecx
; mov [r15 + off.gpr(ra)], rax
);
gst64(ops, cache, off, ra, RAX);
if instr.rc_bit() {
emit_cr0_from_reg(ops, off, false);
}
@@ -593,14 +684,14 @@ pub fn try_emit_native(
// rlwnm: like rlwinm but SH = RB[27:31] (runtime, masked to 0x1F).
PpcOpcode::rlwnmx => {
let mask = rlw_mask(instr.mb(), instr.me()) as i32;
gld32(ops, cache, off, RAX, rd);
gld32(ops, cache, off, RCX, rb);
dynasm!(ops ; .arch x64
; mov eax, [r15 + off.gpr(rd)]
; mov ecx, [r15 + off.gpr(rb)]
; and ecx, 0x1F
; rol eax, cl
; and eax, mask
; mov [r15 + off.gpr(ra)], rax
);
gst64(ops, cache, off, ra, RAX);
if instr.rc_bit() {
emit_cr0_from_reg(ops, off, false);
}
@@ -611,15 +702,15 @@ pub fn try_emit_native(
PpcOpcode::rldiclx => {
let sh = instr.sh64();
let mask = rld_mask_left(instr.mb_md()) as i64;
dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.gpr(rd)]);
gld64(ops, cache, off, RAX, rd);
if sh != 0 {
dynasm!(ops ; .arch x64 ; rol rax, sh as i8);
}
dynasm!(ops ; .arch x64
; mov rcx, QWORD mask
; and rax, rcx
; mov [r15 + off.gpr(ra)], rax
);
gst64(ops, cache, off, ra, RAX);
if instr.rc_bit() {
emit_cr0_from_reg(ops, off, true);
}
@@ -630,15 +721,15 @@ pub fn try_emit_native(
PpcOpcode::rldicrx => {
let sh = instr.sh64();
let mask = rld_mask_right(instr.mb_md()) as i64;
dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.gpr(rd)]);
gld64(ops, cache, off, RAX, rd);
if sh != 0 {
dynasm!(ops ; .arch x64 ; rol rax, sh as i8);
}
dynasm!(ops ; .arch x64
; mov rcx, QWORD mask
; and rax, rcx
; mov [r15 + off.gpr(ra)], rax
);
gst64(ops, cache, off, ra, RAX);
if instr.rc_bit() {
emit_cr0_from_reg(ops, off, true);
}
@@ -654,16 +745,16 @@ pub fn try_emit_native(
if instr.rc_bit() {
return Emit::Fallback;
}
gld32(ops, cache, off, RCX, rb);
gld32(ops, cache, off, RAX, rd);
dynasm!(ops ; .arch x64
; mov ecx, [r15 + off.gpr(rb)]
; and ecx, 0x3F
; mov eax, [r15 + off.gpr(rd)]
; shl eax, cl
; xor edx, edx
; cmp ecx, 32
; cmovae eax, edx
; mov [r15 + off.gpr(ra)], rax
);
gst64(ops, cache, off, ra, RAX);
state.retire();
Emit::Native
}
@@ -672,16 +763,16 @@ pub fn try_emit_native(
if instr.rc_bit() {
return Emit::Fallback;
}
gld32(ops, cache, off, RCX, rb);
gld32(ops, cache, off, RAX, rd);
dynasm!(ops ; .arch x64
; mov ecx, [r15 + off.gpr(rb)]
; and ecx, 0x3F
; mov eax, [r15 + off.gpr(rd)]
; shr eax, cl
; xor edx, edx
; cmp ecx, 32
; cmovae eax, edx
; mov [r15 + off.gpr(ra)], rax
);
gst64(ops, cache, off, ra, RAX);
state.retire();
Emit::Native
}
@@ -690,16 +781,16 @@ pub fn try_emit_native(
if instr.rc_bit() {
return Emit::Fallback;
}
gld64(ops, cache, off, RCX, rb);
gld64(ops, cache, off, RAX, rd);
dynasm!(ops ; .arch x64
; mov rcx, [r15 + off.gpr(rb)]
; and rcx, 0x7F
; mov rax, [r15 + off.gpr(rd)]
; shl rax, cl
; xor edx, edx
; cmp rcx, 64
; cmovae rax, rdx
; mov [r15 + off.gpr(ra)], rax
);
gst64(ops, cache, off, ra, RAX);
state.retire();
Emit::Native
}
@@ -708,16 +799,16 @@ pub fn try_emit_native(
if instr.rc_bit() {
return Emit::Fallback;
}
gld64(ops, cache, off, RCX, rb);
gld64(ops, cache, off, RAX, rd);
dynasm!(ops ; .arch x64
; mov rcx, [r15 + off.gpr(rb)]
; and rcx, 0x7F
; mov rax, [r15 + off.gpr(rd)]
; shr rax, cl
; xor edx, edx
; cmp rcx, 64
; cmovae rax, rdx
; mov [r15 + off.gpr(ra)], rax
);
gst64(ops, cache, off, ra, RAX);
state.retire();
Emit::Native
}
@@ -726,17 +817,24 @@ pub fn try_emit_native(
// mfctr/mtctr); every other SPR has side effects -> fallback. =====
PpcOpcode::mfspr => {
match instr.spr() {
8 => dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.lr] ; mov [r15 + off.gpr(rd)], rax),
9 => dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.ctr] ; mov [r15 + off.gpr(rd)], rax),
8 => dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.lr]),
9 => dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.ctr]),
_ => return Emit::Fallback,
}
gst64(ops, cache, off, rd, RAX);
state.retire();
Emit::Native
}
PpcOpcode::mtspr => {
match instr.spr() {
8 => dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.gpr(rd)] ; mov [r15 + off.lr], rax),
9 => dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.gpr(rd)] ; mov [r15 + off.ctr], rax),
8 => {
gld64(ops, cache, off, RAX, rd);
dynasm!(ops ; .arch x64 ; mov [r15 + off.lr], rax);
}
9 => {
gld64(ops, cache, off, RAX, rd);
dynasm!(ops ; .arch x64 ; mov [r15 + off.ctr], rax);
}
_ => return Emit::Fallback,
}
state.retire();
@@ -870,11 +968,11 @@ enum Ext {
/// address, so the low 32 bits of `rsi` are the guest EA (matching the
/// interpreter's `... as u32` truncation).
#[inline]
fn emit_ea(ops: &mut Asm, off: &Offsets, ra: usize, disp: i32) {
fn emit_ea(ops: &mut Asm, cache: &RegCache, off: &Offsets, ra: usize, disp: i32) {
if ra == 0 {
dynasm!(ops ; .arch x64 ; xor esi, esi);
} else {
dynasm!(ops ; .arch x64 ; mov rsi, [r15 + off.gpr(ra)]);
gld64(ops, cache, off, RSI, ra);
}
if disp != 0 {
dynasm!(ops ; .arch x64 ; add rsi, disp);
@@ -882,8 +980,10 @@ fn emit_ea(ops: &mut Asm, off: &Offsets, ra: usize, disp: i32) {
}
/// Load tail: `rsi=ea` already set; `call read_helper; extend; gpr[rd]=res`.
/// Cached regs survive the call (callee-saved; the mem helper doesn't touch
/// ctx.gpr), so the result store goes through the accessor.
#[inline]
fn emit_load_tail(ops: &mut Asm, off: &Offsets, state: &mut EmitState, helper: i64, rd: usize, ext: Ext) {
fn emit_load_tail(ops: &mut Asm, cache: &RegCache, off: &Offsets, state: &mut EmitState, helper: i64, rd: usize, ext: Ext) {
dynasm!(ops
; .arch x64
; mov rdi, rbx
@@ -898,16 +998,16 @@ fn emit_load_tail(ops: &mut Asm, off: &Offsets, state: &mut EmitState, helper: i
Ext::Sx16 => dynasm!(ops ; .arch x64 ; movsx eax, ax),
Ext::Zx32 => dynasm!(ops ; .arch x64 ; mov eax, eax),
}
dynasm!(ops ; .arch x64 ; mov [r15 + off.gpr(rd)], rax);
gst64(ops, cache, off, rd, RAX);
state.retire();
}
/// Store tail: `rsi=ea` already set; `rdx=gpr[rs]; call store_helper`.
#[inline]
fn emit_store_tail(ops: &mut Asm, off: &Offsets, state: &mut EmitState, helper: i64, rs: usize) {
fn emit_store_tail(ops: &mut Asm, cache: &RegCache, off: &Offsets, state: &mut EmitState, helper: i64, rs: usize) {
gld64(ops, cache, off, RDX, rs);
dynasm!(ops
; .arch x64
; mov rdx, [r15 + off.gpr(rs)]
; mov rdi, rbx
; mov rax, QWORD helper
; call rax
@@ -917,42 +1017,41 @@ fn emit_store_tail(ops: &mut Asm, off: &Offsets, state: &mut EmitState, helper:
/// D-form load: compute EA then the load tail.
#[inline]
fn emit_load(ops: &mut Asm, off: &Offsets, state: &mut EmitState, helper: i64, ra: usize, rd: usize, disp: i32, ext: Ext) {
emit_ea(ops, off, ra, disp);
emit_load_tail(ops, off, state, helper, rd, ext);
fn emit_load(ops: &mut Asm, cache: &RegCache, off: &Offsets, state: &mut EmitState, helper: i64, ra: usize, rd: usize, disp: i32, ext: Ext) {
emit_ea(ops, cache, off, ra, disp);
emit_load_tail(ops, cache, off, state, helper, rd, ext);
}
/// D-form store: compute EA then the store tail.
#[inline]
fn emit_store(ops: &mut Asm, off: &Offsets, state: &mut EmitState, helper: i64, ra: usize, rs: usize, disp: i32) {
emit_ea(ops, off, ra, disp);
emit_store_tail(ops, off, state, helper, rs);
fn emit_store(ops: &mut Asm, cache: &RegCache, off: &Offsets, state: &mut EmitState, helper: i64, ra: usize, rs: usize, disp: i32) {
emit_ea(ops, cache, off, ra, disp);
emit_store_tail(ops, cache, off, state, helper, rs);
}
/// Compute an indexed effective address `(rA==0 ? 0 : gpr[rA]) + gpr[rB]` into
/// `rsi` (low 32 = guest EA). Mirrors the interpreter's X-form EA.
#[inline]
fn emit_ea_x(ops: &mut Asm, off: &Offsets, ra: usize, rb: usize) {
fn emit_ea_x(ops: &mut Asm, cache: &RegCache, off: &Offsets, ra: usize, rb: usize) {
if ra == 0 {
dynasm!(ops ; .arch x64 ; mov rsi, [r15 + off.gpr(rb)]);
gld64(ops, cache, off, RSI, rb);
} else {
dynasm!(ops ; .arch x64
; mov rsi, [r15 + off.gpr(ra)]
; add rsi, [r15 + off.gpr(rb)]
);
gld64(ops, cache, off, RSI, ra);
gld64(ops, cache, off, RCX, rb);
dynasm!(ops ; .arch x64 ; add rsi, rcx);
}
}
/// Emit a 64-bit integer load: `rsi=ea; call read_u64; gpr[rd] = rax`.
/// (Caller sets `rsi` via `emit_ea`/`emit_ea_x` and calls `state.retire()`.)
#[inline]
fn emit_int_load64(ops: &mut Asm, off: &Offsets, helper: i64, rd: usize) {
fn emit_int_load64(ops: &mut Asm, cache: &RegCache, off: &Offsets, helper: i64, rd: usize) {
dynasm!(ops ; .arch x64
; mov rdi, rbx
; mov rax, QWORD helper
; call rax
; mov [r15 + off.gpr(rd)], rax
);
gst64(ops, cache, off, rd, RAX);
}
/// Emit an FP load: `rsi=ea; call helper (returns f64 bits in rax); fpr[rd]=rax`.
@@ -1042,10 +1141,10 @@ fn emit_cr_from_flags(ops: &mut Asm, off: &Offsets, field: usize, signed: bool)
/// Load `gpr[ra]` into `rax`, or zero `rax` when `ra == 0` (the PPC
/// arithmetic-D "RA=0 means literal 0" rule, statically known here).
#[inline]
fn load_ra_or_zero(ops: &mut Asm, off: &Offsets, ra: usize) {
fn load_ra_or_zero(ops: &mut Asm, off: &Offsets, cache: &RegCache, ra: usize) {
if ra == 0 {
dynasm!(ops ; .arch x64 ; xor eax, eax);
} else {
dynasm!(ops ; .arch x64 ; mov rax, [r15 + off.gpr(ra)]);
gld64(ops, cache, off, RAX, ra);
}
}

View File

@@ -302,22 +302,40 @@ fn compile_block(block: &DecodedBlock) -> CompiledBlock {
let mem_helpers = MemHelpers::resolve();
let helper = jit_interpret_one as usize as i64;
// Per-block register cache (a few hot GPRs pinned in r12/r13/r14). Empty
// unless enabled, in which case the accessors read/write the host regs and
// we load at the prologue / flush at exit / flush-reload around fallbacks.
let cache = if regcache_active() {
emit::RegCache::build(&instrs)
} else {
emit::RegCache::disabled()
};
// Whether this block actually caches any GPRs (only then do we spend the
// extra push/pop + load/flush; the default JIT path is unchanged).
let use_cache = !cache.active().is_empty();
let mut ops = dynasmrt::x64::Assembler::new().expect("dynasm assembler");
let entry = ops.offset();
let l_exit = ops.new_dynamic_label();
let l_cont = ops.new_dynamic_label();
// Prologue: save callee-saved regs we use, keep the stack 16-aligned before
// the helper calls (entry rsp%16==8; two pushes -> 8; `sub 8` -> 0), pin
// env in rbx and ctx in r15.
// Prologue: pin env in rbx, ctx in r15, keeping rsp 16-aligned for the helper
// `call`s (entry rsp%16==8). Without caching: 2 pushes + `sub 8`. With
// caching: 5 pushes (rbx/r15 + r12/r13/r14), already aligned.
dynasm!(ops ; .arch x64 ; push rbx ; push r15);
if use_cache {
dynasm!(ops ; .arch x64 ; push r12 ; push r13 ; push r14);
} else {
dynasm!(ops ; .arch x64 ; sub rsp, 8);
}
dynasm!(ops
; .arch x64
; push rbx
; push r15
; sub rsp, 8
; mov rbx, rdi
; mov r15, [rbx + off.env_ctx]
);
// Load the cached guest GPRs into their host regs (no-op if none).
emit::emit_cache_load(&mut ops, &cache, &off);
// Counter/pc deferral: native ops just accumulate `state.pending`; pc and the
// counters are materialized only at observability points (fallbacks, branch
@@ -327,7 +345,7 @@ fn compile_block(block: &DecodedBlock) -> CompiledBlock {
let mut state = emit::EmitState::new();
let mut tail_pc: Option<u32> = None;
for instr in instrs.iter() {
match emit::try_emit_native(&mut ops, &off, &mem_helpers, &mut state, instr) {
match emit::try_emit_native(&mut ops, &off, &mem_helpers, &mut state, &cache, instr) {
// Native non-branch: computation only; pc/counters deferred.
emit::Emit::Native => {
tail_pc = Some(instr.addr.wrapping_add(4));
@@ -356,6 +374,9 @@ fn compile_block(block: &DecodedBlock) -> CompiledBlock {
let addr = instr.addr as i32;
let instr_ptr = instr as *const DecodedInstr as usize as i64;
let expected_next = instr.addr.wrapping_add(4) as i32;
// The interpreter reads/writes ctx.gpr directly, so flush the cached
// regs to memory before the call and reload them after.
emit::emit_cache_flush(&mut ops, &cache, &off);
dynasm!(ops
; .arch x64
; mov DWORD [r15 + off.pc], addr
@@ -364,6 +385,10 @@ fn compile_block(block: &DecodedBlock) -> CompiledBlock {
; mov rsi, QWORD instr_ptr
; mov rax, QWORD helper
; call rax
);
emit::emit_cache_load(&mut ops, &cache, &off); // reload possibly-changed gprs
dynasm!(ops
; .arch x64
// determinism postlude: this instruction retired.
; inc QWORD [r15 + off.cycle]
; inc QWORD [r15 + off.timebase]
@@ -386,13 +411,23 @@ fn compile_block(block: &DecodedBlock) -> CompiledBlock {
dynasm!(ops ; .arch x64 ; mov DWORD [r15 + off.pc], p as i32);
}
// Natural end / discontinuity exit: Continue (eax=0). Shared epilogue.
// Natural end / discontinuity exit: Continue (eax=0). Shared epilogue
// flush the cached GPRs back to ctx.gpr (both l_cont and l_exit reach here),
// then restore the callee-saved regs.
dynasm!(ops
; .arch x64
; =>l_cont
; xor eax, eax
; =>l_exit
; add rsp, 8
);
emit::emit_cache_flush(&mut ops, &cache, &off);
if use_cache {
dynasm!(ops ; .arch x64 ; pop r14 ; pop r13 ; pop r12);
} else {
dynasm!(ops ; .arch x64 ; add rsp, 8);
}
dynasm!(ops
; .arch x64
; pop r15
; pop rbx
; ret
@@ -515,3 +550,38 @@ pub fn env_enabled() -> bool {
.unwrap_or(false)
})
}
/// Override to force the register cache on (1) / off (2) / env-decided (0).
/// Under `cfg(test)` it defaults to ON so every differential test exercises the
/// register-cache codegen path; production defaults to env-decided.
#[cfg(test)]
static REGCACHE_FORCE: std::sync::atomic::AtomicU8 = std::sync::atomic::AtomicU8::new(1);
#[cfg(not(test))]
static REGCACHE_FORCE: std::sync::atomic::AtomicU8 = std::sync::atomic::AtomicU8::new(0);
/// Force the register cache on (1) / off (2) / env-decided (0). Test-only.
#[cfg(test)]
pub(crate) fn set_regcache_force(state: u8) {
REGCACHE_FORCE.store(state, std::sync::atomic::Ordering::Relaxed);
}
/// Whether the per-block register cache is active this run
/// (`XENIA_JIT_REGCACHE=1|true|yes`), overridable in tests.
fn regcache_active() -> bool {
match REGCACHE_FORCE.load(std::sync::atomic::Ordering::Relaxed) {
1 => return true,
2 => return false,
_ => {}
}
use std::sync::OnceLock;
static ON: OnceLock<bool> = OnceLock::new();
*ON.get_or_init(|| {
std::env::var("XENIA_JIT_REGCACHE")
.ok()
.map(|v| {
let v = v.trim().to_ascii_lowercase();
v == "1" || v == "true" || v == "yes"
})
.unwrap_or(false)
})
}

View File

@@ -191,7 +191,7 @@ fn check(raw: u32, gpr: [u64; 32]) {
let helpers = crate::MemHelpers::resolve();
let mut probe = dynasmrt::x64::Assembler::new().unwrap();
assert!(
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &instr) != emit::Emit::Fallback,
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &emit::RegCache::disabled(), &instr) != emit::Emit::Fallback,
"opcode not natively emitted for raw={raw:#010x} ({:?})",
instr.opcode
);
@@ -238,7 +238,7 @@ fn check_cmp(raw: u32, gpr: [u64; 32], xer_so: u8) {
let helpers = crate::MemHelpers::resolve();
let mut probe = dynasmrt::x64::Assembler::new().unwrap();
assert!(
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &instr) != emit::Emit::Fallback,
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &emit::RegCache::disabled(), &instr) != emit::Emit::Fallback,
"compare not natively emitted raw={raw:#010x} ({:?})",
instr.opcode
);
@@ -361,7 +361,7 @@ fn check_rot(raw: u32, gpr: [u64; 32], xer_so: u8) {
let helpers = crate::MemHelpers::resolve();
let mut probe = dynasmrt::x64::Assembler::new().unwrap();
assert!(
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &instr) != emit::Emit::Fallback,
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &emit::RegCache::disabled(), &instr) != emit::Emit::Fallback,
"rotate not natively emitted raw={raw:#010x} ({:?})",
instr.opcode
);
@@ -467,7 +467,7 @@ fn check_branch(raw: u32, lr: u64, ctr: u64, cr_seed: u8) {
let helpers = crate::MemHelpers::resolve();
let mut probe = dynasmrt::x64::Assembler::new().unwrap();
assert_eq!(
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &instr),
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &emit::RegCache::disabled(), &instr),
emit::Emit::Branch,
"branch not emitted as Emit::Branch raw={raw:#010x} ({:?})",
instr.opcode
@@ -678,7 +678,7 @@ fn spr_lr_ctr_matches() {
let helpers = crate::MemHelpers::resolve();
let mut probe = dynasmrt::x64::Assembler::new().unwrap();
assert!(
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &instr) != emit::Emit::Fallback,
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &emit::RegCache::disabled(), &instr) != emit::Emit::Fallback,
"spr op not native raw={raw:#010x} ({:?}) spr={spr}",
instr.opcode
);
@@ -729,7 +729,7 @@ fn recording_and_hint_forms_fall_back() {
for raw in cases {
let instr = decode(raw, 0x8200_1000);
assert!(
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &instr) == emit::Emit::Fallback,
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &emit::RegCache::disabled(), &instr) == emit::Emit::Fallback,
"raw={raw:#010x} ({:?}) should fall back, not native",
instr.opcode
);
@@ -746,7 +746,7 @@ fn check_mem(raw: u32, gpr: [u64; 32]) {
let helpers = crate::MemHelpers::resolve();
let mut probe = dynasmrt::x64::Assembler::new().unwrap();
assert!(
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &instr) != emit::Emit::Fallback,
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &emit::RegCache::disabled(), &instr) != emit::Emit::Fallback,
"mem opcode not natively emitted raw={raw:#010x} ({:?})",
instr.opcode
);
@@ -803,7 +803,7 @@ fn check_fp(raw: u32, gpr: [u64; 32], fpr_bits: [u64; 32]) {
let helpers = crate::MemHelpers::resolve();
let mut probe = dynasmrt::x64::Assembler::new().unwrap();
assert!(
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &instr) != emit::Emit::Fallback,
emit::try_emit_native(&mut probe, &off, &helpers, &mut emit::EmitState::new(), &emit::RegCache::disabled(), &instr) != emit::Emit::Fallback,
"fp opcode not natively emitted raw={raw:#010x} ({:?})",
instr.opcode
);