/** ****************************************************************************** * Xenia : Xbox 360 Emulator Research Project * ****************************************************************************** * Copyright 2026 Ben Vanik. All rights reserved. * * Released under the BSD license - see LICENSE in the root for more details. * ****************************************************************************** */ #include "xenia/cpu/backend/a64/a64_sequences.h" #include "xenia/base/clock.h" #include "xenia/base/cvar.h" #include "xenia/base/memory.h" #include "xenia/cpu/backend/a64/a64_backend.h" #include "xenia/cpu/backend/a64/a64_emitter.h" #include "xenia/cpu/backend/a64/a64_op.h" #include "xenia/cpu/backend/a64/a64_seq_util.h" #include "xenia/cpu/backend/a64/a64_stack_layout.h" #include "xenia/cpu/hir/instr.h" #include "xenia/cpu/ppc/ppc_context.h" #include "xenia/cpu/processor.h" #include "xenia/cpu/xex_module.h" DECLARE_bool(emit_mmio_aware_stores_for_recorded_exception_addresses); DECLARE_bool(emit_inline_mmio_checks); namespace xe { namespace cpu { namespace backend { namespace a64 { volatile int anchor_memory = 0; static bool IsPossibleMMIOInstruction(A64Emitter& e, const hir::Instr* i) { if (!cvars::emit_mmio_aware_stores_for_recorded_exception_addresses) { return false; } uint32_t guest_address = i->GuestAddressFor(); if (!guest_address) { return false; } auto* guest_module = e.GuestModule(); if (!guest_module) { return false; } auto* flags = guest_module->GetInstructionAddressFlags(guest_address); return flags && flags->accessed_mmio; } // ============================================================================ // OPCODE_DELAY_EXECUTION // ============================================================================ struct DELAY_EXECUTION : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { e.yield(); } }; EMITTER_OPCODE_TABLE(OPCODE_DELAY_EXECUTION, DELAY_EXECUTION); // ============================================================================ // OPCODE_MEMORY_BARRIER // ============================================================================ struct MEMORY_BARRIER : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { e.dmb(Xbyak_aarch64::ISH); } }; EMITTER_OPCODE_TABLE(OPCODE_MEMORY_BARRIER, MEMORY_BARRIER); // ============================================================================ // OPCODE_CACHE_CONTROL // ============================================================================ struct CACHE_CONTROL : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { bool is_clflush = false, is_prefetch = false, is_prefetchw = false; switch (CacheControlType(i.instr->flags)) { case CacheControlType::CACHE_CONTROL_TYPE_DATA_TOUCH: is_prefetch = true; break; case CacheControlType::CACHE_CONTROL_TYPE_DATA_TOUCH_FOR_STORE: is_prefetchw = true; break; case CacheControlType::CACHE_CONTROL_TYPE_DATA_STORE: case CacheControlType::CACHE_CONTROL_TYPE_DATA_STORE_AND_FLUSH: is_clflush = true; break; default: return; } auto addr = ComputeMemoryAddress(e, i.src1); e.add(e.x0, e.GetMembaseReg(), addr); size_t cache_line_size = i.src2.value; if (is_clflush) { // dc civac, x0 e.sys(0b011, 0b0111, 0b1110, 0b001, e.x0); } if (is_prefetch) { e.prfm(Xbyak_aarch64::PLDL1KEEP, ptr(e.x0)); } else if (is_prefetchw) { e.prfm(Xbyak_aarch64::PSTL1KEEP, ptr(e.x0)); } if (cache_line_size >= 128) { e.eor(e.x0, e.x0, 64); if (is_clflush) { // dc civac, x0 e.sys(0b011, 0b0111, 0b1110, 0b001, e.x0); } if (is_prefetch) { e.prfm(Xbyak_aarch64::PLDL1KEEP, ptr(e.x0)); } else if (is_prefetchw) { e.prfm(Xbyak_aarch64::PSTL1KEEP, ptr(e.x0)); } } } }; EMITTER_OPCODE_TABLE(OPCODE_CACHE_CONTROL, CACHE_CONTROL); template static void MMIOAwareStore(void* _ctx, unsigned int guestaddr, T value) { if (swap) { value = xe::byte_swap(value); } if (guestaddr >= 0xE0000000) { guestaddr += 0x1000; } auto ctx = reinterpret_cast(_ctx); auto gaddr = ctx->processor->memory()->LookupVirtualMappedRange(guestaddr); if (!gaddr) { *reinterpret_cast(ctx->virtual_membase + guestaddr) = value; } else { value = xe::byte_swap(value); gaddr->write(nullptr, gaddr->callback_context, guestaddr, value); } } template static T MMIOAwareLoad(void* _ctx, unsigned int guestaddr) { T value; if (guestaddr >= 0xE0000000) { guestaddr += 0x1000; } auto ctx = reinterpret_cast(_ctx); auto gaddr = ctx->processor->memory()->LookupVirtualMappedRange(guestaddr); if (!gaddr) { value = *reinterpret_cast(ctx->virtual_membase + guestaddr); if (swap) { value = xe::byte_swap(value); } } else { value = gaddr->read(nullptr, gaddr->callback_context, guestaddr); } return value; } // ============================================================================ // OPCODE_LOAD // ============================================================================ struct LOAD_I8 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); e.ldrb(i.dest, ptr(e.GetMembaseReg(), addr)); } }; struct LOAD_I16 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); e.ldrh(i.dest, ptr(e.GetMembaseReg(), addr)); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { e.rev16(i.dest, i.dest); } } }; struct LOAD_I32 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { if (IsPossibleMMIOInstruction(e, i.instr)) { void* mmio_fn = (void*)&MMIOAwareLoad; if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { mmio_fn = (void*)&MMIOAwareLoad; } if (i.src1.is_constant) { e.mov(e.w1, static_cast(static_cast(i.src1.constant()))); } else { e.mov(e.w1, WReg(i.src1.reg().getIdx())); } e.CallNativeSafe(mmio_fn); e.mov(i.dest, e.w0); return; } if (cvars::emit_inline_mmio_checks) { if (i.src1.is_constant) { e.mov(e.w17, static_cast(static_cast(i.src1.constant()))); } else { e.mov(e.w17, WReg(i.src1.reg().getIdx())); } auto& normal_access = e.NewCachedLabel(); auto& done = e.NewCachedLabel(); e.mov(e.w0, 0x7FC00000u); e.cmp(e.w17, e.w0); e.b(LO, normal_access); e.mov(e.w0, 0x7FFFFFFFu); e.cmp(e.w17, e.w0); e.b(HI, normal_access); // MMIO path void* mmio_fn = (void*)&MMIOAwareLoad; if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { mmio_fn = (void*)&MMIOAwareLoad; } e.mov(e.w1, e.w17); e.CallNativeSafe(mmio_fn); e.mov(i.dest, e.w0); e.b(done); e.L(normal_access); { auto addr = ComputeMemoryAddress(e, i.src1); e.ldr(i.dest, ptr(e.GetMembaseReg(), addr)); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { e.rev(i.dest, i.dest); } } e.L(done); } else { auto addr = ComputeMemoryAddress(e, i.src1); e.ldr(i.dest, ptr(e.GetMembaseReg(), addr)); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { e.rev(i.dest, i.dest); } } } }; struct LOAD_I64 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); e.ldr(i.dest, ptr(e.GetMembaseReg(), addr)); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { e.rev(i.dest, i.dest); } } }; struct LOAD_F32 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { e.ldr(e.w0, ptr(e.GetMembaseReg(), addr)); e.rev(e.w0, e.w0); e.fmov(i.dest, e.w0); } else { e.ldr(i.dest, ptr(e.GetMembaseReg(), addr)); } } }; struct LOAD_F64 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { e.ldr(e.x0, ptr(e.GetMembaseReg(), addr)); e.rev(e.x0, e.x0); e.fmov(i.dest, e.x0); } else { e.ldr(i.dest, ptr(e.GetMembaseReg(), addr)); } } }; struct LOAD_V128 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); e.ldr(i.dest, ptr(e.GetMembaseReg(), addr)); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { // Reverse bytes within each 32-bit word (PPC BE -> ARM64 LE). auto idx = i.dest.reg().getIdx(); e.rev32(VReg16B(idx), VReg16B(idx)); } } }; EMITTER_OPCODE_TABLE(OPCODE_LOAD, LOAD_I8, LOAD_I16, LOAD_I32, LOAD_I64, LOAD_F32, LOAD_F64, LOAD_V128); // ============================================================================ // OPCODE_STORE // ============================================================================ struct STORE_I8 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); if (i.src2.is_constant) { e.mov(e.w17, static_cast(i.src2.constant() & 0xFF)); e.strb(e.w17, ptr(e.GetMembaseReg(), addr)); } else { e.strb(i.src2, ptr(e.GetMembaseReg(), addr)); } } }; struct STORE_I16 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { if (i.src2.is_constant) { uint16_t val = xe::byte_swap(static_cast(i.src2.constant())); e.mov(e.w17, static_cast(val)); } else { e.rev16(e.w17, i.src2); } e.strh(e.w17, ptr(e.GetMembaseReg(), addr)); } else { if (i.src2.is_constant) { e.mov(e.w17, static_cast(i.src2.constant() & 0xFFFF)); e.strh(e.w17, ptr(e.GetMembaseReg(), addr)); } else { e.strh(i.src2, ptr(e.GetMembaseReg(), addr)); } } } }; struct STORE_I32 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { if (IsPossibleMMIOInstruction(e, i.instr)) { void* mmio_fn = (void*)&MMIOAwareStore; if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { mmio_fn = (void*)&MMIOAwareStore; } if (i.src1.is_constant) { e.mov(e.w1, static_cast(static_cast(i.src1.constant()))); } else { e.mov(e.w1, WReg(i.src1.reg().getIdx())); } if (i.src2.is_constant) { e.mov(e.w2, static_cast(static_cast(i.src2.constant()))); } else { e.mov(e.w2, i.src2); } e.CallNativeSafe(mmio_fn); return; } if (cvars::emit_inline_mmio_checks) { if (i.src1.is_constant) { e.mov(e.w17, static_cast(static_cast(i.src1.constant()))); } else { e.mov(e.w17, WReg(i.src1.reg().getIdx())); } auto& normal_access = e.NewCachedLabel(); auto& done = e.NewCachedLabel(); e.mov(e.w0, 0x7FC00000u); e.cmp(e.w17, e.w0); e.b(LO, normal_access); e.mov(e.w0, 0x7FFFFFFFu); e.cmp(e.w17, e.w0); e.b(HI, normal_access); // MMIO path — copy value to w2 before w1 in case src2 is in w1 void* mmio_fn = (void*)&MMIOAwareStore; if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { mmio_fn = (void*)&MMIOAwareStore; } if (i.src2.is_constant) { e.mov(e.w2, static_cast(static_cast(i.src2.constant()))); } else { e.mov(e.w2, i.src2); } e.mov(e.w1, e.w17); e.CallNativeSafe(mmio_fn); e.b(done); e.L(normal_access); { auto addr = ComputeMemoryAddress(e, i.src1); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { if (i.src2.is_constant) { uint32_t val = xe::byte_swap(static_cast(i.src2.constant())); e.mov(e.w17, static_cast(val)); } else { e.rev(e.w17, i.src2); } e.str(e.w17, ptr(e.GetMembaseReg(), addr)); } else { if (i.src2.is_constant) { e.mov(e.w17, static_cast( static_cast(i.src2.constant()))); e.str(e.w17, ptr(e.GetMembaseReg(), addr)); } else { e.str(i.src2, ptr(e.GetMembaseReg(), addr)); } } } e.L(done); } else { auto addr = ComputeMemoryAddress(e, i.src1); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { if (i.src2.is_constant) { uint32_t val = xe::byte_swap(static_cast(i.src2.constant())); e.mov(e.w17, static_cast(val)); } else { e.rev(e.w17, i.src2); } e.str(e.w17, ptr(e.GetMembaseReg(), addr)); } else { if (i.src2.is_constant) { e.mov(e.w17, static_cast( static_cast(i.src2.constant()))); e.str(e.w17, ptr(e.GetMembaseReg(), addr)); } else { e.str(i.src2, ptr(e.GetMembaseReg(), addr)); } } } } }; struct STORE_I64 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { if (i.src2.is_constant) { uint64_t val = xe::byte_swap(static_cast(i.src2.constant())); e.mov(e.x17, val); } else { e.rev(e.x17, i.src2); } e.str(e.x17, ptr(e.GetMembaseReg(), addr)); } else { if (i.src2.is_constant) { e.mov(e.x17, static_cast(i.src2.constant())); e.str(e.x17, ptr(e.GetMembaseReg(), addr)); } else { e.str(i.src2, ptr(e.GetMembaseReg(), addr)); } } } }; struct STORE_F32 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { if (i.src2.is_constant) { uint32_t val = xe::byte_swap(static_cast(i.src2.value->constant.i32)); e.mov(e.w17, static_cast(val)); } else { e.fmov(e.w17, i.src2); e.rev(e.w17, e.w17); } e.str(e.w17, ptr(e.GetMembaseReg(), addr)); } else { if (i.src2.is_constant) { e.mov(e.w17, static_cast(i.src2.value->constant.i32)); e.str(e.w17, ptr(e.GetMembaseReg(), addr)); } else { e.str(i.src2, ptr(e.GetMembaseReg(), addr)); } } } }; struct STORE_F64 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { if (i.src2.is_constant) { uint64_t val = xe::byte_swap(static_cast(i.src2.value->constant.i64)); e.mov(e.x17, val); } else { e.fmov(e.x17, i.src2); e.rev(e.x17, e.x17); } e.str(e.x17, ptr(e.GetMembaseReg(), addr)); } else { if (i.src2.is_constant) { e.mov(e.x17, static_cast(i.src2.value->constant.i64)); e.str(e.x17, ptr(e.GetMembaseReg(), addr)); } else { e.str(i.src2, ptr(e.GetMembaseReg(), addr)); } } } }; struct STORE_V128 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { // ComputeMemoryAddress may return x0, and LoadV128Const/SrcVReg clobber // x0, so save the address to x17 when we need to load a constant source. bool need_src_load = i.src2.is_constant || (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP); auto addr = ComputeMemoryAddress(e, i.src1); if (need_src_load) { e.mov(e.x17, addr); addr = e.x17; } if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { // Reverse bytes within each 32-bit word, store via scratch v0. int idx = SrcVReg(e, i.src2, 0); e.rev32(VReg16B(0), VReg16B(idx)); e.str(QReg(0), ptr(e.GetMembaseReg(), addr)); } else { if (i.src2.is_constant) { LoadV128Const(e, 0, i.src2.constant()); e.str(QReg(0), ptr(e.GetMembaseReg(), addr)); } else { e.str(i.src2, ptr(e.GetMembaseReg(), addr)); } } } }; EMITTER_OPCODE_TABLE(OPCODE_STORE, STORE_I8, STORE_I16, STORE_I32, STORE_I64, STORE_F32, STORE_F64, STORE_V128); // ============================================================================ // OPCODE_LOAD_CLOCK // ============================================================================ struct LOAD_CLOCK : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { // Call QueryGuestTickCount which updates the clock from host ticks. // Reading the cached pointer directly would return stale values for // consecutive mftb instructions. e.CallNative(reinterpret_cast(LoadClock)); e.mov(i.dest, e.x0); } static uint64_t LoadClock(void* raw_context) { return Clock::QueryGuestTickCount(); } }; EMITTER_OPCODE_TABLE(OPCODE_LOAD_CLOCK, LOAD_CLOCK); // ============================================================================ // OPCODE_LOAD_OFFSET / OPCODE_STORE_OFFSET // ============================================================================ struct LOAD_OFFSET_I8 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { AddGuestMemoryOffset(e, ComputeMemoryAddress(e, i.src1), i.src2); e.ldrb(i.dest, ptr(e.GetMembaseReg(), e.x0)); } }; struct LOAD_OFFSET_I16 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { AddGuestMemoryOffset(e, ComputeMemoryAddress(e, i.src1), i.src2); e.ldrh(i.dest, ptr(e.GetMembaseReg(), e.x0)); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { e.rev16(i.dest, i.dest); } } }; struct LOAD_OFFSET_I32 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { if (IsPossibleMMIOInstruction(e, i.instr)) { void* mmio_fn = (void*)&MMIOAwareLoad; if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { mmio_fn = (void*)&MMIOAwareLoad; } if (i.src1.is_constant) { e.mov(e.w1, static_cast(static_cast(i.src1.constant()))); } else { e.mov(e.w1, WReg(i.src1.reg().getIdx())); } if (i.src2.is_constant) { e.mov(e.w17, static_cast(static_cast(i.src2.constant()))); } else { e.mov(e.w17, WReg(i.src2.reg().getIdx())); } e.add(e.w1, e.w1, e.w17); e.CallNativeSafe(mmio_fn); e.mov(i.dest, e.w0); return; } if (cvars::emit_inline_mmio_checks) { // Compute raw guest address (src1 + src2) in w17 for range check. if (i.src1.is_constant) { e.mov(e.w17, static_cast(static_cast(i.src1.constant()))); } else { e.mov(e.w17, WReg(i.src1.reg().getIdx())); } if (i.src2.is_constant) { uint32_t offset = static_cast(i.src2.constant()); if (offset != 0) { e.mov(e.w0, static_cast(offset)); e.add(e.w17, e.w17, e.w0); } } else { e.add(e.w17, e.w17, WReg(i.src2.reg().getIdx())); } auto& normal_access = e.NewCachedLabel(); auto& done = e.NewCachedLabel(); e.mov(e.w0, 0x7FC00000u); e.cmp(e.w17, e.w0); e.b(LO, normal_access); e.mov(e.w0, 0x7FFFFFFFu); e.cmp(e.w17, e.w0); e.b(HI, normal_access); // MMIO path void* mmio_fn = (void*)&MMIOAwareLoad; if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { mmio_fn = (void*)&MMIOAwareLoad; } e.mov(e.w1, e.w17); e.CallNativeSafe(mmio_fn); e.mov(i.dest, e.w0); e.b(done); e.L(normal_access); { AddGuestMemoryOffset(e, ComputeMemoryAddress(e, i.src1), i.src2); e.ldr(i.dest, ptr(e.GetMembaseReg(), e.x0)); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { e.rev(i.dest, i.dest); } } e.L(done); } else { AddGuestMemoryOffset(e, ComputeMemoryAddress(e, i.src1), i.src2); e.ldr(i.dest, ptr(e.GetMembaseReg(), e.x0)); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { e.rev(i.dest, i.dest); } } } }; struct LOAD_OFFSET_I64 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { AddGuestMemoryOffset(e, ComputeMemoryAddress(e, i.src1), i.src2); e.ldr(i.dest, ptr(e.GetMembaseReg(), e.x0)); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { e.rev(i.dest, i.dest); } } }; EMITTER_OPCODE_TABLE(OPCODE_LOAD_OFFSET, LOAD_OFFSET_I8, LOAD_OFFSET_I16, LOAD_OFFSET_I32, LOAD_OFFSET_I64); struct STORE_OFFSET_I8 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { AddGuestMemoryOffset(e, ComputeMemoryAddress(e, i.src1), i.src2); if (i.src3.is_constant) { e.mov(e.w17, static_cast(i.src3.constant() & 0xFF)); e.strb(e.w17, ptr(e.GetMembaseReg(), e.x0)); } else { e.strb(i.src3, ptr(e.GetMembaseReg(), e.x0)); } } }; struct STORE_OFFSET_I16 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { AddGuestMemoryOffset(e, ComputeMemoryAddress(e, i.src1), i.src2); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { if (i.src3.is_constant) { uint16_t val = xe::byte_swap(static_cast(i.src3.constant())); e.mov(e.w17, static_cast(val)); } else { e.rev16(e.w17, i.src3); } e.strh(e.w17, ptr(e.GetMembaseReg(), e.x0)); } else { if (i.src3.is_constant) { e.mov(e.w17, static_cast(i.src3.constant() & 0xFFFF)); e.strh(e.w17, ptr(e.GetMembaseReg(), e.x0)); } else { e.strh(i.src3, ptr(e.GetMembaseReg(), e.x0)); } } } }; struct STORE_OFFSET_I32 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { if (IsPossibleMMIOInstruction(e, i.instr)) { void* mmio_fn = (void*)&MMIOAwareStore; if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { mmio_fn = (void*)&MMIOAwareStore; } if (i.src1.is_constant) { e.mov(e.w1, static_cast(static_cast(i.src1.constant()))); } else { e.mov(e.w1, WReg(i.src1.reg().getIdx())); } if (i.src2.is_constant) { e.mov(e.w17, static_cast(static_cast(i.src2.constant()))); } else { e.mov(e.w17, WReg(i.src2.reg().getIdx())); } e.add(e.w1, e.w1, e.w17); if (i.src3.is_constant) { e.mov(e.w2, static_cast(static_cast(i.src3.constant()))); } else { e.mov(e.w2, i.src3); } e.CallNativeSafe(mmio_fn); return; } if (cvars::emit_inline_mmio_checks) { // Compute raw guest address (src1 + src2) in w17 for range check. if (i.src1.is_constant) { e.mov(e.w17, static_cast(static_cast(i.src1.constant()))); } else { e.mov(e.w17, WReg(i.src1.reg().getIdx())); } if (i.src2.is_constant) { uint32_t offset = static_cast(i.src2.constant()); if (offset != 0) { e.mov(e.w0, static_cast(offset)); e.add(e.w17, e.w17, e.w0); } } else { e.add(e.w17, e.w17, WReg(i.src2.reg().getIdx())); } auto& normal_access = e.NewCachedLabel(); auto& done = e.NewCachedLabel(); e.mov(e.w0, 0x7FC00000u); e.cmp(e.w17, e.w0); e.b(LO, normal_access); e.mov(e.w0, 0x7FFFFFFFu); e.cmp(e.w17, e.w0); e.b(HI, normal_access); // MMIO path — copy value to w2 before w1 in case src3 is in w1 void* mmio_fn = (void*)&MMIOAwareStore; if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { mmio_fn = (void*)&MMIOAwareStore; } if (i.src3.is_constant) { e.mov(e.w2, static_cast(static_cast(i.src3.constant()))); } else { e.mov(e.w2, i.src3); } e.mov(e.w1, e.w17); e.CallNativeSafe(mmio_fn); e.b(done); e.L(normal_access); { AddGuestMemoryOffset(e, ComputeMemoryAddress(e, i.src1), i.src2); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { if (i.src3.is_constant) { uint32_t val = xe::byte_swap(static_cast(i.src3.constant())); e.mov(e.w17, static_cast(val)); } else { e.rev(e.w17, i.src3); } e.str(e.w17, ptr(e.GetMembaseReg(), e.x0)); } else { if (i.src3.is_constant) { e.mov(e.w17, static_cast( static_cast(i.src3.constant()))); e.str(e.w17, ptr(e.GetMembaseReg(), e.x0)); } else { e.str(i.src3, ptr(e.GetMembaseReg(), e.x0)); } } } e.L(done); } else { AddGuestMemoryOffset(e, ComputeMemoryAddress(e, i.src1), i.src2); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { if (i.src3.is_constant) { uint32_t val = xe::byte_swap(static_cast(i.src3.constant())); e.mov(e.w17, static_cast(val)); } else { e.rev(e.w17, i.src3); } e.str(e.w17, ptr(e.GetMembaseReg(), e.x0)); } else { if (i.src3.is_constant) { e.mov(e.w17, static_cast( static_cast(i.src3.constant()))); e.str(e.w17, ptr(e.GetMembaseReg(), e.x0)); } else { e.str(i.src3, ptr(e.GetMembaseReg(), e.x0)); } } } } }; struct STORE_OFFSET_I64 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { AddGuestMemoryOffset(e, ComputeMemoryAddress(e, i.src1), i.src2); if (i.instr->flags & LoadStoreFlags::LOAD_STORE_BYTE_SWAP) { if (i.src3.is_constant) { uint64_t val = xe::byte_swap(static_cast(i.src3.constant())); e.mov(e.x17, val); } else { e.rev(e.x17, i.src3); } e.str(e.x17, ptr(e.GetMembaseReg(), e.x0)); } else { if (i.src3.is_constant) { e.mov(e.x17, static_cast(i.src3.constant())); e.str(e.x17, ptr(e.GetMembaseReg(), e.x0)); } else { e.str(i.src3, ptr(e.GetMembaseReg(), e.x0)); } } } }; EMITTER_OPCODE_TABLE(OPCODE_STORE_OFFSET, STORE_OFFSET_I8, STORE_OFFSET_I16, STORE_OFFSET_I32, STORE_OFFSET_I64); // ============================================================================ // OPCODE_MEMSET // ============================================================================ static const bool zva_enable = (xe_cpu_mrs(DCZID_EL0) & 0b1'0000) == 0; static const uint64_t zva_length = (4ULL << (xe_cpu_mrs(DCZID_EL0) & 0b0'1111)); struct MEMSET_I64 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { assert_true(i.src2.is_constant); assert_true(i.src3.is_constant); assert_true(i.src2.constant() == 0); // memset(membase + guest_addr, 0, length) // Only used by dcbz/dcbz128: constant zero value, constant aligned size. auto addr = ComputeMemoryAddress(e, i.src1); e.add(e.x0, e.GetMembaseReg(), addr); const uint64_t len = i.src3.constant(); uint64_t off = 0; // Use `dc zva` if it writes more bytes at a time than STP if (zva_enable && len >= zva_length && zva_length > 16) { for (; off + zva_length <= len; off += zva_length) { // dc zva, x0 e.sys(0b011, 0b0111, 0b0100, 0b001, e.x0); if (off + zva_length < len) { e.add(e.x0, e.x0, zva_length); } } } // Inline with STP xzr, xzr pairs (16 bytes each) for (; off + 16 <= len; off += 16) { e.stp(e.xzr, e.xzr, AdrPostImm(e.x0, 16)); } // Handle remaining bytes (0-15) if (off + 8 <= len) { e.str(e.xzr, AdrPostImm(e.x0, 8)); off += 8; } if (off + 4 <= len) { e.str(e.wzr, AdrPostImm(e.x0, 4)); off += 4; } // Byte loop for any remaining 0-3 bytes for (; off + 1 <= len; off += 1) { e.strb(e.wzr, AdrPostImm(e.x0, 1)); } } }; EMITTER_OPCODE_TABLE(OPCODE_MEMSET, MEMSET_I64); // ============================================================================ // OPCODE_ATOMIC_EXCHANGE // ============================================================================ // Note: src1 is a HOST address (not guest), matching the x64 backend. struct ATOMIC_EXCHANGE_I8 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { // src1 is already a host address. if (i.src1.is_constant) { e.mov(e.x4, i.src1.constant()); } else { e.mov(e.x4, i.src1); } if (i.src2.is_constant) { e.mov(e.w0, static_cast( static_cast(i.src2.constant()) & 0xFF)); } else { e.and_(e.w0, i.src2, 0xFF); } if (e.IsFeatureEnabled(kA64EmitLSE)) { e.swpalb(e.w0, i.dest, ptr(e.x4)); return; } auto& retry = e.NewCachedLabel(); e.L(retry); e.ldaxrb(e.w1, ptr(e.x4)); e.stlxrb(e.w2, e.w0, ptr(e.x4)); e.cbnz(e.w2, retry); e.mov(i.dest, e.w1); } }; struct ATOMIC_EXCHANGE_I16 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { if (i.src1.is_constant) { e.mov(e.x4, i.src1.constant()); } else { e.mov(e.x4, i.src1); } if (i.src2.is_constant) { e.mov(e.w0, static_cast( static_cast(i.src2.constant()) & 0xFFFF)); } else { e.and_(e.w0, i.src2, 0xFFFF); } if (e.IsFeatureEnabled(kA64EmitLSE)) { e.swpalh(e.w0, i.dest, ptr(e.x4)); return; } auto& retry = e.NewCachedLabel(); e.L(retry); e.ldaxrh(e.w1, ptr(e.x4)); e.stlxrh(e.w2, e.w0, ptr(e.x4)); e.cbnz(e.w2, retry); e.mov(i.dest, e.w1); } }; struct ATOMIC_EXCHANGE_I32 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { // src1 is a host address (not guest). if (i.src1.is_constant) { e.mov(e.x4, i.src1.constant()); } else { e.mov(e.x4, i.src1); } if (i.src2.is_constant) { e.mov(e.w0, static_cast(static_cast(i.src2.constant()))); } else { e.mov(e.w0, i.src2); } if (e.IsFeatureEnabled(kA64EmitLSE)) { e.swpal(e.w0, i.dest, ptr(e.x4)); return; } auto& retry = e.NewCachedLabel(); e.L(retry); e.ldaxr(e.w1, ptr(e.x4)); e.stlxr(e.w2, e.w0, ptr(e.x4)); e.cbnz(e.w2, retry); e.mov(i.dest, e.w1); } }; struct ATOMIC_EXCHANGE_I64 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { if (i.src1.is_constant) { e.mov(e.x4, i.src1.constant()); } else { e.mov(e.x4, i.src1); } if (i.src2.is_constant) { e.mov(e.x0, static_cast(i.src2.constant())); } else { e.mov(e.x0, i.src2); } if (e.IsFeatureEnabled(kA64EmitLSE)) { e.swpal(i.dest, e.x0, ptr(e.x4)); return; } auto& retry = e.NewCachedLabel(); e.L(retry); e.ldaxr(e.x1, ptr(e.x4)); e.stlxr(e.w2, e.x0, ptr(e.x4)); e.cbnz(e.w2, retry); e.mov(i.dest, e.x1); } }; EMITTER_OPCODE_TABLE(OPCODE_ATOMIC_EXCHANGE, ATOMIC_EXCHANGE_I8, ATOMIC_EXCHANGE_I16, ATOMIC_EXCHANGE_I32, ATOMIC_EXCHANGE_I64); // ============================================================================ // OPCODE_ATOMIC_COMPARE_EXCHANGE // ============================================================================ struct ATOMIC_COMPARE_EXCHANGE_I32 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { // Compute full host address (ldxr/stxr need base-only [Xn] addressing). auto addr = ComputeMemoryAddress(e, i.src1); e.add(e.x4, e.GetMembaseReg(), addr); // src2 = expected (use w5), src3 = desired (use w6). if (i.src2.is_constant) { e.mov(e.w5, static_cast(static_cast(i.src2.constant()))); } else { e.mov(e.w5, i.src2); } if (i.src3.is_constant) { e.mov(e.w6, static_cast(static_cast(i.src3.constant()))); } else { e.mov(e.w6, i.src3); } if (e.IsFeatureEnabled(kA64EmitLSE)) { e.mov(e.w0, e.w5); e.casal(e.w5, e.w6, ptr(e.x4)); e.cmp(e.w5, e.w0); e.cset(i.dest, Xbyak_aarch64::EQ); return; } auto& retry = e.NewCachedLabel(); auto& fail = e.NewCachedLabel(); auto& done = e.NewCachedLabel(); e.L(retry); e.ldaxr(e.w2, ptr(e.x4)); e.cmp(e.w2, e.w5); e.b(Xbyak_aarch64::NE, fail); e.stlxr(e.w3, e.w6, ptr(e.x4)); e.cbnz(e.w3, retry); e.mov(i.dest, 1); e.b(done); e.L(fail); e.clrex(15); e.mov(i.dest, 0); e.L(done); } }; struct ATOMIC_COMPARE_EXCHANGE_I64 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); e.add(e.x4, e.GetMembaseReg(), addr); if (i.src2.is_constant) { e.mov(e.x5, static_cast(i.src2.constant())); } else { e.mov(e.x5, i.src2); } if (i.src3.is_constant) { e.mov(e.x6, static_cast(i.src3.constant())); } else { e.mov(e.x6, i.src3); } if (e.IsFeatureEnabled(kA64EmitLSE)) { e.mov(e.x0, e.x5); e.casal(e.x5, e.x6, ptr(e.x4)); e.cmp(e.x5, e.x0); e.cset(i.dest, Xbyak_aarch64::EQ); return; } auto& retry = e.NewCachedLabel(); auto& fail = e.NewCachedLabel(); auto& done = e.NewCachedLabel(); e.L(retry); e.ldaxr(e.x2, ptr(e.x4)); e.cmp(e.x2, e.x5); e.b(Xbyak_aarch64::NE, fail); e.stlxr(e.w3, e.x6, ptr(e.x4)); e.cbnz(e.w3, retry); e.mov(i.dest, 1); e.b(done); e.L(fail); e.clrex(15); e.mov(i.dest, 0); e.L(done); } }; EMITTER_OPCODE_TABLE(OPCODE_ATOMIC_COMPARE_EXCHANGE, ATOMIC_COMPARE_EXCHANGE_I32, ATOMIC_COMPARE_EXCHANGE_I64); // ============================================================================ // OPCODE_LOAD_MMIO / OPCODE_STORE_MMIO // ============================================================================ struct LOAD_MMIO_I32 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto mmio_range = reinterpret_cast(i.src1.value); auto read_address = uint32_t(i.src2.value); // CallNativeSafe: thunk sets x0=PPCContext*, x1/x2/x3 pass through. // MMIOReadCallback(void* ppc_ctx, void* callback_ctx, uint32_t addr). e.mov(e.x1, uint64_t(mmio_range->callback_context)); e.mov(e.w2, static_cast(read_address)); e.CallNativeSafe(reinterpret_cast(mmio_range->read)); e.rev(e.w0, e.w0); e.mov(i.dest, e.w0); } }; EMITTER_OPCODE_TABLE(OPCODE_LOAD_MMIO, LOAD_MMIO_I32); struct STORE_MMIO_I32 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto mmio_range = reinterpret_cast(i.src1.value); auto write_address = uint32_t(i.src2.value); // CallNativeSafe: thunk sets x0=PPCContext*, x1/x2/x3 pass through. // MMIOWriteCallback(void* ppc_ctx, void* callback_ctx, uint32_t addr, // uint32_t value). e.mov(e.x1, uint64_t(mmio_range->callback_context)); e.mov(e.w2, static_cast(write_address)); if (i.src3.is_constant) { e.mov(e.w3, static_cast( xe::byte_swap(static_cast(i.src3.constant())))); } else { e.mov(e.w3, i.src3); e.rev(e.w3, e.w3); } e.CallNativeSafe(reinterpret_cast(mmio_range->write)); } }; EMITTER_OPCODE_TABLE(OPCODE_STORE_MMIO, STORE_MMIO_I32); // ============================================================================ // OPCODE_RESERVED_LOAD / OPCODE_RESERVED_STORE // ============================================================================ // Helper: get pointer to A64BackendContext. // x19 is the dedicated backend context register, so this is a no-op // accessor for readability. The returned register is x19. static const Xbyak_aarch64::XReg& LoadBackendCtxPtr(A64Emitter& e) { return e.GetBackendCtxReg(); } struct RESERVED_LOAD_I32 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); // Save guest address before load — dest may alias addr register. e.mov(e.w0, WReg(addr.getIdx())); // Load the value (may clobber addr if dest == addr). e.ldr(i.dest, ptr(e.GetMembaseReg(), addr)); // Save reservation: address and value in backend context. auto bctx = LoadBackendCtxPtr(e); // Store the guest address (already saved in x0). e.str(e.x0, ptr(bctx, static_cast(offsetof( A64BackendContext, cached_reserve_offset)))); // Store the loaded value (zero-extended to 64-bit). e.mov(e.w1, i.dest); e.str(e.x1, ptr(bctx, static_cast(offsetof( A64BackendContext, cached_reserve_value_)))); // Set the "has reserve" flag (bit 1). e.ldr(e.w1, ptr(bctx, static_cast(offsetof(A64BackendContext, flags)))); e.orr(e.w1, e.w1, static_cast(1u << kA64BackendHasReserveBit)); e.str(e.w1, ptr(bctx, static_cast(offsetof(A64BackendContext, flags)))); } }; struct RESERVED_LOAD_I64 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); // Save guest address before load — dest may alias addr register. e.mov(e.w0, WReg(addr.getIdx())); // Load the value (may clobber addr if dest == addr). e.ldr(i.dest, ptr(e.GetMembaseReg(), addr)); // Save reservation in backend context. auto bctx = LoadBackendCtxPtr(e); e.str(e.x0, ptr(bctx, static_cast(offsetof( A64BackendContext, cached_reserve_offset)))); e.str(i.dest, ptr(bctx, static_cast(offsetof( A64BackendContext, cached_reserve_value_)))); e.ldr(e.w1, ptr(bctx, static_cast(offsetof(A64BackendContext, flags)))); e.orr(e.w1, e.w1, static_cast(1u << kA64BackendHasReserveBit)); e.str(e.w1, ptr(bctx, static_cast(offsetof(A64BackendContext, flags)))); } }; EMITTER_OPCODE_TABLE(OPCODE_RESERVED_LOAD, RESERVED_LOAD_I32, RESERVED_LOAD_I64); struct RESERVED_STORE_I32 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); auto& no_reserve = e.NewCachedLabel(); auto& done = e.NewCachedLabel(); // Check if we have a reservation. auto bctx = LoadBackendCtxPtr(e); e.ldr(e.w4, ptr(bctx, static_cast(offsetof(A64BackendContext, flags)))); e.tbz(e.w4, kA64BackendHasReserveBit, no_reserve); // Clear the reserve flag. e.and_(e.w4, e.w4, static_cast(~(1u << kA64BackendHasReserveBit))); e.str(e.w4, ptr(bctx, static_cast(offsetof(A64BackendContext, flags)))); // Check if address matches. e.ldr(e.x4, ptr(bctx, static_cast(offsetof( A64BackendContext, cached_reserve_offset)))); e.mov(e.w5, WReg(addr.getIdx())); e.cmp(e.x4, e.x5); e.b(Xbyak_aarch64::NE, no_reserve); // Address matches. Do atomic compare-exchange. // Expected value from cached_reserve_value_. e.ldr(e.w5, ptr(bctx, static_cast(offsetof( A64BackendContext, cached_reserve_value_)))); // Desired value. if (i.src2.is_constant) { e.mov(e.w6, static_cast(static_cast(i.src2.constant()))); } else { e.mov(e.w6, WReg(i.src2.reg().getIdx())); } // Compute host address. e.add(e.x4, e.GetMembaseReg(), addr); if (e.IsFeatureEnabled(kA64EmitLSE)) { e.mov(e.w0, e.w5); e.casal(e.w5, e.w6, ptr(e.x4)); e.cmp(e.w5, e.w0); e.cset(i.dest, Xbyak_aarch64::EQ); return; } // LDXR/STXR loop. auto& cas_loop = e.NewCachedLabel(); auto& cas_fail = e.NewCachedLabel(); e.L(cas_loop); e.ldaxr(e.w7, ptr(e.x4)); e.cmp(e.w7, e.w5); e.b(Xbyak_aarch64::NE, cas_fail); e.stlxr(e.w7, e.w6, ptr(e.x4)); e.cbnz(e.w7, cas_loop); // Success. e.mov(i.dest, 1); e.b(done); e.L(cas_fail); e.clrex(15); e.L(no_reserve); e.mov(i.dest, 0); e.L(done); } }; struct RESERVED_STORE_I64 : Sequence> { static void Emit(A64Emitter& e, const EmitArgType& i) { auto addr = ComputeMemoryAddress(e, i.src1); auto& no_reserve = e.NewCachedLabel(); auto& done = e.NewCachedLabel(); auto bctx = LoadBackendCtxPtr(e); e.ldr(e.w4, ptr(bctx, static_cast(offsetof(A64BackendContext, flags)))); e.tbz(e.w4, kA64BackendHasReserveBit, no_reserve); e.and_(e.w4, e.w4, static_cast(~(1u << kA64BackendHasReserveBit))); e.str(e.w4, ptr(bctx, static_cast(offsetof(A64BackendContext, flags)))); e.ldr(e.x4, ptr(bctx, static_cast(offsetof( A64BackendContext, cached_reserve_offset)))); e.mov(e.w5, WReg(addr.getIdx())); e.cmp(e.x4, e.x5); e.b(Xbyak_aarch64::NE, no_reserve); // 64-bit compare-exchange. e.ldr(e.x5, ptr(bctx, static_cast(offsetof( A64BackendContext, cached_reserve_value_)))); if (i.src2.is_constant) { e.mov(e.x6, static_cast(i.src2.constant())); } else { e.mov(e.x6, XReg(i.src2.reg().getIdx())); } e.add(e.x4, e.GetMembaseReg(), addr); if (e.IsFeatureEnabled(kA64EmitLSE)) { e.mov(e.x0, e.x5); e.casal(e.x5, e.x6, ptr(e.x4)); e.cmp(e.x5, e.x0); e.cset(i.dest, Xbyak_aarch64::EQ); return; } auto& cas_loop = e.NewCachedLabel(); auto& cas_fail = e.NewCachedLabel(); e.L(cas_loop); e.ldaxr(e.x7, ptr(e.x4)); e.cmp(e.x7, e.x5); e.b(Xbyak_aarch64::NE, cas_fail); e.stlxr(e.w7, e.x6, ptr(e.x4)); e.cbnz(e.w7, cas_loop); e.mov(i.dest, 1); e.b(done); e.L(cas_fail); e.clrex(15); e.L(no_reserve); e.mov(i.dest, 0); e.L(done); } }; EMITTER_OPCODE_TABLE(OPCODE_RESERVED_STORE, RESERVED_STORE_I32, RESERVED_STORE_I64); } // namespace a64 } // namespace backend } // namespace cpu } // namespace xe