[CPU/X64] Optimize VMX dot products, vrsqrtefp, partial stores, and permutes

This commit is contained in:
WawWeFix
2026-08-12 11:05:39 +03:30
committed by Radosław Gliński
parent e31142bd79
commit ec5c875122
5 changed files with 216 additions and 183 deletions

View File

@@ -9,7 +9,9 @@
#include "xenia/cpu/backend/x64/x64_backend.h"
#include <array>
#include <cstddef>
#include "third_party/capstone/include/capstone/capstone.h"
#include "third_party/capstone/include/capstone/x86.h"
@@ -45,6 +47,55 @@ namespace cpu {
namespace backend {
namespace x64 {
// For positive normal inputs, the VMX reciprocal-square-root estimate ignores
// the low 9 mantissa bits. The remaining 14 mantissa bits and the low exponent
// bit fully determine the estimated mantissa and its exponent adjustment.
// Precompute the exact result of the existing coefficient interpolation for
// those 32768 cases; special values and denormals still use the original path.
static uint32_t ComputeNormalVRsqrteTableValue(
uint32_t input, const uint32_t* coefficient_table) {
const uint32_t mantissa = input & 0x7FFFFF;
const uint32_t coefficient_index =
(((input >> 23) & 1) << 4) | (mantissa >> 19);
const uint32_t coefficient = coefficient_table[coefficient_index];
uint32_t estimate = ((coefficient << 10) & 0x3FFFC00) -
(((mantissa >> 9) & 1023) * (coefficient >> 16));
int32_t output_exponent_adjustment = 0;
if (!(estimate & 0x02000000)) {
const uint32_t normalized_estimate = estimate & 0x1FFFFFF;
uint32_t leading_zero_count = 0;
for (uint32_t bit = 0x80000000; !(normalized_estimate & bit); bit >>= 1) {
++leading_zero_count;
}
output_exponent_adjustment += 6 - int32_t(leading_zero_count);
estimate <<= leading_zero_count - 6;
}
if ((estimate & 5) && (estimate & 2)) {
estimate += 4;
}
return (0x3F800000 + uint32_t(output_exponent_adjustment) * 0x00800000) |
((estimate >> 2) & 0x7FFFFF);
}
static const uint32_t* GetNormalVRsqrteTable(
const uint32_t* coefficient_table) {
alignas(64) static const std::array<uint32_t, 1 << 15> table =
[coefficient_table]() {
std::array<uint32_t, 1 << 15> table;
for (uint32_t index = 0; index < table.size(); ++index) {
const uint32_t exponent_parity = index >> 14;
const uint32_t mantissa = (index & 0x3FFF) << 9;
const uint32_t canonical_exponent = 126 + exponent_parity;
table[index] = ComputeNormalVRsqrteTableValue(
(canonical_exponent << 23) | mantissa, coefficient_table);
}
return table;
}();
return table.data();
}
class X64HelperEmitter : public X64Emitter {
public:
struct _code_offsets {
@@ -1060,9 +1111,14 @@ void* X64HelperEmitter::EmitScalarVRsqrteHelper() {
Xbyak::Label L18, L2, L35, L4, L9, L8, L10, L11, L12, L13, L1;
Xbyak::Label LC1, _LCPI3_1;
Xbyak::Label handle_denormal_input;
Xbyak::Label handle_non_positive_normal;
Xbyak::Label specialcheck_1, convert_to_signed_inf_and_ret,
handle_oddball_denormal;
const uint32_t* normal_table =
GetNormalVRsqrteTable(reinterpret_cast<const uint32_t*>(
backend()->LookupXMMConstantAddress(XMMVRsqrteTableStart)));
auto emulate_lzcnt_helper_unary_reg = [this](auto& reg, auto& scratch_reg) {
inLocalLabel();
Xbyak::Label end_lzcnt;
@@ -1076,6 +1132,24 @@ void* X64HelperEmitter::EmitScalarVRsqrteHelper() {
};
vmovd(r8d, xmm0);
lea(eax, ptr[r8 - 0x00800000]);
cmp(eax, 0x7EFFFFFF);
ja(handle_non_positive_normal, CodeGenerator::T_NEAR);
mov(edx, r8d);
shr(edx, 9);
and_(edx, 0x7FFF);
mov(r9, reinterpret_cast<uintptr_t>(normal_table));
mov(ecx, ptr[r9 + rdx * 4]);
shr(r8d, 24);
sub(r8d, 63);
shl(r8d, 23);
sub(ecx, r8d);
vmovd(xmm0, ecx);
ret();
L(handle_non_positive_normal);
vmovaps(xmm1, xmm0);
mov(ecx, r8d);
// extract mantissa

View File

@@ -354,33 +354,57 @@ struct LVR_V128 : Sequence<LVR_V128, I<OPCODE_LVR, V128Op, I64Op>> {
};
EMITTER_OPCODE_TABLE(OPCODE_LVR, LVR_V128);
// Stores ecx bytes from xmm0 to [rax] without touching adjacent bytes.
static void EmitPartialVectorStore(X64Emitter& e) {
Xbyak::Label skip_8, skip_4, skip_2, done;
e.vmovq(e.r8, e.xmm0);
e.test(e.ecx, 8);
e.jz(skip_8);
e.mov(e.qword[e.rax], e.r8);
e.add(e.rax, 8);
e.vpextrq(e.r8, e.xmm0, 1);
e.L(skip_8);
e.test(e.ecx, 4);
e.jz(skip_4);
e.mov(e.dword[e.rax], e.r8d);
e.add(e.rax, 4);
e.shr(e.r8, 32);
e.L(skip_4);
e.test(e.ecx, 2);
e.jz(skip_2);
e.mov(e.word[e.rax], e.r8w);
e.add(e.rax, 2);
e.shr(e.r8, 16);
e.L(skip_2);
e.test(e.ecx, 1);
e.jz(done);
e.mov(e.byte[e.rax], e.r8b);
e.L(done);
}
struct STVL_V128 : Sequence<STVL_V128, I<OPCODE_STVL, VoidOp, I64Op, V128Op>> {
static void Emit(X64Emitter& e, const EmitArgType& i) {
Xmm src2 = GetInputRegOrConstant(e, i.src2, e.xmm0);
e.StashXmm(0, src2);
// Store bytes offset..15 from the source vector. Xenia's host vector byte
// layout is word-swapped from guest byte order, so convert source byte
// indexes with ^ 3 before reading the stashed XMM value.
e.lea(e.rax, e.ptr[ComputeMemoryAddress(e, i.src1)]);
e.mov(e.ecx, 15);
e.and_(e.ecx, e.eax);
e.mov(e.edx, 15);
e.not_(e.rdx);
e.and_(e.rax, e.rdx);
Xbyak::Label loop, done;
e.mov(e.edx, e.ecx);
e.L(loop);
e.cmp(e.edx, 16);
e.jge(done);
e.mov(e.r8d, e.edx);
e.sub(e.r8d, e.ecx);
e.xor_(e.r8d, 3);
e.movzx(e.r9d, e.byte[e.rsp + X64Emitter::kStashOffset + e.r8]);
e.mov(e.byte[e.rax + e.rdx], e.r9b);
e.inc(e.edx);
e.jmp(loop);
Xmm src2 = GetInputRegOrConstant(e, i.src2, e.xmm0);
e.vpshufb(e.xmm0, src2, e.GetXmmConstPtr(XMMLVLShuffle));
Xbyak::Label partial, done;
e.test(e.ecx, e.ecx);
e.jnz(partial);
e.vmovdqa(e.ptr[e.rax], e.xmm0);
e.jmp(done);
e.L(partial);
e.neg(e.ecx);
e.add(e.ecx, 16);
EmitPartialVectorStore(e);
e.L(done);
}
};
@@ -390,30 +414,25 @@ struct STVR_V128 : Sequence<STVR_V128, I<OPCODE_STVR, VoidOp, I64Op, V128Op>> {
static void Emit(X64Emitter& e, const EmitArgType& i) {
Xbyak::Label skipper{};
e.mov(e.ecx, 15);
e.mov(e.edx, e.ecx);
e.lea(e.rax, e.ptr[ComputeMemoryAddress(e, i.src1)]);
e.and_(e.ecx, e.eax);
e.jz(skipper);
e.not_(e.rdx);
e.and_(e.rax, e.rdx);
e.jz(skipper, X64Emitter::T_NEAR);
e.and_(e.rax, -16);
e.mov(e.edx, 16);
e.sub(e.edx, e.ecx);
e.vmovd(e.xmm1, e.edx);
if (e.IsFeatureEnabled(kX64EmitAVX2)) {
e.vpbroadcastb(e.xmm1, e.xmm1);
} else {
e.vpshufb(e.xmm1, e.xmm1, e.GetXmmConstPtr(XMMZero));
}
e.vpaddb(e.xmm1, e.xmm1, e.GetXmmConstPtr(XMMSTVLShuffle));
e.vpxor(e.xmm1, e.xmm1, e.GetXmmConstPtr(XMMSwapWordMask));
Xmm src2 = GetInputRegOrConstant(e, i.src2, e.xmm0);
e.StashXmm(0, src2);
// Store bytes 0..offset-1 from the tail of the source vector.
Xbyak::Label loop;
e.xor_(e.edx, e.edx);
e.L(loop);
e.cmp(e.edx, e.ecx);
e.jge(skipper);
e.mov(e.r8d, 16);
e.sub(e.r8d, e.ecx);
e.add(e.r8d, e.edx);
e.xor_(e.r8d, 3);
e.movzx(e.r9d, e.byte[e.rsp + X64Emitter::kStashOffset + e.r8]);
e.mov(e.byte[e.rax + e.rdx], e.r9b);
e.inc(e.edx);
e.jmp(loop);
e.vpshufb(e.xmm0, src2, e.xmm1);
EmitPartialVectorStore(e);
e.L(skipper);
}
};

View File

@@ -2425,6 +2425,10 @@ struct PERMUTE_I32
} else {
src3 = i.src3;
}
if (control == MakePermuteMask(0, 2, 0, 3, 1, 0, 1, 1)) {
e.vshufps(i.dest, src2, src3, MakeSwizzleMask(2, 3, 0, 1));
return;
}
if (i.dest != src3) {
e.vpshufd(i.dest, src2, src_control);
e.vpshufd(e.xmm0, src3, src_control);

View File

@@ -33,16 +33,15 @@
#include "xenia/cpu/backend/x64/x64_emitter.h"
#include "xenia/cpu/backend/x64/x64_op.h"
#include "xenia/cpu/backend/x64/x64_tracers.h"
// needed for stmxcsr
// Needed for MXCSR scratch storage.
#include "xenia/cpu/backend/x64/x64_stack_layout.h"
#include "xenia/cpu/backend/x64/x64_util.h"
#include "xenia/cpu/hir/hir_builder.h"
#include "xenia/cpu/processor.h"
DEFINE_bool(use_fast_dot_product, false,
"Experimental optimization, much shorter sequence on dot products, "
"treating inf as overflow instead of using mcxsr"
"four insn dotprod",
"Use less accurate dot-product exception handling that converts "
"all infinite results to QNaN.",
"CPU");
DEFINE_bool(no_round_to_single, false,
@@ -2340,30 +2339,61 @@ EMITTER_OPCODE_TABLE(OPCODE_LOG2, LOG2_F32, LOG2_F64, LOG2_V128);
// ============================================================================
// OPCODE_DOT_PRODUCT_3
// ============================================================================
// Keep the float64 accumulation below: it closely matches Xbox 360 vmsum
// results, which may differ from a host float32 dot product by one bit.
template <typename EmitArgType>
static void EmitDotProductResult(X64Emitter& e, const EmitArgType& i) {
Xbyak::Label& done = e.NewCachedLabel();
Xbyak::Label& exceptional_result =
e.AddToTail([i, &done](X64Emitter& e, Xbyak::Label& exceptional_result) {
e.L(exceptional_result);
if (!cvars::use_fast_dot_product) {
// A dot product of four float32 values can't overflow float64, so
// float32 overflow happened exactly when the float64 sum is finite
// but its float32 conversion has an all-ones exponent. Preserve
// infinities and NaNs originating in the inputs, as the previous
// MXCSR overflow-flag check did.
Xbyak::Label double_result_was_non_finite;
e.vmovq(e.rax, e.xmm2);
e.shr(e.rax, 52);
e.and_(e.eax, 0x7FF);
e.cmp(e.eax, 0x7FF);
e.je(double_result_was_non_finite);
e.vmovaps(i.dest, e.GetXmmConstPtr(XMMQNaN));
e.jmp(done, X64Emitter::T_NEAR);
e.L(double_result_was_non_finite);
} else {
// Preserve the existing opt-in behavior, which maps infinity to the
// canonical quiet NaN but leaves an existing NaN unchanged.
Xbyak::Label input_was_nan;
e.vmovd(e.eax, e.xmm1);
e.test(e.eax, 0x007FFFFF);
e.jnz(input_was_nan);
e.vmovaps(i.dest, e.GetXmmConstPtr(XMMQNaN));
e.jmp(done, X64Emitter::T_NEAR);
e.L(input_was_nan);
}
e.vshufps(i.dest, e.xmm1, e.xmm1, 0);
e.jmp(done, X64Emitter::T_NEAR);
});
// The common finite result needs no MXCSR status round trip. Check only the
// float32 exponent and leave all exceptional handling in cold tail code.
e.vmovd(e.eax, e.xmm1);
e.add(e.eax, e.eax); // Discard the sign bit.
e.cmp(e.eax, 0xFF000000);
e.jae(exceptional_result, X64Emitter::T_NEAR);
e.vshufps(i.dest, e.xmm1, e.xmm1, 0);
e.L(done);
}
struct DOT_PRODUCT_3_V128
: Sequence<DOT_PRODUCT_3_V128,
I<OPCODE_DOT_PRODUCT_3, V128Op, V128Op, V128Op>> {
static void Emit(X64Emitter& e, const EmitArgType& i) {
e.ChangeMxcsrMode(MXCSRMode::Vmx);
// todo: add fast_dot_product path that just checks for infinity instead of
// using mxcsr
auto mxcsr_storage = e.dword[e.rsp + StackLayout::GUEST_SCRATCH];
// this is going to hurt a bit...
/*
this implementation is accurate, it matches the results of xb360 vmsum3
except that vmsum3 is often off by 1 bit, but its extremely slow. it is a
long, unbroken chain of dependencies, and the three uses of mxcsr all cost
about 15-20 cycles at the very least on amd zen processors. on older amd the
figures agner has are pretty horrible. it looks like its just as bad on
modern intel cpus also up until just recently. perhaps a better way of
detecting overflow would be to just compare with inf. todo: test whether cmp
with inf can replace
*/
if (!cvars::use_fast_dot_product) {
e.vstmxcsr(mxcsr_storage);
e.mov(e.eax, 8);
}
e.vmovaps(e.xmm2, e.GetXmmConstPtr(XMMThreeFloatMask));
bool is_lensqr = i.instr->src1.value == i.instr->src2.value;
@@ -2381,9 +2411,6 @@ struct DOT_PRODUCT_3_V128
} else {
src2v = i.src2.reg();
}
if (!cvars::use_fast_dot_product) {
e.not_(e.eax);
}
// todo: maybe the top element should be cleared by the InstrEmit_ function
// so that in the future this could be optimized away if the top is known to
// be zero. Right now im not sure that happens often though and its
@@ -2393,11 +2420,6 @@ struct DOT_PRODUCT_3_V128
e.vandps(e.xmm2, src2v, e.xmm2);
if (!cvars::use_fast_dot_product) {
e.and_(mxcsr_storage, e.eax);
e.vldmxcsr(mxcsr_storage); // overflow flag is cleared, now we're good
// to go
}
e.vcvtps2pd(e.ymm0, e.xmm3);
e.vcvtps2pd(e.ymm1, e.xmm2);
@@ -2408,47 +2430,15 @@ struct DOT_PRODUCT_3_V128
e.vmulpd(e.ymm3, e.ymm0, e.ymm1);
} else {
e.vandps(e.xmm3, src1v, e.xmm2);
if (!cvars::use_fast_dot_product) {
e.and_(mxcsr_storage, e.eax);
e.vldmxcsr(mxcsr_storage); // overflow flag is cleared, now we're good
// to go
}
e.vcvtps2pd(e.ymm0, e.xmm3);
e.vmulpd(e.ymm3, e.ymm0, e.ymm0);
}
e.vextractf128(e.xmm2, e.ymm3, 1);
e.vunpckhpd(e.xmm0, e.xmm3, e.xmm3); // get element [1] in xmm3
e.vaddsd(e.xmm3, e.xmm3, e.xmm2);
if (!cvars::use_fast_dot_product) {
e.not_(e.eax);
}
e.vaddsd(e.xmm2, e.xmm3, e.xmm0);
e.vcvtsd2ss(e.xmm1, e.xmm2);
if (!cvars::use_fast_dot_product) {
e.vstmxcsr(mxcsr_storage);
e.test(mxcsr_storage, e.eax);
Xbyak::Label& done = e.NewCachedLabel();
Xbyak::Label& ret_qnan =
e.AddToTail([i, &done](X64Emitter& e, Xbyak::Label& me) {
e.L(me);
e.vmovaps(i.dest, e.GetXmmConstPtr(XMMQNaN));
e.jmp(done, X64Emitter::T_NEAR);
});
e.jnz(ret_qnan, X64Emitter::T_NEAR); // reorder these jmps later, just
// want to get this fix in
e.vshufps(i.dest, e.xmm1, e.xmm1, 0);
e.L(done);
} else {
e.vandps(e.xmm0, e.xmm1, e.GetXmmConstPtr(XMMAbsMaskPS));
e.vcmpgeps(e.xmm2, e.xmm0, e.GetXmmConstPtr(XMMFloatInf));
e.vblendvps(e.xmm1, e.xmm1, e.GetXmmConstPtr(XMMQNaN), e.xmm2);
e.vshufps(i.dest, e.xmm1, e.xmm1, 0);
}
EmitDotProductResult(e, i);
}
};
EMITTER_OPCODE_TABLE(OPCODE_DOT_PRODUCT_3, DOT_PRODUCT_3_V128);
@@ -2461,10 +2451,6 @@ struct DOT_PRODUCT_4_V128
I<OPCODE_DOT_PRODUCT_4, V128Op, V128Op, V128Op>> {
static void Emit(X64Emitter& e, const EmitArgType& i) {
e.ChangeMxcsrMode(MXCSRMode::Vmx);
// todo: add fast_dot_product path that just checks for infinity instead of
// using mxcsr
auto mxcsr_storage = e.dword[e.rsp + StackLayout::GUEST_SCRATCH];
bool is_lensqr = i.instr->src1.value == i.instr->src2.value;
auto src1v = e.xmm3;
@@ -2481,15 +2467,6 @@ struct DOT_PRODUCT_4_V128
} else {
src2v = i.src2.reg();
}
if (!cvars::use_fast_dot_product) {
e.vstmxcsr(mxcsr_storage);
e.mov(e.eax, 8);
e.not_(e.eax);
e.and_(mxcsr_storage, e.eax);
e.vldmxcsr(mxcsr_storage);
}
if (is_lensqr) {
e.vcvtps2pd(e.ymm0, src1v);
@@ -2504,36 +2481,9 @@ struct DOT_PRODUCT_4_V128
e.vaddpd(e.xmm3, e.xmm3, e.xmm2);
e.vunpckhpd(e.xmm0, e.xmm3, e.xmm3);
if (!cvars::use_fast_dot_product) {
e.not_(e.eax);
}
e.vaddsd(e.xmm2, e.xmm3, e.xmm0);
e.vcvtsd2ss(e.xmm1, e.xmm2);
if (!cvars::use_fast_dot_product) {
e.vstmxcsr(mxcsr_storage);
e.test(mxcsr_storage, e.eax);
Xbyak::Label& done = e.NewCachedLabel();
Xbyak::Label& ret_qnan =
e.AddToTail([i, &done](X64Emitter& e, Xbyak::Label& me) {
e.L(me);
e.vmovaps(i.dest, e.GetXmmConstPtr(XMMQNaN));
e.jmp(done, X64Emitter::T_NEAR);
});
e.jnz(ret_qnan, X64Emitter::T_NEAR); // reorder these jmps later, just
// want to get this fix in
e.vshufps(i.dest, e.xmm1, e.xmm1, 0);
e.L(done);
} else {
e.vandps(e.xmm0, e.xmm1, e.GetXmmConstPtr(XMMAbsMaskPS));
e.vcmpgeps(e.xmm2, e.xmm0, e.GetXmmConstPtr(XMMFloatInf));
e.vblendvps(e.xmm1, e.xmm1, e.GetXmmConstPtr(XMMQNaN), e.xmm2);
e.vshufps(i.dest, e.xmm1, e.xmm1, 0);
}
EmitDotProductResult(e, i);
}
};
EMITTER_OPCODE_TABLE(OPCODE_DOT_PRODUCT_4, DOT_PRODUCT_4_V128);

View File

@@ -785,16 +785,10 @@ int InstrEmit_vlogefp128(PPCHIRBuilder& f, const InstrData& i) {
int InstrEmit_vmaddfp_(PPCHIRBuilder& f, uint32_t vd, uint32_t va, uint32_t vb,
uint32_t vc) {
/*
chrispy: testing on POWER8 revealed that altivec vmaddfp unconditionally
flushes denormal inputs to 0, regardless of NJM setting
*/
Value* a = f.VectorDenormFlush(f.LoadVR(va));
Value* b = f.VectorDenormFlush(f.LoadVR(vb));
Value* c = f.VectorDenormFlush(f.LoadVR(vc));
// POWER8 testing showed that vmaddfp flushes denormal inputs to zero
// regardless of NJM.
// (VD) <- ((VA) * (VC)) + (VB)
Value* v = f.MulAdd(a, c, b);
// todo: do denormal results also unconditionally become 0?
Value* v = f.MulAdd(f.LoadVR(va), f.LoadVR(vc), f.LoadVR(vb));
f.StoreVR(vd, v);
return 0;
}
@@ -810,16 +804,9 @@ int InstrEmit_vmaddfp128(PPCHIRBuilder& f, const InstrData& i) {
}
int InstrEmit_vmaddcfp128(PPCHIRBuilder& f, const InstrData& i) {
/*
see vmaddfp about these denormflushes
*/
Value* a = f.VectorDenormFlush(f.LoadVR(VX128_VA128));
Value* b = f.VectorDenormFlush(f.LoadVR(VX128_VB128));
Value* d = f.VectorDenormFlush(f.LoadVR(VX128_VD128));
// (VD) <- ((VA) * (VD)) + (VB)
Value* v = f.MulAdd(a, d, b);
f.StoreVR(VX128_VD128, v);
return 0;
return InstrEmit_vmaddfp_(f, VX128_VD128, VX128_VA128, VX128_VB128,
VX128_VD128);
}
int InstrEmit_vmaxfp_(PPCHIRBuilder& f, uint32_t vd, uint32_t va, uint32_t vb) {
@@ -1067,9 +1054,8 @@ int InstrEmit_vmsumuhs(PPCHIRBuilder& f, const InstrData& i) {
int InstrEmit_vmsum3fp128(PPCHIRBuilder& f, const InstrData& i) {
// Dot product XYZ.
// (VD.xyzw) = (VA.x * VB.x) + (VA.y * VB.y) + (VA.z * VB.z)
Value* v = f.DotProduct3(f.LoadVR(VX128_VA128), f.LoadVR(VX128_VB128));
// chrispy: denormal outputs for Dot product are unconditionally made 0
v = f.VectorDenormFlush(v);
Value* v = f.DotProduct3(f.LoadVR(VX128_VA128), f.LoadVR(VX128_VB128));
f.StoreVR(VX128_VD128, v);
return 0;
}
@@ -1078,7 +1064,6 @@ int InstrEmit_vmsum4fp128(PPCHIRBuilder& f, const InstrData& i) {
// Dot product XYZW.
// (VD.xyzw) = (VA.x * VB.x) + (VA.y * VB.y) + (VA.z * VB.z) + (VA.w * VB.w)
Value* v = f.DotProduct4(f.LoadVR(VX128_VA128), f.LoadVR(VX128_VB128));
v = f.VectorDenormFlush(v);
f.StoreVR(VX128_VD128, v);
return 0;
}
@@ -1138,16 +1123,7 @@ int InstrEmit_vnmsubfp_(PPCHIRBuilder& f, uint32_t vd, uint32_t va, uint32_t vb,
// NOTE2: we could make vnmsub a new opcode, and then do it in double
// precision, rounding after the neg
/*
chrispy: this is untested, but i believe this has the same DAZ behavior for
inputs as vmadd
*/
Value* a = f.VectorDenormFlush(f.LoadVR(va));
Value* b = f.VectorDenormFlush(f.LoadVR(vb));
Value* c = f.VectorDenormFlush(f.LoadVR(vc));
Value* v = f.Neg(f.MulSub(a, c, b));
Value* v = f.Neg(f.MulSub(f.LoadVR(va), f.LoadVR(vc), f.LoadVR(vb)));
f.StoreVR(vd, v);
return 0;
}
@@ -1465,12 +1441,22 @@ int InstrEmit_vsldoi_(PPCHIRBuilder& f, uint32_t vd, uint32_t va, uint32_t vb,
f.StoreVR(vd, f.LoadVR(vb));
return 0;
}
// TODO(benvanik): optimize for the rotation case:
// vsldoi128 vr63,vr63,vr63,4
// (ABCD ABCD) << 4b = (BCDA)
// (VA << SH) OR (VB >> (16 - SH))
Value* control = f.LoadConstantVec128(__vsldoi_table[sh]);
Value* v = f.Permute(control, f.LoadVR(va), f.LoadVR(vb), INT8_TYPE);
Value* v;
if (!(sh & 3)) {
// Word-aligned shifts can use the cheaper 32-bit permute.
uint32_t control = 0;
uint32_t source_word = sh >> 2;
for (uint32_t output_word = 0; output_word < 4;
++output_word, ++source_word) {
control |= ((source_word & 3) | ((source_word & 4) ? 4 : 0))
<< (output_word * 8);
}
v = f.Permute(f.LoadConstantUint32(control), f.LoadVR(va), f.LoadVR(vb),
INT32_TYPE);
} else {
Value* control = f.LoadConstantVec128(__vsldoi_table[sh]);
v = f.Permute(control, f.LoadVR(va), f.LoadVR(vb), INT8_TYPE);
}
f.StoreVR(vd, v);
return 0;
}