Merge pull request #3186 from Sonicadvance1/afp_support

IR: Adds scalar vector insert operations
This commit is contained in:
Ryan Houdek authored and GitHub committed 2023-10-10 11:28:22 -07:00
commit 6253f4f708
37 files changed
+3915 -874

No files matched your search

@@ -581,6 +581,24 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
}
void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
#ifndef VIXL_SIMULATOR
if (EmitterCTX->HostFeatures.SupportsAFP) {
// Disable AFP features when spilling registers.
//
// Disable FPCR.NEP and FPCR.AH
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
//
// Additional interesting AFP bits:
// FIZ(0): Flush Inputs to Zero
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
bic(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
(1U << 2) | // NEP
(1U << 1)); // AH
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
}
#endif
if (!StaticRegisterAllocation()) {
return;
}
@@ -645,6 +663,26 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
}
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
#ifndef VIXL_SIMULATOR
if (EmitterCTX->HostFeatures.SupportsAFP) {
// Enable AFP features when filling JIT state.
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
auto TmpReg = StaticRegisters[FindFirstSetBit(GPRFillMask)];
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
// Enable FPCR.NEP and FPCR.AH
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
//
// Additional interesting AFP bits:
// FIZ(0): Flush Inputs to Zero
orr(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
(1U << 2) | // NEP
(1U << 1)); // AH
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
}
#endif
if (!StaticRegisterAllocation()) {
return;
}
@@ -53,7 +53,7 @@ uint64_t Dispatcher::GetCompileBlockPtr() {
return CompileBlockPtr.Data;
}
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config)
: Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE)
@@ -243,6 +243,9 @@ HostFeatures::HostFeatures() {
SupportsBMI2 = true;
SupportsCLWB = true;
// TODO: AFP is disabled until the scalar usage in the codebase can be audited to be working as expected.
SupportsAFP = false;
if (!SupportsAtomics) {
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
}
@@ -534,6 +534,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
, HostSupportsSVE128{ctx->HostFeatures.SupportsSVE}
, HostSupportsSVE256{ctx->HostFeatures.SupportsAVX}
, HostSupportsRPRES{ctx->HostFeatures.SupportsRPRES}
, HostSupportsAFP{ctx->HostFeatures.SupportsAFP}
, CTX {ctx} {
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
@@ -26,6 +26,7 @@ $end_info$
#include <array>
#include <cstdint>
#include <utility>
#include <variant>
namespace FEXCore::Core {
struct InternalThreadState;
@@ -59,6 +60,7 @@ private:
const bool HostSupportsSVE128{};
const bool HostSupportsSVE256{};
const bool HostSupportsRPRES{};
const bool HostSupportsAFP{};
ARMEmitter::BiDirectionalLabel *PendingTargetLabel;
FEXCore::Context::ContextImpl *CTX;
@@ -209,6 +211,11 @@ private:
uint32_t SpillSlots{};
using OpType = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
using ScalarUnaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar)>;
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
// Runtime selection;
// Load and store register style.
OpType RT_LoadRegister;
@@ -14,6 +14,786 @@ $end_info$
namespace FEXCore::CPU {
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
void Arm64JITCore::VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2) {
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
if (!Is256Bit) {
LOGMAN_THROW_A_FMT(ZeroUpperBits == false, "128-bit operation doesn't support ZeroUpperBits in {}", __func__);
}
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
constexpr auto Predicate = ARMEmitter::PReg::p0;
if (Dst == Vector1) {
if (ZeroUpperBits) {
// When zeroing the upper 128-bits we just use an ASIMD move.
mov(Dst.Q(), Vector1.Q());
}
if (HostSupportsAFP) {
// If the host CPU supports AFP then scalar does an insert without modifying upper bits.
ScalarEmit(Dst, Vector1, Vector2);
}
else {
// If AFP is unsupported then the operation result goes in to a temporary.
// and then it gets inserted.
ScalarEmit(VTMP1, Vector1, Vector2);
if (!ZeroUpperBits && Is256Bit) {
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
}
else {
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
}
}
}
else if (Dst != Vector2) {
if (!ZeroUpperBits && Is256Bit) {
mov(Dst.Z(), Vector1.Z());
}
else {
mov(Dst.Q(), Vector1.Q());
}
if (HostSupportsAFP) {
ScalarEmit(Dst, Vector1, Vector2);
}
else {
ScalarEmit(VTMP1, Vector1, Vector2);
if (!ZeroUpperBits && Is256Bit) {
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
}
else {
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
}
}
}
else {
// Destination intersects Vector2, can't do anything optimal in this case.
// Do the scalar operation first and then move and insert.
ScalarEmit(VTMP1, Vector1, Vector2);
if (!ZeroUpperBits && Is256Bit) {
mov(Dst.Z(), Vector1.Z());
}
else {
mov(Dst.Q(), Vector1.Q());
}
if (!ZeroUpperBits && Is256Bit) {
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
}
else {
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
}
}
}
void Arm64JITCore::VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2) {
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
if (!Is256Bit) {
LOGMAN_THROW_A_FMT(ZeroUpperBits == false, "128-bit operation doesn't support ZeroUpperBits in {}", __func__);
}
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
constexpr auto Predicate = ARMEmitter::PReg::p0;
bool DstOverlapsVector2 = false;
if (const auto* Vector2Reg = std::get_if<ARMEmitter::VRegister>(&Vector2)) {
DstOverlapsVector2 = Dst == *Vector2Reg;
}
if (Dst == Vector1) {
if (ZeroUpperBits) {
// When zeroing the upper 128-bits we just use an ASIMD move.
mov(Dst.Q(), Vector1.Q());
}
if (HostSupportsAFP) {
// If the host CPU supports AFP then scalar does an insert without modifying upper bits.
ScalarEmit(Dst, Vector2);
}
else {
// If AFP is unsupported then the operation result goes in to a temporary.
// and then it gets inserted.
ScalarEmit(VTMP1, Vector2);
if (!ZeroUpperBits && Is256Bit) {
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
}
else {
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
}
}
}
else if (!DstOverlapsVector2) {
if (!ZeroUpperBits && Is256Bit) {
mov(Dst.Z(), Vector1.Z());
}
else {
mov(Dst.Q(), Vector1.Q());
}
if (HostSupportsAFP) {
ScalarEmit(Dst, Vector2);
}
else {
ScalarEmit(VTMP1, Vector2);
if (!ZeroUpperBits && Is256Bit) {
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
}
else {
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
}
}
}
else {
// Destination intersects Vector2, can't do anything optimal in this case.
// Do the scalar operation first and then move and insert.
ScalarEmit(VTMP1, Vector2);
if (!ZeroUpperBits && Is256Bit) {
mov(Dst.Z(), Vector1.Z());
}
else {
mov(Dst.Q(), Vector1.Q());
}
if (!ZeroUpperBits && Is256Bit) {
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
}
else {
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
}
}
}
DEF_OP(VFAddScalarInsert) {
const auto Op = IROp->C<IR::IROp_VFAddScalarInsert>();
const auto ElementSize = Op->Header.ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
fadd(SubRegSize.Scalar, Dst, Src1, Src2);
};
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
}
DEF_OP(VFSubScalarInsert) {
const auto Op = IROp->C<IR::IROp_VFSubScalarInsert>();
const auto ElementSize = Op->Header.ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
fsub(SubRegSize.Scalar, Dst, Src1, Src2);
};
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
}
DEF_OP(VFMulScalarInsert) {
const auto Op = IROp->C<IR::IROp_VFMulScalarInsert>();
const auto ElementSize = Op->Header.ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
fmul(SubRegSize.Scalar, Dst, Src1, Src2);
};
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
}
DEF_OP(VFDivScalarInsert) {
const auto Op = IROp->C<IR::IROp_VFDivScalarInsert>();
const auto ElementSize = Op->Header.ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
fdiv(SubRegSize.Scalar, Dst, Src1, Src2);
};
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
}
DEF_OP(VFMinScalarInsert) {
const auto Op = IROp->C<IR::IROp_VFMinScalarInsert>();
const auto ElementSize = Op->Header.ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
if (HostSupportsAFP) {
// AFP.AH lets fmin behave like x86 min
fmin(SubRegSize.Scalar, Dst, Src1, Src2);
}
else {
fcmp(SubRegSize.Scalar, Src1, Src2);
fcsel(SubRegSize.Scalar, Dst, Src1, Src2, ARMEmitter::Condition::CC_MI);
}
};
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
}
DEF_OP(VFMaxScalarInsert) {
const auto Op = IROp->C<IR::IROp_VFMaxScalarInsert>();
const auto ElementSize = Op->Header.ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
// AFP can make this more optimal.
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
if (HostSupportsAFP) {
// AFP.AH lets fmax behave like x86 max
fmax(SubRegSize.Scalar, Dst, Src1, Src2);
}
else {
fcmp(SubRegSize.Scalar, Src1, Src2);
fcsel(SubRegSize.Scalar, Dst, Src2, Src1, ARMEmitter::Condition::CC_MI);
}
};
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
}
DEF_OP(VFSqrtScalarInsert) {
const auto Op = IROp->C<IR::IROp_VFSqrtScalarInsert>();
const auto ElementSize = Op->Header.ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
fsqrt(SubRegSize.Scalar, Dst, Src);
};
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
}
DEF_OP(VFRSqrtScalarInsert) {
const auto Op = IROp->C<IR::IROp_VFRSqrtScalarInsert>();
const auto ElementSize = Op->Header.ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
fmov(SubRegSize.Scalar, VTMP1.Q(), 1.0f);
fsqrt(SubRegSize.Scalar, VTMP2, Src);
fdiv(SubRegSize.Scalar, Dst, VTMP1, VTMP2);
};
auto ScalarEmitRPRES = [this, SubRegSize](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
frecpe(SubRegSize.Scalar, Dst.S(), Src.S());
};
std::array<ScalarUnaryOpCaller, 2> Handlers = {
ScalarEmit,
ScalarEmitRPRES,
};
const auto HandlerIndex = ElementSize == 4 && HostSupportsRPRES ? 1 : 0;
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, Handlers[HandlerIndex], Dst, Vector1, Vector2);
}
DEF_OP(VFRecpScalarInsert) {
const auto Op = IROp->C<IR::IROp_VFRecpScalarInsert>();
const auto ElementSize = Op->Header.ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
fmov(SubRegSize.Scalar, VTMP1.Q(), 1.0f);
fdiv(SubRegSize.Scalar, Dst, VTMP1, Src);
};
auto ScalarEmitRPRES = [this, SubRegSize](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
frsqrte(SubRegSize.Scalar, Dst, Src);
};
std::array<ScalarUnaryOpCaller, 2> Handlers = {
ScalarEmit,
ScalarEmitRPRES,
};
const auto HandlerIndex = ElementSize == 4 && HostSupportsRPRES ? 1 : 0;
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, Handlers[HandlerIndex], Dst, Vector1, Vector2);
}
DEF_OP(VFToFScalarInsert) {
const auto Op = IROp->C<IR::IROp_VFToFScalarInsert>();
const auto ElementSize = Op->Header.ElementSize;
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
auto ScalarEmit = [this, Conv](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
switch (Conv) {
case 0x0204: { // Half <- Float
fcvt(Dst.H(), Src.S());
break;
}
case 0x0208: { // Half <- Double
fcvt(Dst.H(), Src.D());
break;
}
case 0x0402: { // Float <- Half
fcvt(Dst.S(), Src.H());
break;
}
case 0x0802: { // Double <- Half
fcvt(Dst.D(), Src.H());
break;
}
case 0x0804: { // Double <- Float
fcvt(Dst.D(), Src.S());
break;
}
case 0x0408: { // Float <- Double
fcvt(Dst.S(), Src.D());
break;
}
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
}
};
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
}
DEF_OP(VSToFVectorInsert) {
const auto Op = IROp->C<IR::IROp_VSToFVectorInsert>();
const auto ElementSize = Op->Header.ElementSize;
const auto HasTwoElements = Op->HasTwoElements;
LOGMAN_THROW_AA_FMT(ElementSize == 4 || ElementSize == 8, "Invalid size");
if (HasTwoElements) {
LOGMAN_THROW_AA_FMT(ElementSize == 4, "Can't have two elements for 8-byte size");
}
auto ScalarEmit = [this, ElementSize, HasTwoElements](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
if (ElementSize == 4) {
if (HasTwoElements) {
scvtf(ARMEmitter::SubRegSize::i32Bit, Dst.D(), Src.D());
}
else {
scvtf(ARMEmitter::ScalarRegSize::i32Bit, Dst.S(), Src.S());
}
}
else {
scvtf(ARMEmitter::ScalarRegSize::i64Bit, Dst.D(), Src.D());
}
};
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
// Claim the element size is 8-bytes.
// Might be scalar 8-byte (cvtsi2ss xmm0, rax)
// Might be vector i32v2 (cvtpi2ps xmm0, mm0)
VFScalarUnaryOperation(IROp->Size, ElementSize * (HasTwoElements ? 2 : 1), Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
}
DEF_OP(VSToFGPRInsert) {
const auto Op = IROp->C<IR::IROp_VSToFGPRInsert>();
const uint16_t ElementSize = Op->Header.ElementSize;
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
auto ScalarEmit = [this, Conv](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
auto Src = *std::get_if<ARMEmitter::Register>(&SrcVar);
switch (Conv) {
case 0x0204: { // Half <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.H(), Src);
break;
}
case 0x0208: { // Half <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.H(), Src);
break;
}
case 0x0404: { // Float <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.S(), Src);
break;
}
case 0x0408: { // Float <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.S(), Src);
break;
}
case 0x0804: { // Double <- int32_t
scvtf(ARMEmitter::Size::i32Bit, Dst.D(), Src);
break;
}
case 0x0808: { // Double <- int64_t
scvtf(ARMEmitter::Size::i64Bit, Dst.D(), Src);
break;
}
default:
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}",
Conv);
break;
}
};
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector.ID());
const auto GPR = GetReg(Op->Src.ID());
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector, GPR);
}
DEF_OP(VFToIScalarInsert) {
const auto Op = IROp->C<IR::IROp_VFToIScalarInsert>();
const auto ElementSize = Op->Header.ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
const auto RoundMode = Op->Round;
auto ScalarEmit = [this, SubRegSize, RoundMode](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
switch (RoundMode) {
case IR::Round_Nearest:
frintn(SubRegSize.Scalar, Dst, Src);
break;
case IR::Round_Negative_Infinity:
frintm(SubRegSize.Scalar, Dst, Src);
break;
case IR::Round_Positive_Infinity:
frintp(SubRegSize.Scalar, Dst, Src);
break;
case IR::Round_Towards_Zero:
frintz(SubRegSize.Scalar, Dst, Src);
break;
case IR::Round_Host:
frinti(SubRegSize.Scalar, Dst, Src);
break;
}
};
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
}
DEF_OP(VFCMPScalarInsert) {
const auto Op = IROp->C<IR::IROp_VFCMPScalarInsert>();
const auto ElementSize = Op->Header.ElementSize;
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
ARMEmitter::SubRegSize::i64Bit);
const auto ZeroUpperBits = Op->ZeroUpperBits;
const auto Is256Bit = IROp->Size == Core::CPUState::XMM_AVX_REG_SIZE;
auto ScalarEmitEQ = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
switch (SubRegSize.Scalar) {
case ARMEmitter::ScalarRegSize::i16Bit: {
fcmeq(Dst.H(), Src1.H(), Src2.H());
break;
}
case ARMEmitter::ScalarRegSize::i32Bit:
case ARMEmitter::ScalarRegSize::i64Bit:
fcmeq(SubRegSize.Scalar, Dst, Src1, Src2);
break;
default:
break;
}
};
auto ScalarEmitLT = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
switch (SubRegSize.Scalar) {
case ARMEmitter::ScalarRegSize::i16Bit: {
fcmgt(Dst.H(), Src2.H(), Src1.H());
break;
}
case ARMEmitter::ScalarRegSize::i32Bit:
case ARMEmitter::ScalarRegSize::i64Bit:
fcmgt(SubRegSize.Scalar, Dst, Src2, Src1);
break;
default:
break;
}
};
auto ScalarEmitLE = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
switch (SubRegSize.Scalar) {
case ARMEmitter::ScalarRegSize::i16Bit: {
fcmge(Dst.H(), Src2.H(), Src1.H());
break;
}
case ARMEmitter::ScalarRegSize::i32Bit:
case ARMEmitter::ScalarRegSize::i64Bit:
fcmge(SubRegSize.Scalar, Dst, Src2, Src1);
break;
default:
break;
}
};
auto ScalarEmitUNO = [this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
switch (SubRegSize.Scalar) {
case ARMEmitter::ScalarRegSize::i16Bit: {
fcmge(VTMP1.H(), Src1.H(), Src2.H());
fcmgt(VTMP2.H(), Src2.H(), Src1.H());
break;
}
case ARMEmitter::ScalarRegSize::i32Bit:
case ARMEmitter::ScalarRegSize::i64Bit:
fcmge(SubRegSize.Scalar, VTMP1, Src1, Src2);
fcmgt(SubRegSize.Scalar, VTMP2, Src2, Src1);
break;
default:
break;
}
// If the destination is a temporary then it is going to do an insert after the operation.
// This means this operation can avoid a redundant insert in this case.
const bool DstIsTemp = Dst == VTMP1;
// Combine results and invert directly in VTMP1.
orr(VTMP1.D(), VTMP1.D(), VTMP2.D());
mvn(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
if (!DstIsTemp) {
// If the destination doesn't overlap VTMP1, then we need to insert the final result.
// This only happens in the case that the host supports AFP.
if (!ZeroUpperBits && Is256Bit) {
constexpr auto Predicate = ARMEmitter::PReg::p0;
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
}
else {
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
}
}
};
auto ScalarEmitNEQ = [this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
switch (SubRegSize.Scalar) {
case ARMEmitter::ScalarRegSize::i16Bit: {
fcmeq(VTMP1.H(), Src1.H(), Src2.H());
break;
}
case ARMEmitter::ScalarRegSize::i32Bit:
case ARMEmitter::ScalarRegSize::i64Bit:
fcmeq(SubRegSize.Scalar, VTMP1, Src1, Src2);
break;
default:
break;
}
// If the destination is a temporary then it is going to do an insert after the operation.
// This means this operation can avoid a redundant insert in this case.
const bool DstIsTemp = Dst == VTMP1;
// Invert directly in VTMP1.
mvn(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
if (!DstIsTemp) {
// If the destination doesn't overlap VTMP1, then we need to insert the final result.
// This only happens in the case that the host supports AFP.
if (!ZeroUpperBits && Is256Bit) {
constexpr auto Predicate = ARMEmitter::PReg::p0;
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
}
else {
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
}
}
};
auto ScalarEmitORD = [this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
switch (SubRegSize.Scalar) {
case ARMEmitter::ScalarRegSize::i16Bit: {
fcmge(VTMP1.H(), Src1.H(), Src2.H());
fcmgt(VTMP2.H(), Src2.H(), Src1.H());
break;
}
case ARMEmitter::ScalarRegSize::i32Bit:
case ARMEmitter::ScalarRegSize::i64Bit:
fcmge(SubRegSize.Scalar, VTMP1, Src1, Src2);
fcmgt(SubRegSize.Scalar, VTMP2, Src2, Src1);
break;
default:
break;
}
// If the destination is a temporary then it is going to do an insert after the operation.
// This means this operation can avoid a redundant insert in this case.
const bool DstIsTemp = Dst == VTMP1;
// Combine results directly in VTMP1.
orr(VTMP1.D(), VTMP1.D(), VTMP2.D());
if (!DstIsTemp) {
// If the destination doesn't overlap VTMP1, then we need to insert the final result.
// This only happens in the case that the host supports AFP.
if (!ZeroUpperBits && Is256Bit) {
constexpr auto Predicate = ARMEmitter::PReg::p0;
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
}
else {
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
}
}
};
std::array<ScalarBinaryOpCaller, 6> Funcs = {{
ScalarEmitEQ,
ScalarEmitLT,
ScalarEmitLE,
ScalarEmitUNO,
ScalarEmitNEQ,
ScalarEmitORD,
}};
// Bit of a tricky detail.
// The upper bits of the destination comes from the first source.
const auto Dst = GetVReg(Node);
const auto Vector1 = GetVReg(Op->Vector1.ID());
const auto Vector2 = GetVReg(Op->Vector2.ID());
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, Funcs[FEXCore::ToUnderlying(Op->Op)], Dst, Vector1, Vector2);
}
DEF_OP(VectorZero) {
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
@@ -5030,7 +5030,7 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
// Now extract the subregister if it was a partial load /smaller/ than SSE size
// TODO: Instead of doing the VMov implicitly on load, hunt down all use cases that require partial loads and do it after load.
// We don't have information here to know if the operation needs zero upper bits or can contain data.
if (OpSize < Core::CPUState::XMM_SSE_REG_SIZE) {
if (!AllowUpperGarbage && OpSize < Core::CPUState::XMM_SSE_REG_SIZE) {
Src = _VMov(OpSize, Src);
}
}
@@ -5908,8 +5908,8 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
{OPD(1, 0b00, 0x29), 1, &OpDispatchBuilder::MOVAPS_MOVAPDOp},
{OPD(1, 0b01, 0x29), 1, &OpDispatchBuilder::MOVAPS_MOVAPDOp},
{OPD(1, 0b10, 0x2A), 1, &OpDispatchBuilder::AVXCVTGPR_To_FPR<4>},
{OPD(1, 0b11, 0x2A), 1, &OpDispatchBuilder::AVXCVTGPR_To_FPR<8>},
{OPD(1, 0b10, 0x2A), 1, &OpDispatchBuilder::AVXInsertCVTGPR_To_FPR<4>},
{OPD(1, 0b11, 0x2A), 1, &OpDispatchBuilder::AVXInsertCVTGPR_To_FPR<8>},
{OPD(1, 0b00, 0x2B), 1, &OpDispatchBuilder::MOVVectorNTOp},
{OPD(1, 0b01, 0x2B), 1, &OpDispatchBuilder::MOVVectorNTOp},
@@ -5928,16 +5928,16 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
{OPD(1, 0b00, 0x50), 1, &OpDispatchBuilder::MOVMSKOp<4>},
{OPD(1, 0b01, 0x50), 1, &OpDispatchBuilder::MOVMSKOp<8>},
{OPD(1, 0b00, 0x51), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 4, false>},
{OPD(1, 0b01, 0x51), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 8, false>},
{OPD(1, 0b10, 0x51), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 4, true>},
{OPD(1, 0b11, 0x51), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 8, true>},
{OPD(1, 0b00, 0x51), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 4>},
{OPD(1, 0b01, 0x51), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 8>},
{OPD(1, 0b10, 0x51), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 4>},
{OPD(1, 0b11, 0x51), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 8>},
{OPD(1, 0b00, 0x52), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRSQRT, 4, false>},
{OPD(1, 0b10, 0x52), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRSQRT, 4, true>},
{OPD(1, 0b00, 0x52), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRSQRT, 4>},
{OPD(1, 0b10, 0x52), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 4>},
{OPD(1, 0b00, 0x53), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRECP, 4, false>},
{OPD(1, 0b10, 0x53), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRECP, 4, true>},
{OPD(1, 0b00, 0x53), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRECP, 4>},
{OPD(1, 0b10, 0x53), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 4>},
{OPD(1, 0b00, 0x54), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VAND, 16>},
{OPD(1, 0b01, 0x54), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VAND, 16>},
@@ -5953,18 +5953,18 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
{OPD(1, 0b00, 0x58), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFADD, 4>},
{OPD(1, 0b01, 0x58), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFADD, 8>},
{OPD(1, 0b10, 0x58), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFADD, 4>},
{OPD(1, 0b11, 0x58), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFADD, 8>},
{OPD(1, 0b10, 0x58), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 4>},
{OPD(1, 0b11, 0x58), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 8>},
{OPD(1, 0b00, 0x59), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFMUL, 4>},
{OPD(1, 0b01, 0x59), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFMUL, 8>},
{OPD(1, 0b10, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMUL, 4>},
{OPD(1, 0b11, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMUL, 8>},
{OPD(1, 0b10, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 4>},
{OPD(1, 0b11, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 8>},
{OPD(1, 0b00, 0x5A), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Float<8, 4>},
{OPD(1, 0b01, 0x5A), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Float<4, 8>},
{OPD(1, 0b10, 0x5A), 1, &OpDispatchBuilder::AVXScalar_CVT_Float_To_Float<8, 4>},
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::AVXScalar_CVT_Float_To_Float<4, 8>},
{OPD(1, 0b10, 0x5A), 1, &OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<8, 4>},
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<4, 8>},
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Int_To_Float<4, false>},
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Float_To_Int<4, false, true>},
@@ -5972,23 +5972,23 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
{OPD(1, 0b00, 0x5C), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFSUB, 4>},
{OPD(1, 0b01, 0x5C), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFSUB, 8>},
{OPD(1, 0b10, 0x5C), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFSUB, 4>},
{OPD(1, 0b11, 0x5C), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFSUB, 8>},
{OPD(1, 0b10, 0x5C), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 4>},
{OPD(1, 0b11, 0x5C), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 8>},
{OPD(1, 0b00, 0x5D), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFMIN, 4>},
{OPD(1, 0b01, 0x5D), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFMIN, 8>},
{OPD(1, 0b10, 0x5D), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMIN, 4>},
{OPD(1, 0b11, 0x5D), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMIN, 8>},
{OPD(1, 0b10, 0x5D), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 4>},
{OPD(1, 0b11, 0x5D), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 8>},
{OPD(1, 0b00, 0x5E), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFDIV, 4>},
{OPD(1, 0b01, 0x5E), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFDIV, 8>},
{OPD(1, 0b10, 0x5E), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFDIV, 4>},
{OPD(1, 0b11, 0x5E), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFDIV, 8>},
{OPD(1, 0b10, 0x5E), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 4>},
{OPD(1, 0b11, 0x5E), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 8>},
{OPD(1, 0b00, 0x5F), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFMAX, 4>},
{OPD(1, 0b01, 0x5F), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFMAX, 8>},
{OPD(1, 0b10, 0x5F), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMAX, 4>},
{OPD(1, 0b11, 0x5F), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMAX, 8>},
{OPD(1, 0b10, 0x5F), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 4>},
{OPD(1, 0b11, 0x5F), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 8>},
{OPD(1, 0b01, 0x60), 1, &OpDispatchBuilder::VPUNPCKLOp<1>},
{OPD(1, 0b01, 0x61), 1, &OpDispatchBuilder::VPUNPCKLOp<2>},
@@ -6030,10 +6030,10 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
{OPD(1, 0b01, 0x7F), 1, &OpDispatchBuilder::MOVAPS_MOVAPDOp},
{OPD(1, 0b10, 0x7F), 1, &OpDispatchBuilder::MOVUPS_MOVUPDOp},
{OPD(1, 0b00, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<4, false>},
{OPD(1, 0b01, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<8, false>},
{OPD(1, 0b10, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<4, true>},
{OPD(1, 0b11, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<8, true>},
{OPD(1, 0b00, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<4>},
{OPD(1, 0b01, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<8>},
{OPD(1, 0b10, 0xC2), 1, &OpDispatchBuilder::AVXInsertScalarFCMPOp<4>},
{OPD(1, 0b11, 0xC2), 1, &OpDispatchBuilder::AVXInsertScalarFCMPOp<8>},
{OPD(1, 0b01, 0xC4), 1, &OpDispatchBuilder::VPINSRWOp},
{OPD(1, 0b01, 0xC5), 1, &OpDispatchBuilder::PExtrOp<2>},
@@ -6124,9 +6124,9 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
{OPD(2, 0b01, 0x18), 1, &OpDispatchBuilder::VBROADCASTOp<4>},
{OPD(2, 0b01, 0x19), 1, &OpDispatchBuilder::VBROADCASTOp<8>},
{OPD(2, 0b01, 0x1A), 1, &OpDispatchBuilder::VBROADCASTOp<16>},
{OPD(2, 0b01, 0x1C), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 1, false>},
{OPD(2, 0b01, 0x1D), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 2, false>},
{OPD(2, 0b01, 0x1E), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 4, false>},
{OPD(2, 0b01, 0x1C), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 1>},
{OPD(2, 0b01, 0x1D), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 2>},
{OPD(2, 0b01, 0x1E), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 4>},
{OPD(2, 0b01, 0x20), 1, &OpDispatchBuilder::ExtendVectorElements<1, 2, true>},
{OPD(2, 0b01, 0x21), 1, &OpDispatchBuilder::ExtendVectorElements<1, 4, true>},
@@ -6190,10 +6190,10 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
{OPD(3, 0b01, 0x04), 1, &OpDispatchBuilder::VPERMILImmOp<4>},
{OPD(3, 0b01, 0x05), 1, &OpDispatchBuilder::VPERMILImmOp<8>},
{OPD(3, 0b01, 0x06), 1, &OpDispatchBuilder::VPERM2Op},
{OPD(3, 0b01, 0x08), 1, &OpDispatchBuilder::AVXVectorRound<4, false>},
{OPD(3, 0b01, 0x09), 1, &OpDispatchBuilder::AVXVectorRound<8, false>},
{OPD(3, 0b01, 0x0A), 1, &OpDispatchBuilder::AVXVectorRound<4, true>},
{OPD(3, 0b01, 0x0B), 1, &OpDispatchBuilder::AVXVectorRound<8, true>},
{OPD(3, 0b01, 0x08), 1, &OpDispatchBuilder::AVXVectorRound<4>},
{OPD(3, 0b01, 0x09), 1, &OpDispatchBuilder::AVXVectorRound<8>},
{OPD(3, 0b01, 0x0A), 1, &OpDispatchBuilder::AVXInsertScalarRound<4>},
{OPD(3, 0b01, 0x0B), 1, &OpDispatchBuilder::AVXInsertScalarRound<8>},
{OPD(3, 0b01, 0x0C), 1, &OpDispatchBuilder::VPBLENDDOp},
{OPD(3, 0b01, 0x0D), 1, &OpDispatchBuilder::VBLENDPDOp},
{OPD(3, 0b01, 0x0E), 1, &OpDispatchBuilder::VPBLENDWOp},
@@ -6441,15 +6441,15 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
{0x15, 1, &OpDispatchBuilder::PUNPCKHOp<4>},
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
{0x28, 2, &OpDispatchBuilder::MOVAPS_MOVAPDOp},
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, false>},
{0x2A, 1, &OpDispatchBuilder::InsertMMX_To_XMM_Vector_CVT_Int_To_Float},
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<4, false, false>},
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<4, false, true>},
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<4>},
{0x50, 1, &OpDispatchBuilder::MOVMSKOp<4>},
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4, false>},
{0x52, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4, false>},
{0x53, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4, false>},
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4>},
{0x52, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4>},
{0x53, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4>},
{0x54, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VAND, 16>},
{0x55, 1, &OpDispatchBuilder::VectorALUROp<IR::OP_VBIC, 8>},
{0x56, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VOR, 16>},
@@ -6481,7 +6481,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
{0x76, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VCMPEQ, 4>},
{0x77, 1, &OpDispatchBuilder::X87EMMS},
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<4, false>},
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<4>},
{0xC6, 1, &OpDispatchBuilder::SHUFOp<4>},
{0xD1, 1, &OpDispatchBuilder::PSRLDOp<2>},
@@ -6668,21 +6668,21 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
{0x12, 1, &OpDispatchBuilder::VMOVSLDUPOp},
{0x16, 1, &OpDispatchBuilder::VMOVSHDUPOp},
{0x19, 7, &OpDispatchBuilder::NOPOp},
{0x2A, 1, &OpDispatchBuilder::CVTGPR_To_FPR<4>},
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<4>},
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<4, false>},
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<4, true>},
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4, true>},
{0x52, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4, true>},
{0x53, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4, true>},
{0x58, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFADD, 4>},
{0x59, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMUL, 4>},
{0x5A, 1, &OpDispatchBuilder::Scalar_CVT_Float_To_Float<8, 4>},
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 4>},
{0x52, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 4>},
{0x53, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 4>},
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 4>},
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 4>},
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<8, 4>},
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>},
{0x5C, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFSUB, 4>},
{0x5D, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMIN, 4>},
{0x5E, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFDIV, 4>},
{0x5F, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMAX, 4>},
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 4>},
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 4>},
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 4>},
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 4>},
{0x6F, 1, &OpDispatchBuilder::MOVUPS_MOVUPDOp},
{0x70, 1, &OpDispatchBuilder::PSHUFWOp<false>},
{0x7E, 1, &OpDispatchBuilder::MOVQOp},
@@ -6690,7 +6690,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
{0xB8, 1, &OpDispatchBuilder::PopcountOp},
{0xBC, 1, &OpDispatchBuilder::TZCNT},
{0xBD, 1, &OpDispatchBuilder::LZCNT},
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<4, true>},
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<4>},
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<true>},
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<4, true>},
};
@@ -6699,25 +6699,25 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
{0x10, 2, &OpDispatchBuilder::MOVSDOp},
{0x12, 1, &OpDispatchBuilder::MOVDDUPOp},
{0x19, 7, &OpDispatchBuilder::NOPOp},
{0x2A, 1, &OpDispatchBuilder::CVTGPR_To_FPR<8>},
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<8>},
{0x2B, 1, &OpDispatchBuilder::MOVVectorOp},
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<8, false>},
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<8, true>},
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8, true>},
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 8>},
//x52 = Invalid
{0x58, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFADD, 8>},
{0x59, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMUL, 8>},
{0x5A, 1, &OpDispatchBuilder::Scalar_CVT_Float_To_Float<4, 8>},
{0x5C, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFSUB, 8>},
{0x5D, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMIN, 8>},
{0x5E, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFDIV, 8>},
{0x5F, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMAX, 8>},
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 8>},
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 8>},
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<4, 8>},
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 8>},
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 8>},
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 8>},
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 8>},
{0x70, 1, &OpDispatchBuilder::PSHUFWOp<true>},
{0x7C, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFADDP, 4>},
{0x7D, 1, &OpDispatchBuilder::HSUBP<4>},
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<4>},
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<false>},
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<8, true>},
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<8>},
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, true>},
{0xF0, 1, &OpDispatchBuilder::MOVVectorOp},
};
@@ -6730,7 +6730,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
{0x19, 7, &OpDispatchBuilder::NOPOp},
{0x28, 2, &OpDispatchBuilder::MOVAPS_MOVAPDOp},
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, true>},
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float},
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, false>},
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, true>},
@@ -6738,7 +6738,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
{0x40, 16, &OpDispatchBuilder::CMOVOp},
{0x50, 1, &OpDispatchBuilder::MOVMSKOp<8>},
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8, false>},
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8>},
{0x54, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VAND, 16>},
{0x55, 1, &OpDispatchBuilder::VectorALUROp<IR::OP_VBIC, 8>},
{0x56, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VOR, 16>},
@@ -6777,7 +6777,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
{0x7D, 1, &OpDispatchBuilder::HSUBP<8>},
{0x7E, 1, &OpDispatchBuilder::MOVBetweenGPR_FPR},
{0x7F, 1, &OpDispatchBuilder::MOVUPS_MOVUPDOp},
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<8, false>},
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<8>},
{0xC4, 1, &OpDispatchBuilder::PINSROp<2>},
{0xC5, 1, &OpDispatchBuilder::PExtrOp<2>},
{0xC6, 1, &OpDispatchBuilder::SHUFOp<8>},
@@ -7444,12 +7444,12 @@ constexpr uint16_t PF_F2 = 3;
{OPD(PF_38_66, 0x14), 1, &OpDispatchBuilder::VectorVariableBlend<4>},
{OPD(PF_38_66, 0x15), 1, &OpDispatchBuilder::VectorVariableBlend<8>},
{OPD(PF_38_66, 0x17), 1, &OpDispatchBuilder::PTestOp},
{OPD(PF_38_NONE, 0x1C), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 1, false>},
{OPD(PF_38_66, 0x1C), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 1, false>},
{OPD(PF_38_NONE, 0x1D), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2, false>},
{OPD(PF_38_66, 0x1D), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2, false>},
{OPD(PF_38_NONE, 0x1E), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4, false>},
{OPD(PF_38_66, 0x1E), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4, false>},
{OPD(PF_38_NONE, 0x1C), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 1>},
{OPD(PF_38_66, 0x1C), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 1>},
{OPD(PF_38_NONE, 0x1D), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2>},
{OPD(PF_38_66, 0x1D), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2>},
{OPD(PF_38_NONE, 0x1E), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4>},
{OPD(PF_38_66, 0x1E), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4>},
{OPD(PF_38_66, 0x20), 1, &OpDispatchBuilder::ExtendVectorElements<1, 2, true>},
{OPD(PF_38_66, 0x21), 1, &OpDispatchBuilder::ExtendVectorElements<1, 4, true>},
{OPD(PF_38_66, 0x22), 1, &OpDispatchBuilder::ExtendVectorElements<1, 8, true>},
@@ -7491,10 +7491,10 @@ constexpr uint16_t PF_F2 = 3;
#define PF_3A_NONE 0
#define PF_3A_66 1
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F3ATable[] = {
{OPD(0, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<4, false>},
{OPD(0, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<8, false>},
{OPD(0, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::VectorRound<4, true>},
{OPD(0, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::VectorRound<8, true>},
{OPD(0, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<4>},
{OPD(0, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<8>},
{OPD(0, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<4>},
{OPD(0, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::InsertScalarRound<8>},
{OPD(0, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::VectorBlend<4>},
{OPD(0, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::VectorBlend<8>},
{OPD(0, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::VectorBlend<2>},
@@ -7535,8 +7535,8 @@ constexpr uint16_t PF_F2 = 3;
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>},
{0x86, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4, false>},
{0x87, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4, false>},
{0x86, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4>},
{0x87, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4>},
{0x8A, 1, &OpDispatchBuilder::PFNACCOp},
{0x8E, 1, &OpDispatchBuilder::PFPNACCOp},
@@ -333,8 +333,6 @@ public:
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void VectorALUROp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void VectorScalarALUOp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize, bool Scalar>
void VectorUnaryOp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void VectorUnaryDuplicateOp(OpcodeArgs);
@@ -379,7 +377,6 @@ public:
void Vector_CVT_Float_To_Float(OpcodeArgs);
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
void Vector_CVT_Float_To_Int(OpcodeArgs);
template<size_t SrcElementSize, bool Widen>
void MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs);
@@ -387,7 +384,7 @@ public:
void MOVBetweenGPR_FPR(OpcodeArgs);
void TZCNT(OpcodeArgs);
void LZCNT(OpcodeArgs);
template<size_t ElementSize, bool Scalar>
template<size_t ElementSize>
void VFCMPOp(OpcodeArgs);
template<size_t ElementSize>
void SHUFOp(OpcodeArgs);
@@ -424,11 +421,9 @@ public:
template <IROps IROp, size_t ElementSize>
void AVXVectorALUOp(OpcodeArgs);
template <IROps IROp, size_t ElementSize>
void AVXVectorScalarALUOp(OpcodeArgs);
template <IROps IROp, size_t ElementSize, bool Scalar>
void AVXVectorUnaryOp(OpcodeArgs);
template <size_t ElementSize, bool Scalar>
template <size_t ElementSize>
void AVXVectorRound(OpcodeArgs);
template <size_t DstElementSize, size_t SrcElementSize>
@@ -440,10 +435,41 @@ public:
template <size_t SrcElementSize, bool Widen>
void AVXVector_CVT_Int_To_Float(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void VectorScalarInsertALUOp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void AVXVectorScalarInsertALUOp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void VectorScalarUnaryInsertALUOp(OpcodeArgs);
template<FEXCore::IR::IROps IROp, size_t ElementSize>
void AVXVectorScalarUnaryInsertALUOp(OpcodeArgs);
void InsertMMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
template<size_t DstElementSize>
void InsertCVTGPR_To_FPR(OpcodeArgs);
template <size_t DstElementSize>
void AVXInsertCVTGPR_To_FPR(OpcodeArgs);
template<size_t DstElementSize, size_t SrcElementSize>
void InsertScalar_CVT_Float_To_Float(OpcodeArgs);
template<size_t DstElementSize, size_t SrcElementSize>
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs);
template <size_t ElementSize>
void InsertScalarRound(OpcodeArgs);
template <size_t ElementSize>
void AVXInsertScalarRound(OpcodeArgs);
template <size_t ElementSize>
void InsertScalarFCMPOp(OpcodeArgs);
template <size_t ElementSize>
void AVXInsertScalarFCMPOp(OpcodeArgs);
template <size_t DstElementSize>
void AVXCVTGPR_To_FPR(OpcodeArgs);
template <size_t ElementSize, bool Scalar>
template <size_t ElementSize>
void AVXVFCMPOp(OpcodeArgs);
template <size_t ElementSize>
@@ -787,7 +813,7 @@ public:
template<size_t ElementSize, size_t DstElementSize, bool Signed>
void ExtendVectorElements(OpcodeArgs);
template<size_t ElementSize, bool Scalar>
template<size_t ElementSize>
void VectorRound(OpcodeArgs);
template<size_t ElementSize>
@@ -889,8 +915,7 @@ private:
OrderedNode *Src1, OrderedNode *Src2);
void AVXVectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
void AVXVectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
void AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar);
void AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
template <size_t ElementSize>
void AVXVectorVariableBlend(OpcodeArgs);
@@ -997,19 +1022,63 @@ private:
void MOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
void VMOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
OrderedNode* VFCMPOpImpl(OpcodeArgs, size_t ElementSize, bool Scalar,
OrderedNode* VFCMPOpImpl(OpcodeArgs, size_t ElementSize,
OrderedNode *Src1, OrderedNode *Src2, uint8_t CompType);
void VTESTOpImpl(OpcodeArgs, size_t ElementSize);
void VectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
void VectorALUROpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
void VectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
void VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar);
void VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
void VectorUnaryDuplicateOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
// x86 ALU scalar operations operate in three different ways
// - AVX512: Writemask shenanigans that we don't care about.
// - AVX/VEX: Two source
// - Example 32bit VADDSS Dest, Src1, Src2
// - Dest[31:0] = Src1[31:0] + Src2[31:0]
// - Dest[127:32] = Src1[127:32]
// - SSE: Scalar operation inserts in to the low bits, upper bits completely unaffected.
// - Example 32bit ADDSS Dest, Src
// - Dest[31:0] = Dest[31:0] + Src[31:0]
// - Dest[{256,128}:32] = (Unmodified)
OrderedNode* VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IROp,
size_t DstSize, size_t ElementSize,
const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op,
bool ZeroUpperBits);
OrderedNode* VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IROps IROp,
size_t DstSize, size_t ElementSize,
const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op,
bool ZeroUpperBits);
OrderedNode* InsertCVTGPR_To_FPRImpl(OpcodeArgs,
size_t DstSize, size_t DstElementSize,
const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op,
bool ZeroUpperBits);
OrderedNode* InsertScalar_CVT_Float_To_FloatImpl(OpcodeArgs,
size_t DstSize, size_t DstElementSize, size_t SrcElementSize,
const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op,
bool ZeroUpperBits);
OrderedNode* InsertScalarRoundImpl(OpcodeArgs,
size_t DstSize, size_t ElementSize,
const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op,
uint64_t Mode, bool ZeroUpperBits);
OrderedNode* InsertScalarFCMPOpImpl(OpcodeArgs,
size_t DstSize, size_t ElementSize,
const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op,
uint8_t CompType, bool ZeroUpperBits);
OrderedNode* VectorRoundImpl(OpcodeArgs, size_t ElementSize,
OrderedNode *Src, uint64_t Mode, bool IsScalar);
OrderedNode *Src, uint64_t Mode);
OrderedNode* Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize,
const X86Tables::DecodedOperand& Src1Op,
@@ -1067,6 +1136,10 @@ private:
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
constexpr OpSize GetGuestVectorLength() const {
return CTX->HostFeatures.SupportsAVX ? OpSize::i256Bit : OpSize::i128Bit;
}
[[nodiscard]] static uint32_t GPROffset(X86State::X86Reg reg) {
LOGMAN_THROW_AA_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
return static_cast<uint32_t>(offsetof(Core::CPUState, gregs[static_cast<size_t>(reg)]));
@@ -500,236 +500,487 @@ void OpDispatchBuilder::VectorALUROp<IR::OP_VFSUB, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorALUROp<IR::OP_VFSUB, 8>(OpcodeArgs);
void OpDispatchBuilder::VectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize) {
OrderedNode* OpDispatchBuilder::VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IROp,
size_t DstSize, size_t ElementSize,
const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op,
bool ZeroUpperBits) {
// We load the full vector width when dealing with a source vector,
// so that we don't do any unnecessary zero extension to the scalar
// element that we're going to operate on.
const auto SrcSize = Op->Src[0].IsGPR() ? 16U : GetSrcSize(Op);
const auto DstSize = GetDstSize(Op);
const auto SrcSize = GetSrcSize(Op);
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags, -1);
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, -1, true, false, MemoryAccessType::ACCESS_DEFAULT, true);
// If OpSize == ElementSize then it only does the lower scalar op
auto ALUOp = _VFAddScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, ZeroUpperBits);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
return ALUOp;
}
template <IROps IROp, size_t ElementSize>
void OpDispatchBuilder::VectorScalarInsertALUOp(OpcodeArgs) {
const auto DstSize = GetGuestVectorLength();
auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Dest, Op->Src[0], false);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
}
template
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 8>(OpcodeArgs);
template <IROps IROp, size_t ElementSize>
void OpDispatchBuilder::AVXVectorScalarInsertALUOp(OpcodeArgs) {
const auto DstSize = GetGuestVectorLength();
auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Src[0], Op->Src[1], true);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
}
template
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 8>(OpcodeArgs);
OrderedNode* OpDispatchBuilder::VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IROps IROp,
size_t DstSize, size_t ElementSize,
const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op,
bool ZeroUpperBits) {
// We load the full vector width when dealing with a source vector,
// so that we don't do any unnecessary zero extension to the scalar
// element that we're going to operate on.
const auto SrcSize = GetSrcSize(Op);
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags, -1);
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, -1, true, false, MemoryAccessType::ACCESS_DEFAULT, true);
// If OpSize == ElementSize then it only does the lower scalar op
auto ALUOp = _VFSqrtScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, ZeroUpperBits);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
return ALUOp;
}
template <IROps IROp, size_t ElementSize>
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp(OpcodeArgs) {
const auto DstSize = GetGuestVectorLength();
auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Dest, Op->Src[0], false);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
}
template
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 8>(OpcodeArgs);
template <IROps IROp, size_t ElementSize>
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp(OpcodeArgs) {
const auto DstSize = GetGuestVectorLength();
auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Src[0], Op->Src[1], true);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
}
template
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 8>(OpcodeArgs);
void OpDispatchBuilder::InsertMMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) {
// We load the full vector width when dealing with a source vector,
// so that we don't do any unnecessary zero extension to the scalar
// element that we're going to operate on.
const auto DstSize = GetGuestVectorLength();
const auto SrcSize = Op->Src[0].IsGPR() ? 8 : GetSrcSize(Op);
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1);
// If OpSize == ElementSize then it only does the lower scalar op
auto ALUOp = _VAdd(ElementSize, ElementSize, Dest, Src);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
// Always 32-bit.
const size_t ElementSize = 4;
// Always signed
Dest = _VSToFVectorInsert(IR::SizeToOpSize(DstSize), ElementSize, ElementSize, Dest, Src, true, false);
OrderedNode* Result = ALUOp;
if (DstSize != ElementSize) {
// Insert the lower bits
Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, ALUOp);
}
StoreResult(FPRClass, Op, Result, -1);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Dest, DstSize, -1);
}
template <IROps IROp, size_t ElementSize>
void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs) {
VectorScalarALUOpImpl(Op, IROp, ElementSize);
}
template
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFADD, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFADD, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFSUB, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFSUB, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMUL, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMUL, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFDIV, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFDIV, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMIN, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMIN, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMAX, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMAX, 8>(OpcodeArgs);
void OpDispatchBuilder::AVXVectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize) {
OrderedNode* OpDispatchBuilder::InsertCVTGPR_To_FPRImpl(OpcodeArgs,
size_t DstSize, size_t DstElementSize,
const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op,
bool ZeroUpperBits) {
// We load the full vector width when dealing with a source vector,
// so that we don't do any unnecessary zero extension to the scalar
// element that we're going to operate on.
const auto SrcSize = Op->Src[1].IsGPR() ? 16U : GetSrcSize(Op);
const auto DstSize = GetDstSize(Op);
const auto SrcSize = GetSrcSize(Op);
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags, -1);
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags, -1);
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags, -1);
// If OpSize == ElementSize then it only does the lower scalar op
auto ALUOp = _VAdd(ElementSize, ElementSize, Src1, Src2);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
OrderedNode* Result = ALUOp;
if (DstSize != ElementSize) {
// Insert the lower bits
Result = _VInsElement(DstSize, ElementSize, 0, 0, Src1, ALUOp);
if (Src2Op.IsGPR()) {
// If the source is a GPR then convert directly from the GPR.
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, CTX->GetGPRSize(), Op->Flags, -1);
return _VSToFGPRInsert(IR::SizeToOpSize(DstSize), DstElementSize, SrcSize, Src1, Src2, ZeroUpperBits);
}
else if (SrcSize != DstElementSize) {
// If the source is from memory but the Source size and destination size aren't the same,
// then it is more optimal to load in to a GPR and convert between GPR->FPR.
// ARM GPR->FPR conversion supports different size source and destinations while FPR->FPR doesn't.
auto Src2 = LoadSource(GPRClass, Op, Src2Op, Op->Flags, -1);
return _VSToFGPRInsert(IR::SizeToOpSize(DstSize), DstElementSize, SrcSize, Src1, Src2, ZeroUpperBits);
}
StoreResult(FPRClass, Op, Result, -1);
// In the case of cvtsi2s{s,d} where the source and destination are the same size,
// then it is more optimal to load in to the FPR register directly and convert there.
auto Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
// Always signed
return _VSToFVectorInsert(IR::SizeToOpSize(DstSize), DstElementSize, DstElementSize, Src1, Src2, false, ZeroUpperBits);
}
template <IROps IROp, size_t ElementSize>
void OpDispatchBuilder::AVXVectorScalarALUOp(OpcodeArgs) {
AVXVectorScalarALUOpImpl(Op, IROp, ElementSize);
template<size_t DstElementSize>
void OpDispatchBuilder::InsertCVTGPR_To_FPR(OpcodeArgs) {
const auto DstSize = GetGuestVectorLength();
auto Result = InsertCVTGPR_To_FPRImpl(Op, DstSize, DstElementSize, Op->Dest, Op->Src[0], false);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
}
template
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFADD, 4>(OpcodeArgs);
void OpDispatchBuilder::InsertCVTGPR_To_FPR<4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFADD, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFDIV, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFDIV, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMAX, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMAX, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMIN, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMIN, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMUL, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMUL, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFSUB, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFSUB, 8>(OpcodeArgs);
void OpDispatchBuilder::InsertCVTGPR_To_FPR<8>(OpcodeArgs);
void OpDispatchBuilder::VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar) {
template <size_t DstElementSize>
void OpDispatchBuilder::AVXInsertCVTGPR_To_FPR(OpcodeArgs) {
const auto DstSize = GetGuestVectorLength();
OrderedNode *Result = InsertCVTGPR_To_FPRImpl(Op, DstSize, DstElementSize, Op->Src[0], Op->Src[1], true);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
}
template
void OpDispatchBuilder::AVXInsertCVTGPR_To_FPR<4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXInsertCVTGPR_To_FPR<8>(OpcodeArgs);
OrderedNode* OpDispatchBuilder::InsertScalar_CVT_Float_To_FloatImpl(OpcodeArgs,
size_t DstSize, size_t DstElementSize, size_t SrcElementSize,
const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op,
bool ZeroUpperBits) {
// We load the full vector width when dealing with a source vector,
// so that we don't do any unnecessary zero extension to the scalar
// element that we're going to operate on.
const auto SrcSize = GetSrcSize(Op);
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags, -1);
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, -1, true, false, MemoryAccessType::ACCESS_DEFAULT, true);
return _VFToFScalarInsert(IR::SizeToOpSize(DstSize), DstElementSize, SrcElementSize, Src1, Src2, ZeroUpperBits);
}
template<size_t DstElementSize, size_t SrcElementSize>
void OpDispatchBuilder::InsertScalar_CVT_Float_To_Float(OpcodeArgs) {
const auto DstSize = GetGuestVectorLength();
OrderedNode *Result = InsertScalar_CVT_Float_To_FloatImpl(Op, DstSize, DstElementSize, SrcElementSize, Op->Dest, Op->Src[0], false);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
}
template
void OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<4, 8>(OpcodeArgs);
template
void OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<8, 4>(OpcodeArgs);
template <size_t DstElementSize, size_t SrcElementSize>
void OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs) {
const auto DstSize = GetGuestVectorLength();
OrderedNode *Result = InsertScalar_CVT_Float_To_FloatImpl(Op, DstSize, DstElementSize, SrcElementSize, Op->Src[0], Op->Src[1], true);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
}
template
void OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<4, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<8, 4>(OpcodeArgs);
OrderedNode* OpDispatchBuilder::InsertScalarRoundImpl(OpcodeArgs,
size_t DstSize, size_t ElementSize,
const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op,
uint64_t Mode, bool ZeroUpperBits) {
// We load the full vector width when dealing with a source vector,
// so that we don't do any unnecessary zero extension to the scalar
// element that we're going to operate on.
const auto SrcSize = GetSrcSize(Op);
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags, -1);
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, -1, true, false, MemoryAccessType::ACCESS_DEFAULT, true);
const uint64_t RoundControlSource = (Mode >> 2) & 1;
uint64_t RoundControl = Mode & 0b11;
static constexpr std::array SourceModes = {
FEXCore::IR::Round_Nearest,
FEXCore::IR::Round_Negative_Infinity,
FEXCore::IR::Round_Positive_Infinity,
FEXCore::IR::Round_Towards_Zero,
};
const auto SourceMode = RoundControlSource ? Round_Host : SourceModes[RoundControl];
auto ALUOp = _VFToIScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, SourceMode, ZeroUpperBits);
return ALUOp;
}
template<size_t ElementSize>
void OpDispatchBuilder::InsertScalarRound(OpcodeArgs) {
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
const uint64_t Mode = Op->Src[1].Data.Literal.Value;
const auto DstSize = GetGuestVectorLength();
OrderedNode *Result = InsertScalarRoundImpl(Op, DstSize, ElementSize, Op->Dest, Op->Src[0], Mode, false);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
}
template
void OpDispatchBuilder::InsertScalarRound<4>(OpcodeArgs);
template
void OpDispatchBuilder::InsertScalarRound<8>(OpcodeArgs);
template<size_t ElementSize>
void OpDispatchBuilder::AVXInsertScalarRound(OpcodeArgs) {
LOGMAN_THROW_A_FMT(Op->Src[2].IsLiteral(), "Src1 needs to be literal here");
const uint64_t Mode = Op->Src[2].Data.Literal.Value;
const auto DstSize = GetGuestVectorLength();
OrderedNode *Result = InsertScalarRoundImpl(Op, DstSize, ElementSize, Op->Dest, Op->Src[0], Mode, true);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
}
template
void OpDispatchBuilder::AVXInsertScalarRound<4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXInsertScalarRound<8>(OpcodeArgs);
OrderedNode* OpDispatchBuilder::InsertScalarFCMPOpImpl(OpcodeArgs,
size_t DstSize, size_t ElementSize,
const X86Tables::DecodedOperand& Src1Op,
const X86Tables::DecodedOperand& Src2Op,
uint8_t CompType, bool ZeroUpperBits) {
// We load the full vector width when dealing with a source vector,
// so that we don't do any unnecessary zero extension to the scalar
// element that we're going to operate on.
const auto SrcSize = GetSrcSize(Op);
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags, -1);
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, -1, true, false, MemoryAccessType::ACCESS_DEFAULT, true);
switch (CompType) {
case 0x00: case 0x08: case 0x10: case 0x18: // EQ
return _VFCMPScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, FloatCompareOp::EQ, ZeroUpperBits);
case 0x01: case 0x09: case 0x11: case 0x19: // LT, GT(Swapped operand)
return _VFCMPScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, FloatCompareOp::LT, ZeroUpperBits);
case 0x02: case 0x0A: case 0x12: case 0x1A: // LE, GE(Swapped operand)
return _VFCMPScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, FloatCompareOp::LE, ZeroUpperBits);
case 0x03: case 0x0B: case 0x13: case 0x1B: // Unordered
return _VFCMPScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, FloatCompareOp::UNO, ZeroUpperBits);
case 0x04: case 0x0C: case 0x14: case 0x1C: // NEQ
return _VFCMPScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, FloatCompareOp::NEQ, ZeroUpperBits);
case 0x05: case 0x0D: case 0x15: case 0x1D: { // NLT, NGT(Swapped operand)
OrderedNode *Result = _VFCMPLT(ElementSize, ElementSize, Src1, Src2);
Result = _VNot(ElementSize, ElementSize, Result);
// Insert the lower bits
return _VInsElement(GetDstSize(Op), ElementSize, 0, 0, Src1, Result);
}
case 0x06: case 0x0E: case 0x16: case 0x1E: { // NLE, NGE(Swapped operand)
OrderedNode *Result = _VFCMPLE(ElementSize, ElementSize, Src1, Src2);
Result = _VNot(ElementSize, ElementSize, Result);
// Insert the lower bits
return _VInsElement(GetDstSize(Op), ElementSize, 0, 0, Src1, Result);
}
case 0x07: case 0x0F: case 0x17: case 0x1F: // Ordered
return _VFCMPScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, FloatCompareOp::ORD, ZeroUpperBits);
default:
LOGMAN_MSG_A_FMT("Unknown Comparison type: {}", CompType);
break;
}
FEX_UNREACHABLE;
}
template<size_t ElementSize>
void OpDispatchBuilder::InsertScalarFCMPOp(OpcodeArgs) {
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src[2] needs to be literal");
const uint8_t CompType = Op->Src[1].Data.Literal.Value;
const auto DstSize = GetGuestVectorLength();
OrderedNode *Result = InsertScalarFCMPOpImpl(Op, DstSize, ElementSize, Op->Dest, Op->Src[0], CompType, false);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
}
template
void OpDispatchBuilder::InsertScalarFCMPOp<4>(OpcodeArgs);
template
void OpDispatchBuilder::InsertScalarFCMPOp<8>(OpcodeArgs);
template<size_t ElementSize>
void OpDispatchBuilder::AVXInsertScalarFCMPOp(OpcodeArgs) {
LOGMAN_THROW_A_FMT(Op->Src[2].IsLiteral(), "Src[2] needs to be literal");
const uint8_t CompType = Op->Src[2].Data.Literal.Value;
const auto DstSize = GetGuestVectorLength();
OrderedNode *Result = InsertScalarFCMPOpImpl(Op, DstSize, ElementSize, Op->Src[0], Op->Src[1], CompType, true);
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
}
template
void OpDispatchBuilder::AVXInsertScalarFCMPOp<4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXInsertScalarFCMPOp<8>(OpcodeArgs);
void OpDispatchBuilder::VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize) {
// In the event of a scalar operation and a vector source, then
// we can specify the entire vector length in order to avoid
// unnecessary sign extension on the element to be operated on.
// In the event of a memory operand, we load the exact element size.
const auto SrcSize = Scalar && Op->Src[0].IsGPR() ? 16U : GetSrcSize(Op);
const auto OpSize = Scalar ? ElementSize : GetSrcSize(Op);
const auto DstSize = GetDstSize(Op);
const auto SrcSize = GetSrcSize(Op);
const auto OpSize = GetSrcSize(Op);
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1);
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
auto ALUOp = _VFSqrt(OpSize, ElementSize, Src);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
if (Scalar) {
// Insert the lower bits
auto Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, ALUOp);
StoreResult(FPRClass, Op, Result, -1);
} else {
StoreResult(FPRClass, Op, ALUOp, -1);
}
StoreResult(FPRClass, Op, ALUOp, -1);
}
template <IROps IROp, size_t ElementSize, bool Scalar>
template <IROps IROp, size_t ElementSize>
void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs) {
VectorUnaryOpImpl(Op, IROp, ElementSize, Scalar);
VectorUnaryOpImpl(Op, IROp, ElementSize);
}
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4, false>(OpcodeArgs);
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4, false>(OpcodeArgs);
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4, false>(OpcodeArgs);
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4, true>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4, true>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4, true>(OpcodeArgs);
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8, false>(OpcodeArgs);
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 1>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8, true>(OpcodeArgs);
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 1, false>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2, false>(OpcodeArgs);
template
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4, false>(OpcodeArgs);
void OpDispatchBuilder::AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar) {
void OpDispatchBuilder::AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize) {
// In the event of a scalar operation and a vector source, then
// we can specify the entire vector length in order to avoid
// unnecessary sign extension on the element to be operated on.
// In the event of a memory operand, we load the exact element size.
const auto SrcSize = Scalar && Op->Src[1].IsGPR() ? 16U : GetSrcSize(Op);
const auto OpSize = Scalar ? ElementSize : GetSrcSize(Op);
const auto DstSize = GetDstSize(Op);
const auto SrcSize = GetSrcSize(Op);
const auto OpSize = GetSrcSize(Op);
OrderedNode *Src = [&] {
const auto SrcIndex = Scalar ? 1 : 0;
return LoadSource_WithOpSize(FPRClass, Op, Op->Src[SrcIndex], SrcSize, Op->Flags, -1);
}();
OrderedNode *Dest = [&] {
const auto& Operand = Scalar ? Op->Src[0] : Op->Dest;
return LoadSource_WithOpSize(FPRClass, Op, Operand, DstSize, Op->Flags, -1);
}();
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1);
auto ALUOp = _VFSqrt(OpSize, ElementSize, Src);
// Overwrite our IR's op type
ALUOp.first->Header.Op = IROp;
OrderedNode* Result = ALUOp;
if (Scalar) {
// Insert the lower bits
Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, Result);
}
// NOTE: We don't need to clear the upper lanes here, since the
// IR ops make use of 128-bit AdvSimd for 128-bit cases,
// which, on hardware with SVE, zero-extends as part of
// storing into the destination.
StoreResult(FPRClass, Op, Result, -1);
StoreResult(FPRClass, Op, ALUOp, -1);
}
template <IROps IROp, size_t ElementSize, bool Scalar>
template <IROps IROp, size_t ElementSize>
void OpDispatchBuilder::AVXVectorUnaryOp(OpcodeArgs) {
AVXVectorUnaryOpImpl(Op, IROp, ElementSize, Scalar);
AVXVectorUnaryOpImpl(Op, IROp, ElementSize);
}
template
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 1, false>(OpcodeArgs);
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 1>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 2, false>(OpcodeArgs);
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 2>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 4, false>(OpcodeArgs);
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRECP, 4, false>(OpcodeArgs);
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRECP, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRECP, 4, true>(OpcodeArgs);
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 4, false>(OpcodeArgs);
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 8>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 4, true>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 8, false>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 8, true>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRSQRT, 4, false>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRSQRT, 4, true>(OpcodeArgs);
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRSQRT, 4>(OpcodeArgs);
void OpDispatchBuilder::VectorUnaryDuplicateOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize) {
const auto Size = GetSrcSize(Op);
@@ -2411,38 +2662,22 @@ void OpDispatchBuilder::Vector_CVT_Float_To_Float<4, 8>(OpcodeArgs);
template
void OpDispatchBuilder::Vector_CVT_Float_To_Float<8, 4>(OpcodeArgs);
template<size_t SrcElementSize, bool Widen>
void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
size_t ElementSize = SrcElementSize;
// Always 32-bit.
size_t ElementSize = 4;
size_t DstSize = GetDstSize(Op);
if constexpr (Widen) {
Src = _VSXTL(DstSize, ElementSize, Src);
ElementSize <<= 1;
}
Src = _VSXTL(DstSize, ElementSize, Src);
ElementSize <<= 1;
// Always signed
Src = _Vector_SToF(DstSize, ElementSize, Src);
OrderedNode *Dest{};
if constexpr (Widen) {
Dest = Src;
}
else {
Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
// Insert the lower bits
Dest = _VInsElement(GetDstSize(Op), 8, 0, 0, Dest, Src);
}
StoreResult(FPRClass, Op, Dest, -1);
StoreResult(FPRClass, Op, Src, -1);
}
template
void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, false>(OpcodeArgs);
template
void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, true>(OpcodeArgs);
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs) {
// If loading a vector, use the full size, so we don't
@@ -2578,7 +2813,7 @@ void OpDispatchBuilder::MOVBetweenGPR_FPR(OpcodeArgs) {
}
}
OrderedNode* OpDispatchBuilder::VFCMPOpImpl(OpcodeArgs, size_t ElementSize, bool Scalar,
OrderedNode* OpDispatchBuilder::VFCMPOpImpl(OpcodeArgs, size_t ElementSize,
OrderedNode *Src1, OrderedNode *Src2, uint8_t CompType) {
const auto Size = GetSrcSize(Op);
@@ -2615,44 +2850,35 @@ OrderedNode* OpDispatchBuilder::VFCMPOpImpl(OpcodeArgs, size_t ElementSize, bool
break;
}
if (Scalar) {
// Insert the lower bits
Result = _VInsElement(GetDstSize(Op), ElementSize, 0, 0, Src1, Result);
}
return Result;
}
template<size_t ElementSize, bool Scalar>
template<size_t ElementSize>
void OpDispatchBuilder::VFCMPOp(OpcodeArgs) {
// No need for zero-extending in the scalar case, since
// all we need is an insert at the end of the operation.
const auto SrcSize = Scalar && Op->Src[0].IsGPR() ? 16U : GetSrcSize(Op);
const auto SrcSize = GetSrcSize(Op);
const auto DstSize = GetDstSize(Op);
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1);
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
const uint8_t CompType = Op->Src[1].Data.Literal.Value;
OrderedNode* Result = VFCMPOpImpl(Op, ElementSize, Scalar, Dest, Src, CompType);
OrderedNode* Result = VFCMPOpImpl(Op, ElementSize, Dest, Src, CompType);
StoreResult(FPRClass, Op, Result, -1);
}
template
void OpDispatchBuilder::VFCMPOp<4, false>(OpcodeArgs);
void OpDispatchBuilder::VFCMPOp<4>(OpcodeArgs);
template
void OpDispatchBuilder::VFCMPOp<4, true>(OpcodeArgs);
template
void OpDispatchBuilder::VFCMPOp<8, false>(OpcodeArgs);
template
void OpDispatchBuilder::VFCMPOp<8, true>(OpcodeArgs);
void OpDispatchBuilder::VFCMPOp<8>(OpcodeArgs);
template <size_t ElementSize, bool Scalar>
template <size_t ElementSize>
void OpDispatchBuilder::AVXVFCMPOp(OpcodeArgs) {
// No need for zero-extending in the scalar case, since
// all we need is an insert at the end of the operation.
const auto SrcSize = Scalar && Op->Src[1].IsGPR() ? 16U : GetSrcSize(Op);
const auto SrcSize = GetSrcSize(Op);
const auto DstSize = GetDstSize(Op);
LOGMAN_THROW_A_FMT(Op->Src[2].IsLiteral(), "Src[2] needs to be literal");
@@ -2660,19 +2886,15 @@ void OpDispatchBuilder::AVXVFCMPOp(OpcodeArgs) {
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags, -1);
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags, -1);
OrderedNode *Result = VFCMPOpImpl(Op, ElementSize, Scalar, Src1, Src2, CompType);
OrderedNode *Result = VFCMPOpImpl(Op, ElementSize, Src1, Src2, CompType);
StoreResult(FPRClass, Op, Result, -1);
}
template
void OpDispatchBuilder::AVXVFCMPOp<4, false>(OpcodeArgs);
void OpDispatchBuilder::AVXVFCMPOp<4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVFCMPOp<4, true>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVFCMPOp<8, false>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVFCMPOp<8, true>(OpcodeArgs);
void OpDispatchBuilder::AVXVFCMPOp<8>(OpcodeArgs);
void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
@@ -3928,8 +4150,7 @@ template
void OpDispatchBuilder::ExtendVectorElements<4, 8, true>(OpcodeArgs);
OrderedNode* OpDispatchBuilder::VectorRoundImpl(OpcodeArgs, size_t ElementSize,
OrderedNode *Src, uint64_t Mode,
bool IsScalar) {
OrderedNode *Src, uint64_t Mode) {
const auto Size = GetDstSize(Op);
const uint64_t RoundControlSource = (Mode >> 2) & 1;
uint64_t RoundControl = Mode & 0b11;
@@ -3947,82 +4168,48 @@ OrderedNode* OpDispatchBuilder::VectorRoundImpl(OpcodeArgs, size_t ElementSize,
};
const auto SourceMode = SourceModes[(RoundControlSource << 2) | RoundControl];
const auto OpSize = IsScalar ? ElementSize : Size;
return _Vector_FToI(OpSize, ElementSize, Src, SourceMode);
return _Vector_FToI(Size, ElementSize, Src, SourceMode);
}
template<size_t ElementSize, bool Scalar>
template<size_t ElementSize>
void OpDispatchBuilder::VectorRound(OpcodeArgs) {
// No need to zero extend the vector in the event we have a
// scalar source, especially since it's only inserted into another vector.
const auto SrcSize = Scalar && Op->Src[0].IsGPR() ? 16U : GetSrcSize(Op);
const auto DstSize = GetDstSize(Op);
const auto SrcSize = GetSrcSize(Op);
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1);
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
const uint64_t Mode = Op->Src[1].Data.Literal.Value;
Src = VectorRoundImpl(Op, ElementSize, Src, Mode, Scalar);
Src = VectorRoundImpl(Op, ElementSize, Src, Mode);
if constexpr (Scalar) {
// Insert the lower bits
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
auto Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, Src);
StoreResult(FPRClass, Op, Result, -1);
} else {
StoreResult(FPRClass, Op, Src, -1);
}
StoreResult(FPRClass, Op, Src, -1);
}
template
void OpDispatchBuilder::VectorRound<4, false>(OpcodeArgs);
void OpDispatchBuilder::VectorRound<4>(OpcodeArgs);
template
void OpDispatchBuilder::VectorRound<8, false>(OpcodeArgs);
void OpDispatchBuilder::VectorRound<8>(OpcodeArgs);
template
void OpDispatchBuilder::VectorRound<4, true>(OpcodeArgs);
template
void OpDispatchBuilder::VectorRound<8, true>(OpcodeArgs);
template <size_t ElementSize, bool Scalar>
template <size_t ElementSize>
void OpDispatchBuilder::AVXVectorRound(OpcodeArgs) {
const auto GetMode = [&] {
if constexpr (Scalar) {
LOGMAN_THROW_A_FMT(Op->Src[2].IsLiteral(), "Src2 needs to be literal here");
return Op->Src[2].Data.Literal.Value;
} else {
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
return Op->Src[1].Data.Literal.Value;
}
};
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
const auto Mode = Op->Src[1].Data.Literal.Value;
// No need to zero extend the vector in the event we have a
// scalar source, especially since it's only inserted into another vector.
const auto SrcIdx = Scalar ? 1 : 0;
const auto SrcSize = Scalar && Op->Src[SrcIdx].IsGPR() ? 16U : GetSrcSize(Op);
const auto DstSize = GetDstSize(Op);
const auto SrcSize = GetSrcSize(Op);
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[SrcIdx], SrcSize, Op->Flags, -1);
OrderedNode *Result = VectorRoundImpl(Op, ElementSize, Src, GetMode(), Scalar);
if constexpr (Scalar) {
// Insert the lower bits
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags, -1);
Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, Result);
}
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1);
OrderedNode *Result = VectorRoundImpl(Op, ElementSize, Src, Mode);
StoreResult(FPRClass, Op, Result, -1);
}
template
void OpDispatchBuilder::AVXVectorRound<4, false>(OpcodeArgs);
void OpDispatchBuilder::AVXVectorRound<4>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorRound<8, false>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorRound<4, true>(OpcodeArgs);
template
void OpDispatchBuilder::AVXVectorRound<8, true>(OpcodeArgs);
void OpDispatchBuilder::AVXVectorRound<8>(OpcodeArgs);
template<size_t ElementSize>
void OpDispatchBuilder::VectorBlend(OpcodeArgs) {
+153
View File
@@ -156,6 +156,7 @@
"MemOffsetType": "MemOffsetType",
"BreakDefinition": "BreakDefinition",
"RoundType": "RoundType",
"FloatCompareOp": "FloatCompareOp",
"NamedVectorConstant": "FEXCore::IR::NamedVectorConstant",
"IndexNamedVectorConstant": "FEXCore::IR::IndexNamedVectorConstant"
},
@@ -1268,6 +1269,158 @@
"DestSize": "4"
}
},
"VectorScalar": {
"FPR = VFAddScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar 'add' between Vector1 and Vector2.",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFSubScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar 'sub' between Vector1 and Vector2.",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFMulScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar 'mul' between Vector1 and Vector2.",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFDivScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar 'div' between Vector1 and Vector2.",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFMinScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar 'min' between Vector1 and Vector2.",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics.",
"Additionally matches x86 zero and NaN semantics",
"If both source operands are zero, return the second operand (in the case of negative and positive zero)",
"If either source operand is NaN then return the second operand."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFMaxScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar 'max' between Vector1 and Vector2.",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics.",
"Additionally matches x86 zero and NaN semantics",
"If both source operands are zero, return the second operand (in the case of negative and positive zero)",
"If either source operand is NaN then return the second operand."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFSqrtScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar 'sqrt' on Vector2, inserting in to Vector1 and storing in to the destination.",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFRSqrtScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar 'rsqrt' on Vector2, inserting in to Vector1 and storing in to the destination.",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFRecpScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar 'recip' on Vector2, inserting in to Vector1 and storing in to the destination.",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFToFScalarInsert OpSize:#RegisterSize, u8:#DstElementSize, u8:$SrcElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar 'cvt' between Vector1 and Vector2.",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / DstElementSize"
},
"FPR = VSToFVectorInsert OpSize:#RegisterSize, u8:#DstElementSize, u8:$SrcElementSize, FPR:$Vector1, FPR:$Vector2, i8:$HasTwoElements, i1:$ZeroUpperBits": {
"Desc": ["Does a Vector 'scvt' between Vector1 and Vector2.",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics.",
"HasTwoElements is slightly different than most of these scalar operations.",
"Handles the edge case of cvtpi2ps xmm0, mm0 which is two elements in the lower 64-bits"
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / DstElementSize"
},
"FPR = VSToFGPRInsert OpSize:#RegisterSize, u8:#DstElementSize, u8:$SrcElementSize, FPR:$Vector, GPR:$Src, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar 'cvt' between Vector1 and GPR.",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / DstElementSize"
},
"FPR = VFToIScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, RoundType:$Round, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar round float to integral on Vector2, inserting in to Vector1 and storing in to the destination.",
"Rounding mode determined by argument",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFCMPScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FloatCompareOp:$Op, i1:$ZeroUpperBits": {
"Desc": ["Does a scalar 'cmp' between Vector1 and Vecto2, inserting in to Vector1 and storing in to the destination.",
"Compare op determined by argument",
"Inserting the result in to the lower element of Vector1 and returning the results.",
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
"For 128-bit operation this matches SSE insert semantics.",
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
],
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
}
},
"Vector": {
"FPR = VMov u8:#RegisterSize, FPR:$Source": {
"Desc" : ["Copy vector register",
+12
View File
@@ -238,6 +238,18 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
}
}
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::FloatCompareOp Arg) {
switch (Arg) {
case FloatCompareOp::EQ: *out << "FEQ"; break;
case FloatCompareOp::LT: *out << "FLT"; break;
case FloatCompareOp::LE: *out << "FLE"; break;
case FloatCompareOp::UNO: *out << "UNO"; break;
case FloatCompareOp::NEQ: *out << "NEQ"; break;
case FloatCompareOp::ORD: *out << "ORD"; break;
default: *out << "<Unknown OpSize Type>"; break;
}
}
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::BreakDefinition Arg) {
*out << "{" << Arg.ErrorRegister << ".";
*out << static_cast<uint32_t>(Arg.Signal) << ".";
+10
View File
@@ -555,6 +555,15 @@ enum OpSize : uint8_t {
i256Bit = 32,
};
enum class FloatCompareOp : uint8_t {
EQ = 0,
LT,
LE,
UNO,
NEQ,
ORD,
};
// Converts a size stored as an integer in to an OpSize enum.
// This is a nop operation and will be eliminated by the compiler.
static inline OpSize SizeToOpSize(uint8_t Size) {
@@ -568,6 +577,7 @@ static inline OpSize SizeToOpSize(uint8_t Size) {
default: FEX_UNREACHABLE;
}
}
#define IROP_ENUM
#define IROP_STRUCTS
#define IROP_SIZES
+124
View File
@@ -0,0 +1,124 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"AFP"
],
"DisabledHostFeatures": [
"SVE128",
"SVE256"
]
},
"Instructions": {
"roundss xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"Nearest rounding",
"0x66 0x0f 0x3a 0x0a"
],
"ExpectedArm64ASM": [
"frintn s16, s17"
]
},
"roundss xmm0, xmm1, 00000001b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"-inf rounding",
"0x66 0x0f 0x3a 0x0a"
],
"ExpectedArm64ASM": [
"frintm s16, s17"
]
},
"roundss xmm0, xmm1, 00000010b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"+inf rounding",
"0x66 0x0f 0x3a 0x0a"
],
"ExpectedArm64ASM": [
"frintp s16, s17"
]
},
"roundss xmm0, xmm1, 00000011b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"truncate rounding",
"0x66 0x0f 0x3a 0x0a"
],
"ExpectedArm64ASM": [
"frintz s16, s17"
]
},
"roundss xmm0, xmm1, 00000100b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"host rounding mode rounding",
"0x66 0x0f 0x3a 0x0a"
],
"ExpectedArm64ASM": [
"frinti s16, s17"
]
},
"roundsd xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"Nearest rounding",
"0x66 0x0f 0x3a 0x0b"
],
"ExpectedArm64ASM": [
"frintn d16, d17"
]
},
"roundsd xmm0, xmm1, 00000001b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"-inf rounding",
"0x66 0x0f 0x3a 0x0b"
],
"ExpectedArm64ASM": [
"frintm d16, d17"
]
},
"roundsd xmm0, xmm1, 00000010b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"+inf rounding",
"0x66 0x0f 0x3a 0x0b"
],
"ExpectedArm64ASM": [
"frintp d16, d17"
]
},
"roundsd xmm0, xmm1, 00000011b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"truncate rounding",
"0x66 0x0f 0x3a 0x0b"
],
"ExpectedArm64ASM": [
"frintz d16, d17"
]
},
"roundsd xmm0, xmm1, 00000100b": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"host rounding mode rounding",
"0x66 0x0f 0x3a 0x0b"
],
"ExpectedArm64ASM": [
"frinti d16, d17"
]
}
}
}
@@ -0,0 +1,35 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"SVE128",
"SVE256",
"AFP"
],
"DisabledHostFeatures": []
},
"Instructions": {
"cvtpi2ps xmm0, [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr d2, [x4]",
"scvtf v16.2s, v2.2s"
]
},
"cvtpi2ps xmm0, mm0": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #752]",
"scvtf v16.2s, v2.2s"
]
}
}
}
@@ -0,0 +1,251 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"SVE128",
"SVE256",
"AFP"
],
"DisabledHostFeatures": []
},
"Instructions": {
"cvtsi2ss xmm0, eax": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"scvtf s16, w4"
]
},
"cvtsi2ss xmm0, dword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr s2, [x4]",
"scvtf s16, s2"
]
},
"cvtsi2ss xmm0, qword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr x20, [x4]",
"scvtf s16, x20"
]
},
"sqrtss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x51",
"ExpectedArm64ASM": [
"fsqrt s16, s17"
]
},
"rsqrtss xmm0, xmm1": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"FEAT_FPRES could make this more optimal",
"0xf3 0x0f 0x52"
],
"ExpectedArm64ASM": [
"fmov s0, #0x70 (1.0000)",
"fsqrt s1, s17",
"fdiv s16, s0, s1"
]
},
"rcpss xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"FEAT_FPRES could make this more optimal",
"0xf3 0x0f 0x53"
],
"ExpectedArm64ASM": [
"fmov s0, #0x70 (1.0000)",
"fdiv s16, s0, s17"
]
},
"addss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x58"
],
"ExpectedArm64ASM": [
"fadd s16, s16, s17"
]
},
"mulss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x59"
],
"ExpectedArm64ASM": [
"fmul s16, s16, s17"
]
},
"cvtss2sd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x5a",
"ExpectedArm64ASM": [
"fcvt d16, s17"
]
},
"cvtss2sd xmm0, [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x5a",
"ExpectedArm64ASM": [
"ldr d2, [x4]",
"fcvt d16, s2"
]
},
"subss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5c"
],
"ExpectedArm64ASM": [
"fsub s16, s16, s17"
]
},
"minss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5d"
],
"ExpectedArm64ASM": [
"fmin s16, s16, s17"
]
},
"divss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5e"
],
"ExpectedArm64ASM": [
"fdiv s16, s16, s17"
]
},
"maxss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5f"
],
"ExpectedArm64ASM": [
"fmax s16, s16, s17"
]
},
"cmpss xmm0, xmm1, 0": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmeq s16, s16, s17"
]
},
"cmpss xmm0, xmm1, 1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmgt s16, s17, s16"
]
},
"cmpss xmm0, xmm1, 2": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge s16, s17, s16"
]
},
"cmpss xmm0, xmm1, 3": {
"ExpectedInstructionCount": 6,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge s0, s16, s17",
"fcmgt s1, s17, s16",
"orr v0.8b, v0.8b, v1.8b",
"mvn v0.8b, v0.8b",
"ptrue p0.s, vl1",
"mov z16.s, p0/m, z0.s"
]
},
"cmpss xmm0, xmm1, 4": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmeq s0, s16, s17",
"mvn v0.8b, v0.8b",
"ptrue p0.s, vl1",
"mov z16.s, p0/m, z0.s"
]
},
"cmpss xmm0, xmm1, 5": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmgt s2, s17, s16",
"mvn v2.16b, v2.16b",
"mov v16.s[0], v2.s[0]"
]
},
"cmpss xmm0, xmm1, 6": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge s2, s17, s16",
"mvn v2.16b, v2.16b",
"mov v16.s[0], v2.s[0]"
]
},
"cmpss xmm0, xmm1, 7": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge s0, s16, s17",
"fcmgt s1, s17, s16",
"orr v0.8b, v0.8b, v1.8b",
"ptrue p0.s, vl1",
"mov z16.s, p0/m, z0.s"
]
}
}
}
@@ -0,0 +1,242 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"SVE128",
"SVE256",
"AFP"
],
"DisabledHostFeatures": []
},
"Instructions": {
"cvtsi2sd xmm0, eax": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"scvtf d16, w4"
]
},
"cvtsi2sd xmm0, dword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr w20, [x4]",
"scvtf d16, w20"
]
},
"cvtsi2sd xmm0, rax": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"scvtf d16, x4"
]
},
"cvtsi2sd xmm0, qword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr d2, [x4]",
"scvtf d16, d2"
]
},
"sqrtsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x51"
],
"ExpectedArm64ASM": [
"fsqrt d16, d17"
]
},
"addsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x58"
],
"ExpectedArm64ASM": [
"fadd d16, d16, d17"
]
},
"mulsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x59"
],
"ExpectedArm64ASM": [
"fmul d16, d16, d17"
]
},
"cvtsd2ss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5a"
],
"ExpectedArm64ASM": [
"fcvt s16, d17"
]
},
"cvtsd2ss xmm0, [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5a"
],
"ExpectedArm64ASM": [
"ldr q2, [x4]",
"fcvt s16, d2"
]
},
"subsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5c"
],
"ExpectedArm64ASM": [
"fsub d16, d16, d17"
]
},
"minsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5d"
],
"ExpectedArm64ASM": [
"fmin d16, d16, d17"
]
},
"divsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5e"
],
"ExpectedArm64ASM": [
"fdiv d16, d16, d17"
]
},
"maxsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5f"
],
"ExpectedArm64ASM": [
"fmax d16, d16, d17"
]
},
"cmpsd xmm0, xmm1, 0": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmeq d16, d16, d17"
]
},
"cmpsd xmm0, xmm1, 1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmgt d16, d17, d16"
]
},
"cmpsd xmm0, xmm1, 2": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge d16, d17, d16"
]
},
"cmpsd xmm0, xmm1, 3": {
"ExpectedInstructionCount": 6,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge d0, d16, d17",
"fcmgt d1, d17, d16",
"orr v0.8b, v0.8b, v1.8b",
"mvn v0.8b, v0.8b",
"ptrue p0.d, vl1",
"mov z16.d, p0/m, z0.d"
]
},
"cmpsd xmm0, xmm1, 4": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmeq d0, d16, d17",
"mvn v0.8b, v0.8b",
"ptrue p0.d, vl1",
"mov z16.d, p0/m, z0.d"
]
},
"cmpsd xmm0, xmm1, 5": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmgt d2, d17, d16",
"mvn v2.16b, v2.16b",
"mov v16.d[0], v2.d[0]"
]
},
"cmpsd xmm0, xmm1, 6": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge d2, d17, d16",
"mvn v2.16b, v2.16b",
"mov v16.d[0], v2.d[0]"
]
},
"cmpsd xmm0, xmm1, 7": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge d0, d16, d17",
"fcmgt d1, d17, d16",
"orr v0.8b, v0.8b, v1.8b",
"ptrue p0.d, vl1",
"mov z16.d, p0/m, z0.d"
]
}
}
}
@@ -0,0 +1,36 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"AFP"
],
"DisabledHostFeatures": [
"SVE128",
"SVE256"
]
},
"Instructions": {
"cvtpi2ps xmm0, [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr d2, [x4]",
"scvtf v16.2s, v2.2s"
]
},
"cvtpi2ps xmm0, mm0": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #752]",
"scvtf v16.2s, v2.2s"
]
}
}
}
@@ -0,0 +1,249 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"AFP"
],
"DisabledHostFeatures": [
"SVE128",
"SVE256"
]
},
"Instructions": {
"cvtsi2ss xmm0, eax": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"scvtf s16, w4"
]
},
"cvtsi2ss xmm0, dword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr s2, [x4]",
"scvtf s16, s2"
]
},
"cvtsi2ss xmm0, qword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr x20, [x4]",
"scvtf s16, x20"
]
},
"sqrtss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x51",
"ExpectedArm64ASM": [
"fsqrt s16, s17"
]
},
"rsqrtss xmm0, xmm1": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"FEAT_FPRES could make this more optimal",
"0xf3 0x0f 0x52"
],
"ExpectedArm64ASM": [
"fmov s0, #0x70 (1.0000)",
"fsqrt s1, s17",
"fdiv s16, s0, s1"
]
},
"rcpss xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"FEAT_FPRES could make this more optimal",
"0xf3 0x0f 0x53"
],
"ExpectedArm64ASM": [
"fmov s0, #0x70 (1.0000)",
"fdiv s16, s0, s17"
]
},
"addss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x58"
],
"ExpectedArm64ASM": [
"fadd s16, s16, s17"
]
},
"mulss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x59"
],
"ExpectedArm64ASM": [
"fmul s16, s16, s17"
]
},
"cvtss2sd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x5a",
"ExpectedArm64ASM": [
"fcvt d16, s17"
]
},
"cvtss2sd xmm0, [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x5a",
"ExpectedArm64ASM": [
"ldr d2, [x4]",
"fcvt d16, s2"
]
},
"subss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5c"
],
"ExpectedArm64ASM": [
"fsub s16, s16, s17"
]
},
"minss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5d"
],
"ExpectedArm64ASM": [
"fmin s16, s16, s17"
]
},
"divss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5e"
],
"ExpectedArm64ASM": [
"fdiv s16, s16, s17"
]
},
"maxss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x5f"
],
"ExpectedArm64ASM": [
"fmax s16, s16, s17"
]
},
"cmpss xmm0, xmm1, 0": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmeq s16, s16, s17"
]
},
"cmpss xmm0, xmm1, 1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmgt s16, s17, s16"
]
},
"cmpss xmm0, xmm1, 2": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge s16, s17, s16"
]
},
"cmpss xmm0, xmm1, 3": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge s0, s16, s17",
"fcmgt s1, s17, s16",
"orr v0.8b, v0.8b, v1.8b",
"mvn v0.8b, v0.8b",
"mov v16.s[0], v0.s[0]"
]
},
"cmpss xmm0, xmm1, 4": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmeq s0, s16, s17",
"mvn v0.8b, v0.8b",
"mov v16.s[0], v0.s[0]"
]
},
"cmpss xmm0, xmm1, 5": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmgt s2, s17, s16",
"mvn v2.16b, v2.16b",
"mov v16.s[0], v2.s[0]"
]
},
"cmpss xmm0, xmm1, 6": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge s2, s17, s16",
"mvn v2.16b, v2.16b",
"mov v16.s[0], v2.s[0]"
]
},
"cmpss xmm0, xmm1, 7": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge s0, s16, s17",
"fcmgt s1, s17, s16",
"orr v0.8b, v0.8b, v1.8b",
"mov v16.s[0], v0.s[0]"
]
}
}
}
@@ -0,0 +1,240 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"AFP"
],
"DisabledHostFeatures": [
"SVE128",
"SVE256"
]
},
"Instructions": {
"cvtsi2sd xmm0, eax": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"scvtf d16, w4"
]
},
"cvtsi2sd xmm0, dword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr w20, [x4]",
"scvtf d16, w20"
]
},
"cvtsi2sd xmm0, rax": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"scvtf d16, x4"
]
},
"cvtsi2sd xmm0, qword [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr d2, [x4]",
"scvtf d16, d2"
]
},
"sqrtsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x51"
],
"ExpectedArm64ASM": [
"fsqrt d16, d17"
]
},
"addsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x58"
],
"ExpectedArm64ASM": [
"fadd d16, d16, d17"
]
},
"mulsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x59"
],
"ExpectedArm64ASM": [
"fmul d16, d16, d17"
]
},
"cvtsd2ss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5a"
],
"ExpectedArm64ASM": [
"fcvt s16, d17"
]
},
"cvtsd2ss xmm0, [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5a"
],
"ExpectedArm64ASM": [
"ldr q2, [x4]",
"fcvt s16, d2"
]
},
"subsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5c"
],
"ExpectedArm64ASM": [
"fsub d16, d16, d17"
]
},
"minsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5d"
],
"ExpectedArm64ASM": [
"fmin d16, d16, d17"
]
},
"divsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5e"
],
"ExpectedArm64ASM": [
"fdiv d16, d16, d17"
]
},
"maxsd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0x5f"
],
"ExpectedArm64ASM": [
"fmax d16, d16, d17"
]
},
"cmpsd xmm0, xmm1, 0": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmeq d16, d16, d17"
]
},
"cmpsd xmm0, xmm1, 1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmgt d16, d17, d16"
]
},
"cmpsd xmm0, xmm1, 2": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge d16, d17, d16"
]
},
"cmpsd xmm0, xmm1, 3": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge d0, d16, d17",
"fcmgt d1, d17, d16",
"orr v0.8b, v0.8b, v1.8b",
"mvn v0.8b, v0.8b",
"mov v16.d[0], v0.d[0]"
]
},
"cmpsd xmm0, xmm1, 4": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmeq d0, d16, d17",
"mvn v0.8b, v0.8b",
"mov v16.d[0], v0.d[0]"
]
},
"cmpsd xmm0, xmm1, 5": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmgt d2, d17, d16",
"mvn v2.16b, v2.16b",
"mov v16.d[0], v2.d[0]"
]
},
"cmpsd xmm0, xmm1, 6": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge d2, d17, d16",
"mvn v2.16b, v2.16b",
"mov v16.d[0], v2.d[0]"
]
},
"cmpsd xmm0, xmm1, 7": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge d0, d16, d17",
"fcmgt d1, d17, d16",
"orr v0.8b, v0.8b, v1.8b",
"mov v16.d[0], v0.d[0]"
]
}
}
}
@@ -0,0 +1,440 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"SVE128",
"SVE256",
"AFP"
],
"DisabledHostFeatures": []
},
"Instructions": {
"vsqrtss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x51 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fsqrt s16, s18"
]
},
"vsqrtsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x51 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fsqrt d16, d18"
]
},
"vrsqrtss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"FEAT_FPRES could make this more optimal",
"Map 1 0b10 0x52 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fmov s0, #0x70 (1.0000)",
"fsqrt s1, s18",
"fdiv s16, s0, s1"
]
},
"vrcpss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"FEAT_FPRES could make this more optimal",
"Map 1 0b10 0x53 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fmov s0, #0x70 (1.0000)",
"fdiv s16, s0, s18"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x00": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmeq s16, s17, s18"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x01": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmgt s16, s18, s17"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x02": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmge s16, s18, s17"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x03": {
"ExpectedInstructionCount": 6,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmge s0, s17, s18",
"fcmgt s1, s18, s17",
"orr v0.8b, v0.8b, v1.8b",
"mvn v0.8b, v0.8b",
"mov v16.s[0], v0.s[0]"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x04": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmeq s0, s17, s18",
"mvn v0.8b, v0.8b",
"mov v16.s[0], v0.s[0]"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x05": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmgt s2, s18, s17",
"mvn v2.16b, v2.16b",
"mov v16.16b, v17.16b",
"mov v16.s[0], v2.s[0]"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x06": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmge s2, s18, s17",
"mvn v2.16b, v2.16b",
"mov v16.16b, v17.16b",
"mov v16.s[0], v2.s[0]"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x07": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmge s0, s17, s18",
"fcmgt s1, s18, s17",
"orr v0.8b, v0.8b, v1.8b",
"mov v16.s[0], v0.s[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x00": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmeq d16, d17, d18"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x01": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmgt d16, d18, d17"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x02": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmge d16, d18, d17"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x03": {
"ExpectedInstructionCount": 6,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmge d0, d17, d18",
"fcmgt d1, d18, d17",
"orr v0.8b, v0.8b, v1.8b",
"mvn v0.8b, v0.8b",
"mov v16.d[0], v0.d[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x04": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmeq d0, d17, d18",
"mvn v0.8b, v0.8b",
"mov v16.d[0], v0.d[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x05": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmgt d2, d18, d17",
"mvn v2.16b, v2.16b",
"mov v16.16b, v17.16b",
"mov v16.d[0], v2.d[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x06": {
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmge d2, d18, d17",
"mvn v2.16b, v2.16b",
"mov v16.16b, v17.16b",
"mov v16.d[0], v2.d[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x07": {
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmge d0, d17, d18",
"fcmgt d1, d18, d17",
"orr v0.8b, v0.8b, v1.8b",
"mov v16.d[0], v0.d[0]"
]
},
"vcvtsi2ss xmm0, xmm1, eax": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x2A 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"scvtf s16, w4"
]
},
"vcvtsi2ss xmm0, xmm1, rax": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x2A 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"scvtf s16, x4"
]
},
"vcvtsi2sd xmm0, xmm1, eax": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x2A 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"scvtf d16, w4"
]
},
"vcvtsi2sd xmm0, xmm1, rax": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x2A 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"scvtf d16, x4"
]
},
"vmulss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x59 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fmul s16, s17, s18"
]
},
"vmulsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x59 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fmul d16, d17, d18"
]
},
"vcvtss2sd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5a 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcvt d16, s18"
]
},
"vcvtsd2ss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5a 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcvt s16, d18"
]
},
"vsubss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5c 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fsub s16, s17, s18"
]
},
"vsubsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5c 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fsub d16, d17, d18"
]
},
"vminss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5d 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fmin s16, s17, s18"
]
},
"vminsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5d 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fmin d16, d17, d18"
]
},
"vdivss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5e 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fdiv s16, s17, s18"
]
},
"vdivsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5e 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fdiv d16, d17, d18"
]
},
"vmaxss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5f 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fmax s16, s17, s18"
]
},
"vmaxsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5f 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fmax d16, d17, d18"
]
}
}
}
@@ -0,0 +1,133 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"SVE128",
"SVE256",
"AFP"
],
"DisabledHostFeatures": []
},
"Instructions": {
"vroundss xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"nearest rounding",
"Map 3 0b01 0x0a 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v16.16b",
"frintn s16, s16"
]
},
"vroundss xmm0, xmm1, 00000001b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"-inf rounding",
"Map 3 0b01 0x0a 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v16.16b",
"frintm s16, s16"
]
},
"vroundss xmm0, xmm1, 00000010b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"+inf rounding",
"Map 3 0b01 0x0a 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v16.16b",
"frintp s16, s16"
]
},
"vroundss xmm0, xmm1, 00000011b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"truncate rounding",
"Map 3 0b01 0x0a 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v16.16b",
"frintz s16, s16"
]
},
"vroundss xmm0, xmm1, 00000100b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"host mode rounding",
"Map 3 0b01 0x0a 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v16.16b",
"frinti s16, s16"
]
},
"vroundsd xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"nearest rounding",
"Map 3 0b01 0x0b 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v16.16b",
"frintn d16, d16"
]
},
"vroundsd xmm0, xmm1, 00000001b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"-inf rounding",
"Map 3 0b01 0x0b 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v16.16b",
"frintm d16, d16"
]
},
"vroundsd xmm0, xmm1, 00000010b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"+inf rounding",
"Map 3 0b01 0x0b 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v16.16b",
"frintp d16, d16"
]
},
"vroundsd xmm0, xmm1, 00000011b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"truncate rounding",
"Map 3 0b01 0x0b 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v16.16b",
"frintz d16, d16"
]
},
"vroundsd xmm0, xmm1, 00000100b": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"host mode rounding",
"Map 3 0b01 0x0b 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v16.16b",
"frinti d16, d16"
]
}
}
}
+22 -21
View File
@@ -4,7 +4,8 @@
"EnabledHostFeatures": [],
"DisabledHostFeatures": [
"SVE128",
"SVE256"
"SVE256",
"AFP"
]
},
"Comment": [
@@ -167,8 +168,8 @@
"0x66 0x0f 0x3a 0x0a"
],
"ExpectedArm64ASM": [
"frintn s2, s17",
"mov v16.s[0], v2.s[0]"
"frintn s0, s17",
"mov v16.s[0], v0.s[0]"
]
},
"roundss xmm0, xmm1, 00000001b": {
@@ -179,8 +180,8 @@
"0x66 0x0f 0x3a 0x0a"
],
"ExpectedArm64ASM": [
"frintm s2, s17",
"mov v16.s[0], v2.s[0]"
"frintm s0, s17",
"mov v16.s[0], v0.s[0]"
]
},
"roundss xmm0, xmm1, 00000010b": {
@@ -191,8 +192,8 @@
"0x66 0x0f 0x3a 0x0a"
],
"ExpectedArm64ASM": [
"frintp s2, s17",
"mov v16.s[0], v2.s[0]"
"frintp s0, s17",
"mov v16.s[0], v0.s[0]"
]
},
"roundss xmm0, xmm1, 00000011b": {
@@ -203,8 +204,8 @@
"0x66 0x0f 0x3a 0x0a"
],
"ExpectedArm64ASM": [
"frintz s2, s17",
"mov v16.s[0], v2.s[0]"
"frintz s0, s17",
"mov v16.s[0], v0.s[0]"
]
},
"roundss xmm0, xmm1, 00000100b": {
@@ -215,8 +216,8 @@
"0x66 0x0f 0x3a 0x0a"
],
"ExpectedArm64ASM": [
"frinti s2, s17",
"mov v16.s[0], v2.s[0]"
"frinti s0, s17",
"mov v16.s[0], v0.s[0]"
]
},
"roundsd xmm0, xmm1, 00000000b": {
@@ -227,8 +228,8 @@
"0x66 0x0f 0x3a 0x0b"
],
"ExpectedArm64ASM": [
"frintn d2, d17",
"mov v16.d[0], v2.d[0]"
"frintn d0, d17",
"mov v16.d[0], v0.d[0]"
]
},
"roundsd xmm0, xmm1, 00000001b": {
@@ -239,8 +240,8 @@
"0x66 0x0f 0x3a 0x0b"
],
"ExpectedArm64ASM": [
"frintm d2, d17",
"mov v16.d[0], v2.d[0]"
"frintm d0, d17",
"mov v16.d[0], v0.d[0]"
]
},
"roundsd xmm0, xmm1, 00000010b": {
@@ -251,8 +252,8 @@
"0x66 0x0f 0x3a 0x0b"
],
"ExpectedArm64ASM": [
"frintp d2, d17",
"mov v16.d[0], v2.d[0]"
"frintp d0, d17",
"mov v16.d[0], v0.d[0]"
]
},
"roundsd xmm0, xmm1, 00000011b": {
@@ -263,8 +264,8 @@
"0x66 0x0f 0x3a 0x0b"
],
"ExpectedArm64ASM": [
"frintz d2, d17",
"mov v16.d[0], v2.d[0]"
"frintz d0, d17",
"mov v16.d[0], v0.d[0]"
]
},
"roundsd xmm0, xmm1, 00000100b": {
@@ -275,8 +276,8 @@
"0x66 0x0f 0x3a 0x0b"
],
"ExpectedArm64ASM": [
"frinti d2, d17",
"mov v16.d[0], v2.d[0]"
"frinti d0, d17",
"mov v16.d[0], v0.d[0]"
]
},
"blendps xmm0, xmm1, 0000b": {
@@ -6,7 +6,8 @@
],
"DisabledHostFeatures": [
"SVE128",
"SVE256"
"SVE256",
"AFP"
]
},
"Instructions": {
@@ -18,8 +19,8 @@
"0xf3 0x0f 0x52"
],
"ExpectedArm64ASM": [
"frsqrte s2, s17",
"mov v16.s[0], v2.s[0]"
"frecpe s0, s17",
"mov v16.s[0], v0.s[0]"
]
},
"rcpss xmm0, xmm1": {
@@ -30,8 +31,8 @@
"0xf3 0x0f 0x53"
],
"ExpectedArm64ASM": [
"frecpe s2, s17",
"mov v16.s[0], v2.s[0]"
"frsqrte s0, s17",
"mov v16.s[0], v0.s[0]"
]
}
}
@@ -0,0 +1,37 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"RPRES",
"AFP"
],
"DisabledHostFeatures": [
"SVE128",
"SVE256"
]
},
"Instructions": {
"rsqrtss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"AFP can make this more optimal",
"0xf3 0x0f 0x52"
],
"ExpectedArm64ASM": [
"frecpe s16, s17"
]
},
"rcpss xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"AFP can make this more optimal",
"0xf3 0x0f 0x53"
],
"ExpectedArm64ASM": [
"frsqrte s16, s17"
]
}
}
}
@@ -6,7 +6,9 @@
"SVE256",
"RPRES"
],
"DisabledHostFeatures": []
"DisabledHostFeatures": [
"AFP"
]
},
"Instructions": {
"vrsqrtps xmm0, xmm1": {
@@ -31,18 +33,16 @@
]
},
"vrsqrtss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"AFP can make this more optimal",
"Map 1 0b10 0x52 128-bit"
],
"ExpectedArm64ASM": [
"frsqrte s2, s18",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"frecpe s0, s18",
"mov v16.s[0], v0.s[0]"
]
},
"vrcpps xmm0, xmm1": {
@@ -67,17 +67,15 @@
]
},
"vrcpss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"Map 1 0b10 0x53 128-bit"
],
"ExpectedArm64ASM": [
"frecpe s2, s18",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"frsqrte s0, s18",
"mov v16.s[0], v0.s[0]"
]
}
}
@@ -0,0 +1,79 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"SVE128",
"SVE256",
"RPRES",
"AFP"
],
"DisabledHostFeatures": []
},
"Instructions": {
"vrsqrtps xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Comment": [
"Map 1 0b00 0x52 128-bit"
],
"ExpectedArm64ASM": [
"frsqrte v2.4s, v17.4s",
"mov v16.16b, v2.16b"
]
},
"vrsqrtps ymm0, ymm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"Map 1 0b00 0x52 256-bit"
],
"ExpectedArm64ASM": [
"frsqrte z16.s, z17.s"
]
},
"vrsqrtss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Comment": [
"AFP can make this more optimal",
"Map 1 0b10 0x52 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"frecpe s16, s18"
]
},
"vrcpps xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Comment": [
"Map 1 0b00 0x53 128-bit"
],
"ExpectedArm64ASM": [
"frecpe v2.4s, v17.4s",
"mov v16.16b, v2.16b"
]
},
"vrcpps ymm0, ymm1": {
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"Map 1 0b00 0x53 256-bit"
],
"ExpectedArm64ASM": [
"frecpe z16.s, z17.s"
]
},
"vrcpss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Comment": [
"Map 1 0b10 0x53 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"frsqrte s16, s18"
]
}
}
}
+11
View File
@@ -0,0 +1,11 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [],
"DisabledHostFeatures": [
"SVE128",
"SVE256"
]
},
"Instructions": {}
}
+6 -7
View File
@@ -5,7 +5,8 @@
"DisabledHostFeatures": [
"SVE128",
"SVE256",
"RPRES"
"RPRES",
"AFP"
]
},
"Comment": [
@@ -130,26 +131,24 @@
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr d2, [x4]",
"scvtf v2.4s, v2.4s",
"mov v16.d[0], v2.d[0]"
"scvtf v0.2s, v2.2s",
"mov v16.d[0], v0.d[0]"
]
},
"cvtpi2ps xmm0, mm0": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #752]",
"scvtf v2.4s, v2.4s",
"mov v16.d[0], v2.d[0]"
"scvtf v0.2s, v2.2s",
"mov v16.d[0], v0.d[0]"
]
},
"movntps [rax], xmm0": {
@@ -6,7 +6,8 @@
],
"DisabledHostFeatures": [
"SVE128",
"SVE256"
"SVE256",
"AFP"
]
},
"Instructions": {
+60 -76
View File
@@ -5,7 +5,8 @@
"DisabledHostFeatures": [
"SVE128",
"SVE256",
"RPRES"
"RPRES",
"AFP"
]
},
"Instructions": {
@@ -71,25 +72,23 @@
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"scvtf s2, w4",
"mov v16.s[0], v2.s[0]"
"scvtf s0, w4",
"mov v16.s[0], v0.s[0]"
]
},
"cvtsi2ss xmm0, dword [rax]": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr s2, [x4]",
"scvtf s2, s2",
"mov v16.s[0], v2.s[0]"
"scvtf s0, s2",
"mov v16.s[0], v0.s[0]"
]
},
"cvtsi2ss xmm0, rax": {
@@ -97,21 +96,20 @@
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x2a",
"ExpectedArm64ASM": [
"scvtf s2, x4",
"mov v16.s[0], v2.s[0]"
"scvtf s0, x4",
"mov v16.s[0], v0.s[0]"
]
},
"cvtsi2ss xmm0, qword [rax]": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr x20, [x4]",
"scvtf s2, x20",
"mov v16.s[0], v2.s[0]"
"scvtf s0, x20",
"mov v16.s[0], v0.s[0]"
]
},
"movntss [rax], xmm0": {
@@ -196,60 +194,58 @@
},
"sqrtss xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x51",
"ExpectedArm64ASM": [
"fsqrt s2, s17",
"mov v16.s[0], v2.s[0]"
"fsqrt s0, s17",
"mov v16.s[0], v0.s[0]"
]
},
"rsqrtss xmm0, xmm1": {
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x52"
],
"ExpectedArm64ASM": [
"fmov s0, #0x70 (1.0000)",
"fsqrt s1, s17",
"fdiv s2, s0, s1",
"mov v16.s[0], v2.s[0]"
"fdiv s0, s0, s1",
"mov v16.s[0], v0.s[0]"
]
},
"rcpss xmm0, xmm1": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"0xf3 0x0f 0x53"
],
"ExpectedArm64ASM": [
"fmov s0, #0x70 (1.0000)",
"fdiv s2, s0, s17",
"mov v16.s[0], v2.s[0]"
"fdiv s0, s0, s17",
"mov v16.s[0], v0.s[0]"
]
},
"addss xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0x58"
],
"ExpectedArm64ASM": [
"fadd s2, s16, s17",
"mov v16.s[0], v2.s[0]"
"fadd s0, s16, s17",
"mov v16.s[0], v0.s[0]"
]
},
"mulss xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0x59"
],
"ExpectedArm64ASM": [
"fmul s2, s16, s17",
"mov v16.s[0], v2.s[0]"
"fmul s0, s16, s17",
"mov v16.s[0], v0.s[0]"
]
},
"cvtss2sd xmm0, xmm1": {
@@ -257,8 +253,8 @@
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x5a",
"ExpectedArm64ASM": [
"fcvt d2, s17",
"mov v16.d[0], v2.d[0]"
"fcvt d0, s17",
"mov v16.d[0], v0.d[0]"
]
},
"cvtss2sd xmm0, [rax]": {
@@ -266,9 +262,9 @@
"Optimal": "Yes",
"Comment": "0xf3 0x0f 0x5a",
"ExpectedArm64ASM": [
"ldr s2, [x4]",
"fcvt d2, s2",
"mov v16.d[0], v2.d[0]"
"ldr d2, [x4]",
"fcvt d0, s2",
"mov v16.d[0], v0.d[0]"
]
},
"cvttps2dq xmm0, xmm1": {
@@ -283,50 +279,46 @@
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0x5c"
],
"ExpectedArm64ASM": [
"fsub s2, s16, s17",
"mov v16.s[0], v2.s[0]"
"fsub s0, s16, s17",
"mov v16.s[0], v0.s[0]"
]
},
"minss xmm0, xmm1": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0x5d"
],
"ExpectedArm64ASM": [
"fcmp s16, s17",
"fcsel s2, s16, s17, mi",
"mov v16.s[0], v2.s[0]"
"fcsel s0, s16, s17, mi",
"mov v16.s[0], v0.s[0]"
]
},
"divss xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0x5e"
],
"ExpectedArm64ASM": [
"fdiv s2, s16, s17",
"mov v16.s[0], v2.s[0]"
"fdiv s0, s16, s17",
"mov v16.s[0], v0.s[0]"
]
},
"maxss xmm0, xmm1": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0x5f"
],
"ExpectedArm64ASM": [
"fcmp s16, s17",
"fcsel s2, s17, s16, mi",
"mov v16.s[0], v2.s[0]"
"fcsel s0, s17, s16, mi",
"mov v16.s[0], v0.s[0]"
]
},
"movdqu xmm0, xmm1": {
@@ -588,73 +580,67 @@
},
"cmpss xmm0, xmm1, 0": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmeq s2, s16, s17",
"mov v16.s[0], v2.s[0]"
"fcmeq s0, s16, s17",
"mov v16.s[0], v0.s[0]"
]
},
"cmpss xmm0, xmm1, 1": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmgt s2, s17, s16",
"mov v16.s[0], v2.s[0]"
"fcmgt s0, s17, s16",
"mov v16.s[0], v0.s[0]"
]
},
"cmpss xmm0, xmm1, 2": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge s2, s17, s16",
"mov v16.s[0], v2.s[0]"
"fcmge s0, s17, s16",
"mov v16.s[0], v0.s[0]"
]
},
"cmpss xmm0, xmm1, 3": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge s0, s16, s17",
"fcmgt s1, s17, s16",
"orr v2.8b, v0.8b, v1.8b",
"mvn v2.8b, v2.8b",
"mov v16.s[0], v2.s[0]"
"orr v0.8b, v0.8b, v1.8b",
"mvn v0.8b, v0.8b",
"mov v16.s[0], v0.s[0]"
]
},
"cmpss xmm0, xmm1, 4": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmeq s2, s16, s17",
"mvn v2.8b, v2.8b",
"mov v16.s[0], v2.s[0]"
"fcmeq s0, s16, s17",
"mvn v0.8b, v0.8b",
"mov v16.s[0], v0.s[0]"
]
},
"cmpss xmm0, xmm1, 5": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
@@ -665,9 +651,8 @@
},
"cmpss xmm0, xmm1, 6": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
@@ -678,16 +663,15 @@
},
"cmpss xmm0, xmm1, 7": {
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf3 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge s0, s16, s17",
"fcmgt s1, s17, s16",
"orr v2.8b, v0.8b, v1.8b",
"mov v16.s[0], v2.s[0]"
"orr v0.8b, v0.8b, v1.8b",
"mov v16.s[0], v0.s[0]"
]
},
"movq2dq xmm0, mm0": {
@@ -5,7 +5,8 @@
"DisabledHostFeatures": [
"SVE128",
"SVE256",
"FCMA"
"FCMA",
"AFP"
]
},
"Instructions": {
@@ -54,50 +55,46 @@
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"scvtf d2, w4",
"mov v16.d[0], v2.d[0]"
"scvtf d0, w4",
"mov v16.d[0], v0.d[0]"
]
},
"cvtsi2sd xmm0, dword [rax]": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr w20, [x4]",
"scvtf d2, w20",
"mov v16.d[0], v2.d[0]"
"scvtf d0, w20",
"mov v16.d[0], v0.d[0]"
]
},
"cvtsi2sd xmm0, rax": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"scvtf d2, x4",
"mov v16.d[0], v2.d[0]"
"scvtf d0, x4",
"mov v16.d[0], v0.d[0]"
]
},
"cvtsi2sd xmm0, qword [rax]": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x2a"
],
"ExpectedArm64ASM": [
"ldr d2, [x4]",
"scvtf d2, d2",
"mov v16.d[0], v2.d[0]"
"scvtf d0, d2",
"mov v16.d[0], v0.d[0]"
]
},
"movntsd [rax], xmm0": {
@@ -182,113 +179,104 @@
},
"sqrtsd xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x51"
],
"ExpectedArm64ASM": [
"fsqrt d2, d17",
"mov v16.d[0], v2.d[0]"
"fsqrt d0, d17",
"mov v16.d[0], v0.d[0]"
]
},
"addsd xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x58"
],
"ExpectedArm64ASM": [
"fadd d2, d16, d17",
"mov v16.d[0], v2.d[0]"
"fadd d0, d16, d17",
"mov v16.d[0], v0.d[0]"
]
},
"mulsd xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x59"
],
"ExpectedArm64ASM": [
"fmul d2, d16, d17",
"mov v16.d[0], v2.d[0]"
"fmul d0, d16, d17",
"mov v16.d[0], v0.d[0]"
]
},
"cvtsd2ss xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x5a"
],
"ExpectedArm64ASM": [
"fcvt s2, d17",
"mov v16.s[0], v2.s[0]"
"fcvt s0, d17",
"mov v16.s[0], v0.s[0]"
]
},
"cvtsd2ss xmm0, [rax]": {
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x5a"
],
"ExpectedArm64ASM": [
"ldr d2, [x4]",
"fcvt s2, d2",
"mov v16.s[0], v2.s[0]"
"ldr q2, [x4]",
"fcvt s0, d2",
"mov v16.s[0], v0.s[0]"
]
},
"subsd xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x5c"
],
"ExpectedArm64ASM": [
"fsub d2, d16, d17",
"mov v16.d[0], v2.d[0]"
"fsub d0, d16, d17",
"mov v16.d[0], v0.d[0]"
]
},
"minsd xmm0, xmm1": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x5d"
],
"ExpectedArm64ASM": [
"fcmp d16, d17",
"fcsel d2, d16, d17, mi",
"mov v16.d[0], v2.d[0]"
"fcsel d0, d16, d17, mi",
"mov v16.d[0], v0.d[0]"
]
},
"divsd xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x5e"
],
"ExpectedArm64ASM": [
"fdiv d2, d16, d17",
"mov v16.d[0], v2.d[0]"
"fdiv d0, d16, d17",
"mov v16.d[0], v0.d[0]"
]
},
"maxsd xmm0, xmm1": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0x5f"
],
"ExpectedArm64ASM": [
"fcmp d16, d17",
"fcsel d2, d17, d16, mi",
"mov v16.d[0], v2.d[0]"
"fcsel d0, d17, d16, mi",
"mov v16.d[0], v0.d[0]"
]
},
"pshuflw xmm0, xmm1, 0": {
@@ -408,73 +396,67 @@
},
"cmpsd xmm0, xmm1, 0": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmeq d2, d16, d17",
"mov v16.d[0], v2.d[0]"
"fcmeq d0, d16, d17",
"mov v16.d[0], v0.d[0]"
]
},
"cmpsd xmm0, xmm1, 1": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmgt d2, d17, d16",
"mov v16.d[0], v2.d[0]"
"fcmgt d0, d17, d16",
"mov v16.d[0], v0.d[0]"
]
},
"cmpsd xmm0, xmm1, 2": {
"ExpectedInstructionCount": 2,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge d2, d17, d16",
"mov v16.d[0], v2.d[0]"
"fcmge d0, d17, d16",
"mov v16.d[0], v0.d[0]"
]
},
"cmpsd xmm0, xmm1, 3": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge d0, d16, d17",
"fcmgt d1, d17, d16",
"orr v2.8b, v0.8b, v1.8b",
"mvn v2.8b, v2.8b",
"mov v16.d[0], v2.d[0]"
"orr v0.8b, v0.8b, v1.8b",
"mvn v0.8b, v0.8b",
"mov v16.d[0], v0.d[0]"
]
},
"cmpsd xmm0, xmm1, 4": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmeq d2, d16, d17",
"mvn v2.8b, v2.8b",
"mov v16.d[0], v2.d[0]"
"fcmeq d0, d16, d17",
"mvn v0.8b, v0.8b",
"mov v16.d[0], v0.d[0]"
]
},
"cmpsd xmm0, xmm1, 5": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
@@ -485,9 +467,8 @@
},
"cmpsd xmm0, xmm1, 6": {
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
@@ -498,16 +479,15 @@
},
"cmpsd xmm0, xmm1, 7": {
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"With AFP mode FEX can remove an insert after the operation.",
"0xf2 0x0f 0xc2"
],
"ExpectedArm64ASM": [
"fcmge d0, d16, d17",
"fcmgt d1, d17, d16",
"orr v2.8b, v0.8b, v1.8b",
"mov v16.d[0], v2.d[0]"
"orr v0.8b, v0.8b, v1.8b",
"mov v16.d[0], v0.d[0]"
]
},
"addsubps xmm0, xmm1": {
+27
View File
@@ -0,0 +1,27 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [],
"DisabledHostFeatures": [
"SVE128",
"SVE256"
]
},
"Instructions": {
"push ax, bx": {
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Comment": "0x50",
"x86Insts": [
"push ax",
"push bx"
],
"ExpectedArm64ASM": [
"uxth w20, w4",
"strh w20, [x8, #-2]!",
"uxth w20, w7",
"strh w20, [x8, #-2]!"
]
}
}
}
+192 -269
View File
@@ -7,7 +7,8 @@
],
"DisabledHostFeatures": [
"FCMA",
"RPRES"
"RPRES",
"AFP"
]
},
"Instructions": {
@@ -672,33 +673,27 @@
]
},
"vsqrtss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Insert in to first element could be more optimal, which is the common case.",
"Map 1 0b10 0x51 128-bit"
],
"ExpectedArm64ASM": [
"fsqrt s2, s18",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fsqrt s0, s18",
"mov v16.s[0], v0.s[0]"
]
},
"vsqrtsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Insert in to first element could be more optimal, which is the common case.",
"Map 1 0b11 0x51 128-bit"
],
"ExpectedArm64ASM": [
"fsqrt d2, d18",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fsqrt d0, d18",
"mov v16.d[0], v0.d[0]"
]
},
"vrsqrtps xmm0, xmm1": {
@@ -727,19 +722,17 @@
]
},
"vrsqrtss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x52 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fmov s0, #0x70 (1.0000)",
"fsqrt s1, s18",
"fdiv s2, s0, s1",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"fdiv s0, s0, s1",
"mov v16.s[0], v0.s[0]"
]
},
"vrcpps xmm0, xmm1": {
@@ -767,18 +760,16 @@
]
},
"vrcpss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x53 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fmov s0, #0x70 (1.0000)",
"fdiv s2, s0, s18",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"fdiv s0, s0, s18",
"mov v16.s[0], v0.s[0]"
]
},
"vandps xmm0, xmm1": {
@@ -2397,243 +2388,211 @@
]
},
"vcmpss xmm0, xmm1, xmm2, 0x00": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmeq s2, s17, s18",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fcmeq s0, s17, s18",
"mov v16.s[0], v0.s[0]"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x01": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmgt s2, s18, s17",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fcmgt s0, s18, s17",
"mov v16.s[0], v0.s[0]"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x02": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmge s2, s18, s17",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fcmge s0, s18, s17",
"mov v16.s[0], v0.s[0]"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x03": {
"ExpectedInstructionCount": 8,
"Optimal": "No",
"ExpectedInstructionCount": 6,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmge s0, s17, s18",
"fcmgt s1, s18, s17",
"orr v2.8b, v0.8b, v1.8b",
"mvn v2.8b, v2.8b",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"orr v0.8b, v0.8b, v1.8b",
"mvn v0.8b, v0.8b",
"mov v16.s[0], v0.s[0]"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x04": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmeq s2, s17, s18",
"mvn v2.8b, v2.8b",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fcmeq s0, s17, s18",
"mvn v0.8b, v0.8b",
"mov v16.s[0], v0.s[0]"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x05": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmgt s2, s18, s17",
"mvn v2.16b, v2.16b",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"mov v16.s[0], v2.s[0]"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x06": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmge s2, s18, s17",
"mvn v2.16b, v2.16b",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"mov v16.s[0], v2.s[0]"
]
},
"vcmpss xmm0, xmm1, xmm2, 0x07": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmge s0, s17, s18",
"fcmgt s1, s18, s17",
"orr v2.8b, v0.8b, v1.8b",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"orr v0.8b, v0.8b, v1.8b",
"mov v16.s[0], v0.s[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x00": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmeq d2, d17, d18",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fcmeq d0, d17, d18",
"mov v16.d[0], v0.d[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x01": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmgt d2, d18, d17",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fcmgt d0, d18, d17",
"mov v16.d[0], v0.d[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x02": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmge d2, d18, d17",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fcmge d0, d18, d17",
"mov v16.d[0], v0.d[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x03": {
"ExpectedInstructionCount": 8,
"Optimal": "No",
"ExpectedInstructionCount": 6,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmge d0, d17, d18",
"fcmgt d1, d18, d17",
"orr v2.8b, v0.8b, v1.8b",
"mvn v2.8b, v2.8b",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"orr v0.8b, v0.8b, v1.8b",
"mvn v0.8b, v0.8b",
"mov v16.d[0], v0.d[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x04": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmeq d2, d17, d18",
"mvn v2.8b, v2.8b",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fcmeq d0, d17, d18",
"mvn v0.8b, v0.8b",
"mov v16.d[0], v0.d[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x05": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmgt d2, d18, d17",
"mvn v2.16b, v2.16b",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"mov v16.d[0], v2.d[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x06": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"fcmge d2, d18, d17",
"mvn v2.16b, v2.16b",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"mov v16.d[0], v2.d[0]"
]
},
"vcmpsd xmm0, xmm1, xmm2, 0x07": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"ExpectedInstructionCount": 5,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0xC2 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmge d0, d17, d18",
"fcmgt d1, d18, d17",
"orr v2.8b, v0.8b, v1.8b",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"orr v0.8b, v0.8b, v1.8b",
"mov v16.d[0], v0.d[0]"
]
},
"vpinsrw xmm0, xmm1, eax, 000b": {
@@ -3153,59 +3112,51 @@
]
},
"vcvtsi2ss xmm0, xmm1, eax": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x2A 128-bit"
],
"ExpectedArm64ASM": [
"scvtf s2, w4",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"scvtf s0, w4",
"mov v16.s[0], v0.s[0]"
]
},
"vcvtsi2ss xmm0, xmm1, rax": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x2A 128-bit"
],
"ExpectedArm64ASM": [
"scvtf s2, x4",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"scvtf s0, x4",
"mov v16.s[0], v0.s[0]"
]
},
"vcvtsi2sd xmm0, xmm1, eax": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x2A 128-bit"
],
"ExpectedArm64ASM": [
"scvtf d2, w4",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"scvtf d0, w4",
"mov v16.d[0], v0.d[0]"
]
},
"vcvtsi2sd xmm0, xmm1, rax": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x2A 128-bit"
],
"ExpectedArm64ASM": [
"scvtf d2, x4",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"scvtf d0, x4",
"mov v16.d[0], v0.d[0]"
]
},
"vmovntps [rax], xmm0": {
@@ -3496,31 +3447,27 @@
]
},
"vaddss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x58 128-bit"
],
"ExpectedArm64ASM": [
"fadd s2, s17, s18",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fadd s0, s17, s18",
"mov v16.s[0], v0.s[0]"
]
},
"vaddsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x58 128-bit"
],
"ExpectedArm64ASM": [
"fadd d2, d17, d18",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fadd d0, d17, d18",
"mov v16.d[0], v0.d[0]"
]
},
"vmulps xmm0, xmm1, xmm2": {
@@ -3566,31 +3513,27 @@
]
},
"vmulss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x59 128-bit"
],
"ExpectedArm64ASM": [
"fmul s2, s17, s18",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fmul s0, s17, s18",
"mov v16.s[0], v0.s[0]"
]
},
"vmulsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x59 128-bit"
],
"ExpectedArm64ASM": [
"fmul d2, d17, d18",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fmul d0, d17, d18",
"mov v16.d[0], v0.d[0]"
]
},
"vcvtps2pd xmm0, xmm1": {
@@ -3616,31 +3559,27 @@
]
},
"vcvtss2sd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5a 128-bit"
],
"ExpectedArm64ASM": [
"fcvt d2, s18",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fcvt d0, s18",
"mov v16.d[0], v0.d[0]"
]
},
"vcvtsd2ss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5a 128-bit"
],
"ExpectedArm64ASM": [
"fcvt s2, d18",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fcvt s0, d18",
"mov v16.s[0], v0.s[0]"
]
},
"vcvtdq2ps xmm0, xmm1": {
@@ -3751,31 +3690,27 @@
]
},
"vsubss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5c 128-bit"
],
"ExpectedArm64ASM": [
"fsub s2, s17, s18",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fsub s0, s17, s18",
"mov v16.s[0], v0.s[0]"
]
},
"vsubsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5c 128-bit"
],
"ExpectedArm64ASM": [
"fsub d2, d17, d18",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fsub d0, d17, d18",
"mov v16.d[0], v0.d[0]"
]
},
"vminps xmm0, xmm1, xmm2": {
@@ -3833,33 +3768,29 @@
]
},
"vminss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5d 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmp s17, s18",
"fcsel s2, s17, s18, mi",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"fcsel s0, s17, s18, mi",
"mov v16.s[0], v0.s[0]"
]
},
"vminsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5d 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmp d17, d18",
"fcsel d2, d17, d18, mi",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"fcsel d0, d17, d18, mi",
"mov v16.d[0], v0.d[0]"
]
},
"vdivps xmm0, xmm1, xmm2": {
@@ -3909,31 +3840,27 @@
]
},
"vdivss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5e 128-bit"
],
"ExpectedArm64ASM": [
"fdiv s2, s17, s18",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fdiv s0, s17, s18",
"mov v16.s[0], v0.s[0]"
]
},
"vdivsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"ExpectedInstructionCount": 3,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5e 128-bit"
],
"ExpectedArm64ASM": [
"fdiv d2, d17, d18",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v17.16b",
"fdiv d0, d17, d18",
"mov v16.d[0], v0.d[0]"
]
},
"vmaxps xmm0, xmm1, xmm2": {
@@ -3989,33 +3916,29 @@
]
},
"vmaxss xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b10 0x5f 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmp s17, s18",
"fcsel s2, s18, s17, mi",
"mov v0.16b, v17.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"fcsel s0, s18, s17, mi",
"mov v16.s[0], v0.s[0]"
]
},
"vmaxsd xmm0, xmm1, xmm2": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"ExpectedInstructionCount": 4,
"Optimal": "Yes",
"Comment": [
"Map 1 0b11 0x5f 128-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"fcmp d17, d18",
"fcsel d2, d18, d17, mi",
"mov v0.16b, v17.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"fcsel d0, d18, d17, mi",
"mov v16.d[0], v0.d[0]"
]
},
"vpunpckhbw xmm0, xmm1, xmm2": {
+3 -1
View File
@@ -4,7 +4,9 @@
"EnabledHostFeatures": [
"SVE256"
],
"DisabledHostFeatures": []
"DisabledHostFeatures": [
"AFP"
]
},
"Instructions": {
"vpshufb xmm0, xmm1, xmm2": {
+43 -61
View File
@@ -4,7 +4,9 @@
"EnabledHostFeatures": [
"SVE256"
],
"DisabledHostFeatures": []
"DisabledHostFeatures": [
"AFP"
]
},
"Instructions": {
"vpermq ymm0, ymm1, 00000000b": {
@@ -1745,153 +1747,133 @@
]
},
"vroundss xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"nearest rounding",
"Map 3 0b01 0x0a 128-bit"
],
"ExpectedArm64ASM": [
"frintn s2, s17",
"mov v0.16b, v16.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v16.16b",
"frintn s0, s16",
"mov v16.s[0], v0.s[0]"
]
},
"vroundss xmm0, xmm1, 00000001b": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"-inf rounding",
"Map 3 0b01 0x0a 128-bit"
],
"ExpectedArm64ASM": [
"frintm s2, s17",
"mov v0.16b, v16.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v16.16b",
"frintm s0, s16",
"mov v16.s[0], v0.s[0]"
]
},
"vroundss xmm0, xmm1, 00000010b": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"+inf rounding",
"Map 3 0b01 0x0a 128-bit"
],
"ExpectedArm64ASM": [
"frintp s2, s17",
"mov v0.16b, v16.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v16.16b",
"frintp s0, s16",
"mov v16.s[0], v0.s[0]"
]
},
"vroundss xmm0, xmm1, 00000011b": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"truncate rounding",
"Map 3 0b01 0x0a 128-bit"
],
"ExpectedArm64ASM": [
"frintz s2, s17",
"mov v0.16b, v16.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v16.16b",
"frintz s0, s16",
"mov v16.s[0], v0.s[0]"
]
},
"vroundss xmm0, xmm1, 00000100b": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"host mode rounding",
"Map 3 0b01 0x0a 128-bit"
],
"ExpectedArm64ASM": [
"frinti s2, s17",
"mov v0.16b, v16.16b",
"mov v0.s[0], v2.s[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v16.16b",
"frinti s0, s16",
"mov v16.s[0], v0.s[0]"
]
},
"vroundsd xmm0, xmm1, 00000000b": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"nearest rounding",
"Map 3 0b01 0x0b 128-bit"
],
"ExpectedArm64ASM": [
"frintn d2, d17",
"mov v0.16b, v16.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v16.16b",
"frintn d0, d16",
"mov v16.d[0], v0.d[0]"
]
},
"vroundsd xmm0, xmm1, 00000001b": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"-inf rounding",
"Map 3 0b01 0x0b 128-bit"
],
"ExpectedArm64ASM": [
"frintm d2, d17",
"mov v0.16b, v16.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v16.16b",
"frintm d0, d16",
"mov v16.d[0], v0.d[0]"
]
},
"vroundsd xmm0, xmm1, 00000010b": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"+inf rounding",
"Map 3 0b01 0x0b 128-bit"
],
"ExpectedArm64ASM": [
"frintp d2, d17",
"mov v0.16b, v16.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v16.16b",
"frintp d0, d16",
"mov v16.d[0], v0.d[0]"
]
},
"vroundsd xmm0, xmm1, 00000011b": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"truncate rounding",
"Map 3 0b01 0x0b 128-bit"
],
"ExpectedArm64ASM": [
"frintz d2, d17",
"mov v0.16b, v16.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v16.16b",
"frintz d0, d16",
"mov v16.d[0], v0.d[0]"
]
},
"vroundsd xmm0, xmm1, 00000100b": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 3,
"Optimal": "No",
"Comment": [
"host mode rounding",
"Map 3 0b01 0x0b 128-bit"
],
"ExpectedArm64ASM": [
"frinti d2, d17",
"mov v0.16b, v16.16b",
"mov v0.d[0], v2.d[0]",
"mov v2.16b, v0.16b",
"mov v16.16b, v2.16b"
"mov v16.16b, v16.16b",
"frinti d0, d16",
"mov v16.d[0], v0.d[0]"
]
},
"vblendps xmm0, xmm1, xmm2, 0000b": {
+2 -1
View File
@@ -5,7 +5,8 @@
"DisabledHostFeatures": [
"SVE128",
"SVE256",
"CSSC"
"CSSC",
"AFP"
]
},
"Instructions": {
+2 -1
View File
@@ -7,7 +7,8 @@
"EnabledHostFeatures": [],
"DisabledHostFeatures": [
"SVE128",
"SVE256"
"SVE256",
"AFP"
]
},
"Instructions": {