mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 14:00:16 +02:00
Merge pull request #3186 from Sonicadvance1/afp_support
IR: Adds scalar vector insert operations
This commit is contained in:
37 files changed
+3915
-874
No files matched your search
@@ -581,6 +581,24 @@ void Arm64Emitter::PopCalleeSavedRegisters() {
|
||||
}
|
||||
|
||||
void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FPRs, uint32_t GPRSpillMask, uint32_t FPRSpillMask) {
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Disable AFP features when spilling registers.
|
||||
//
|
||||
// Disable FPCR.NEP and FPCR.AH
|
||||
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
|
||||
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
|
||||
//
|
||||
// Additional interesting AFP bits:
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
|
||||
bic(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
@@ -645,6 +663,26 @@ void Arm64Emitter::SpillStaticRegs(FEXCore::ARMEmitter::Register TmpReg, bool FP
|
||||
}
|
||||
|
||||
void Arm64Emitter::FillStaticRegs(bool FPRs, uint32_t GPRFillMask, uint32_t FPRFillMask) {
|
||||
#ifndef VIXL_SIMULATOR
|
||||
if (EmitterCTX->HostFeatures.SupportsAFP) {
|
||||
// Enable AFP features when filling JIT state.
|
||||
LOGMAN_THROW_A_FMT(GPRFillMask != 0, "Must fill at least 1 GPR for a temp");
|
||||
auto TmpReg = StaticRegisters[FindFirstSetBit(GPRFillMask)];
|
||||
mrs(TmpReg, ARMEmitter::SystemRegister::FPCR);
|
||||
|
||||
// Enable FPCR.NEP and FPCR.AH
|
||||
// NEP(2): Changes ASIMD scalar instructions to insert in to the lower bits of the destination.
|
||||
// AH(1): Changes NaN behaviour in some instructions. Specifically fmin, fmax.
|
||||
//
|
||||
// Additional interesting AFP bits:
|
||||
// FIZ(0): Flush Inputs to Zero
|
||||
orr(ARMEmitter::Size::i64Bit, TmpReg, TmpReg,
|
||||
(1U << 2) | // NEP
|
||||
(1U << 1)); // AH
|
||||
msr(ARMEmitter::SystemRegister::FPCR, TmpReg);
|
||||
}
|
||||
#endif
|
||||
|
||||
if (!StaticRegisterAllocation()) {
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -53,7 +53,7 @@ uint64_t Dispatcher::GetCompileBlockPtr() {
|
||||
return CompileBlockPtr.Data;
|
||||
}
|
||||
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
|
||||
constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
|
||||
|
||||
Dispatcher::Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config)
|
||||
: Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE)
|
||||
|
||||
@@ -243,6 +243,9 @@ HostFeatures::HostFeatures() {
|
||||
SupportsBMI2 = true;
|
||||
SupportsCLWB = true;
|
||||
|
||||
// TODO: AFP is disabled until the scalar usage in the codebase can be audited to be working as expected.
|
||||
SupportsAFP = false;
|
||||
|
||||
if (!SupportsAtomics) {
|
||||
WARN_ONCE_FMT("Host CPU doesn't support atomics. Expect bad performance");
|
||||
}
|
||||
|
||||
@@ -534,6 +534,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::In
|
||||
, HostSupportsSVE128{ctx->HostFeatures.SupportsSVE}
|
||||
, HostSupportsSVE256{ctx->HostFeatures.SupportsAVX}
|
||||
, HostSupportsRPRES{ctx->HostFeatures.SupportsRPRES}
|
||||
, HostSupportsAFP{ctx->HostFeatures.SupportsAFP}
|
||||
, CTX {ctx} {
|
||||
|
||||
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
|
||||
|
||||
@@ -26,6 +26,7 @@ $end_info$
|
||||
#include <array>
|
||||
#include <cstdint>
|
||||
#include <utility>
|
||||
#include <variant>
|
||||
|
||||
namespace FEXCore::Core {
|
||||
struct InternalThreadState;
|
||||
@@ -59,6 +60,7 @@ private:
|
||||
const bool HostSupportsSVE128{};
|
||||
const bool HostSupportsSVE256{};
|
||||
const bool HostSupportsRPRES{};
|
||||
const bool HostSupportsAFP{};
|
||||
|
||||
ARMEmitter::BiDirectionalLabel *PendingTargetLabel;
|
||||
FEXCore::Context::ContextImpl *CTX;
|
||||
@@ -209,6 +211,11 @@ private:
|
||||
uint32_t SpillSlots{};
|
||||
using OpType = void (Arm64JITCore::*)(IR::IROp_Header const *IROp, IR::NodeID Node);
|
||||
|
||||
using ScalarBinaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2)>;
|
||||
void VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2);
|
||||
using ScalarUnaryOpCaller = std::function<void(ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar)>;
|
||||
void VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2);
|
||||
|
||||
// Runtime selection;
|
||||
// Load and store register style.
|
||||
OpType RT_LoadRegister;
|
||||
|
||||
@@ -14,6 +14,786 @@ $end_info$
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const *IROp, IR::NodeID Node)
|
||||
|
||||
void Arm64JITCore::VFScalarOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarBinaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, ARMEmitter::VRegister Vector2) {
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (!Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(ZeroUpperBits == false, "128-bit operation doesn't support ZeroUpperBits in {}", __func__);
|
||||
}
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
constexpr auto Predicate = ARMEmitter::PReg::p0;
|
||||
|
||||
if (Dst == Vector1) {
|
||||
if (ZeroUpperBits) {
|
||||
// When zeroing the upper 128-bits we just use an ASIMD move.
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
}
|
||||
|
||||
if (HostSupportsAFP) {
|
||||
// If the host CPU supports AFP then scalar does an insert without modifying upper bits.
|
||||
ScalarEmit(Dst, Vector1, Vector2);
|
||||
}
|
||||
else {
|
||||
// If AFP is unsupported then the operation result goes in to a temporary.
|
||||
// and then it gets inserted.
|
||||
ScalarEmit(VTMP1, Vector1, Vector2);
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
|
||||
}
|
||||
else {
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (Dst != Vector2) {
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
mov(Dst.Z(), Vector1.Z());
|
||||
}
|
||||
else {
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
}
|
||||
|
||||
if (HostSupportsAFP) {
|
||||
ScalarEmit(Dst, Vector1, Vector2);
|
||||
}
|
||||
else {
|
||||
ScalarEmit(VTMP1, Vector1, Vector2);
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
|
||||
}
|
||||
else {
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Destination intersects Vector2, can't do anything optimal in this case.
|
||||
// Do the scalar operation first and then move and insert.
|
||||
ScalarEmit(VTMP1, Vector1, Vector2);
|
||||
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
mov(Dst.Z(), Vector1.Z());
|
||||
}
|
||||
else {
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
}
|
||||
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
|
||||
}
|
||||
else {
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
void Arm64JITCore::VFScalarUnaryOperation(uint8_t OpSize, uint8_t ElementSize, bool ZeroUpperBits, ScalarUnaryOpCaller ScalarEmit, ARMEmitter::VRegister Dst, ARMEmitter::VRegister Vector1, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> Vector2) {
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
if (!Is256Bit) {
|
||||
LOGMAN_THROW_A_FMT(ZeroUpperBits == false, "128-bit operation doesn't support ZeroUpperBits in {}", __func__);
|
||||
}
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
constexpr auto Predicate = ARMEmitter::PReg::p0;
|
||||
bool DstOverlapsVector2 = false;
|
||||
if (const auto* Vector2Reg = std::get_if<ARMEmitter::VRegister>(&Vector2)) {
|
||||
DstOverlapsVector2 = Dst == *Vector2Reg;
|
||||
}
|
||||
|
||||
if (Dst == Vector1) {
|
||||
if (ZeroUpperBits) {
|
||||
// When zeroing the upper 128-bits we just use an ASIMD move.
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
}
|
||||
|
||||
if (HostSupportsAFP) {
|
||||
// If the host CPU supports AFP then scalar does an insert without modifying upper bits.
|
||||
ScalarEmit(Dst, Vector2);
|
||||
}
|
||||
else {
|
||||
// If AFP is unsupported then the operation result goes in to a temporary.
|
||||
// and then it gets inserted.
|
||||
ScalarEmit(VTMP1, Vector2);
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
|
||||
}
|
||||
else {
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (!DstOverlapsVector2) {
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
mov(Dst.Z(), Vector1.Z());
|
||||
}
|
||||
else {
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
}
|
||||
|
||||
if (HostSupportsAFP) {
|
||||
ScalarEmit(Dst, Vector2);
|
||||
}
|
||||
else {
|
||||
ScalarEmit(VTMP1, Vector2);
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
|
||||
}
|
||||
else {
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
else {
|
||||
// Destination intersects Vector2, can't do anything optimal in this case.
|
||||
// Do the scalar operation first and then move and insert.
|
||||
ScalarEmit(VTMP1, Vector2);
|
||||
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
mov(Dst.Z(), Vector1.Z());
|
||||
}
|
||||
else {
|
||||
mov(Dst.Q(), Vector1.Q());
|
||||
}
|
||||
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
|
||||
}
|
||||
else {
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VFAddScalarInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VFAddScalarInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
fadd(SubRegSize.Scalar, Dst, Src1, Src2);
|
||||
};
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VFSubScalarInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VFSubScalarInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
fsub(SubRegSize.Scalar, Dst, Src1, Src2);
|
||||
};
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VFMulScalarInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VFMulScalarInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
fmul(SubRegSize.Scalar, Dst, Src1, Src2);
|
||||
};
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VFDivScalarInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VFDivScalarInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
fdiv(SubRegSize.Scalar, Dst, Src1, Src2);
|
||||
};
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VFMinScalarInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VFMinScalarInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
if (HostSupportsAFP) {
|
||||
// AFP.AH lets fmin behave like x86 min
|
||||
fmin(SubRegSize.Scalar, Dst, Src1, Src2);
|
||||
}
|
||||
else {
|
||||
fcmp(SubRegSize.Scalar, Src1, Src2);
|
||||
fcsel(SubRegSize.Scalar, Dst, Src1, Src2, ARMEmitter::Condition::CC_MI);
|
||||
}
|
||||
};
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VFMaxScalarInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VFMaxScalarInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
// AFP can make this more optimal.
|
||||
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
if (HostSupportsAFP) {
|
||||
// AFP.AH lets fmax behave like x86 max
|
||||
fmax(SubRegSize.Scalar, Dst, Src1, Src2);
|
||||
}
|
||||
else {
|
||||
fcmp(SubRegSize.Scalar, Src1, Src2);
|
||||
fcsel(SubRegSize.Scalar, Dst, Src2, Src1, ARMEmitter::Condition::CC_MI);
|
||||
}
|
||||
};
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VFSqrtScalarInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VFSqrtScalarInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
fsqrt(SubRegSize.Scalar, Dst, Src);
|
||||
};
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VFRSqrtScalarInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VFRSqrtScalarInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
|
||||
fmov(SubRegSize.Scalar, VTMP1.Q(), 1.0f);
|
||||
fsqrt(SubRegSize.Scalar, VTMP2, Src);
|
||||
fdiv(SubRegSize.Scalar, Dst, VTMP1, VTMP2);
|
||||
};
|
||||
|
||||
auto ScalarEmitRPRES = [this, SubRegSize](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
frecpe(SubRegSize.Scalar, Dst.S(), Src.S());
|
||||
};
|
||||
|
||||
std::array<ScalarUnaryOpCaller, 2> Handlers = {
|
||||
ScalarEmit,
|
||||
ScalarEmitRPRES,
|
||||
};
|
||||
const auto HandlerIndex = ElementSize == 4 && HostSupportsRPRES ? 1 : 0;
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, Handlers[HandlerIndex], Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VFRecpScalarInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VFRecpScalarInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
auto ScalarEmit = [this, SubRegSize](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
|
||||
fmov(SubRegSize.Scalar, VTMP1.Q(), 1.0f);
|
||||
fdiv(SubRegSize.Scalar, Dst, VTMP1, Src);
|
||||
};
|
||||
|
||||
auto ScalarEmitRPRES = [this, SubRegSize](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
frsqrte(SubRegSize.Scalar, Dst, Src);
|
||||
};
|
||||
|
||||
std::array<ScalarUnaryOpCaller, 2> Handlers = {
|
||||
ScalarEmit,
|
||||
ScalarEmitRPRES,
|
||||
};
|
||||
const auto HandlerIndex = ElementSize == 4 && HostSupportsRPRES ? 1 : 0;
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, Handlers[HandlerIndex], Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VFToFScalarInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VFToFScalarInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (Op->Header.ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
auto ScalarEmit = [this, Conv](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0204: { // Half <- Float
|
||||
fcvt(Dst.H(), Src.S());
|
||||
break;
|
||||
}
|
||||
case 0x0208: { // Half <- Double
|
||||
fcvt(Dst.H(), Src.D());
|
||||
break;
|
||||
}
|
||||
case 0x0402: { // Float <- Half
|
||||
fcvt(Dst.S(), Src.H());
|
||||
break;
|
||||
}
|
||||
case 0x0802: { // Double <- Half
|
||||
fcvt(Dst.D(), Src.H());
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- Float
|
||||
fcvt(Dst.D(), Src.S());
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- Double
|
||||
fcvt(Dst.S(), Src.D());
|
||||
break;
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown FCVT sizes: 0x{:x}", Conv);
|
||||
}
|
||||
};
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VSToFVectorInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VSToFVectorInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
const auto HasTwoElements = Op->HasTwoElements;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
if (HasTwoElements) {
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 4, "Can't have two elements for 8-byte size");
|
||||
}
|
||||
|
||||
auto ScalarEmit = [this, ElementSize, HasTwoElements](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
if (ElementSize == 4) {
|
||||
if (HasTwoElements) {
|
||||
scvtf(ARMEmitter::SubRegSize::i32Bit, Dst.D(), Src.D());
|
||||
}
|
||||
else {
|
||||
scvtf(ARMEmitter::ScalarRegSize::i32Bit, Dst.S(), Src.S());
|
||||
}
|
||||
}
|
||||
else {
|
||||
scvtf(ARMEmitter::ScalarRegSize::i64Bit, Dst.D(), Src.D());
|
||||
}
|
||||
};
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
// Claim the element size is 8-bytes.
|
||||
// Might be scalar 8-byte (cvtsi2ss xmm0, rax)
|
||||
// Might be vector i32v2 (cvtpi2ps xmm0, mm0)
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize * (HasTwoElements ? 2 : 1), Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VSToFGPRInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VSToFGPRInsert>();
|
||||
|
||||
const uint16_t ElementSize = Op->Header.ElementSize;
|
||||
const uint16_t Conv = (ElementSize << 8) | Op->SrcElementSize;
|
||||
|
||||
auto ScalarEmit = [this, Conv](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
|
||||
auto Src = *std::get_if<ARMEmitter::Register>(&SrcVar);
|
||||
|
||||
switch (Conv) {
|
||||
case 0x0204: { // Half <- int32_t
|
||||
scvtf(ARMEmitter::Size::i32Bit, Dst.H(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0208: { // Half <- int64_t
|
||||
scvtf(ARMEmitter::Size::i64Bit, Dst.H(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0404: { // Float <- int32_t
|
||||
scvtf(ARMEmitter::Size::i32Bit, Dst.S(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0408: { // Float <- int64_t
|
||||
scvtf(ARMEmitter::Size::i64Bit, Dst.S(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0804: { // Double <- int32_t
|
||||
scvtf(ARMEmitter::Size::i32Bit, Dst.D(), Src);
|
||||
break;
|
||||
}
|
||||
case 0x0808: { // Double <- int64_t
|
||||
scvtf(ARMEmitter::Size::i64Bit, Dst.D(), Src);
|
||||
break;
|
||||
}
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unhandled conversion mask: Mask=0x{:04x}",
|
||||
Conv);
|
||||
break;
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector.ID());
|
||||
const auto GPR = GetReg(Op->Src.ID());
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector, GPR);
|
||||
}
|
||||
|
||||
DEF_OP(VFToIScalarInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VFToIScalarInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
const auto RoundMode = Op->Round;
|
||||
|
||||
auto ScalarEmit = [this, SubRegSize, RoundMode](ARMEmitter::VRegister Dst, std::variant<ARMEmitter::VRegister, ARMEmitter::Register> SrcVar) {
|
||||
auto Src = *std::get_if<ARMEmitter::VRegister>(&SrcVar);
|
||||
|
||||
switch (RoundMode) {
|
||||
case IR::Round_Nearest:
|
||||
frintn(SubRegSize.Scalar, Dst, Src);
|
||||
break;
|
||||
case IR::Round_Negative_Infinity:
|
||||
frintm(SubRegSize.Scalar, Dst, Src);
|
||||
break;
|
||||
case IR::Round_Positive_Infinity:
|
||||
frintp(SubRegSize.Scalar, Dst, Src);
|
||||
break;
|
||||
case IR::Round_Towards_Zero:
|
||||
frintz(SubRegSize.Scalar, Dst, Src);
|
||||
break;
|
||||
case IR::Round_Host:
|
||||
frinti(SubRegSize.Scalar, Dst, Src);
|
||||
break;
|
||||
}
|
||||
};
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
VFScalarUnaryOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, ScalarEmit, Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VFCMPScalarInsert) {
|
||||
const auto Op = IROp->C<IR::IROp_VFCMPScalarInsert>();
|
||||
const auto ElementSize = Op->Header.ElementSize;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size");
|
||||
const auto SubRegSize = ARMEmitter::ToVectorSizePair(
|
||||
ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
|
||||
ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
|
||||
ARMEmitter::SubRegSize::i64Bit);
|
||||
|
||||
const auto ZeroUpperBits = Op->ZeroUpperBits;
|
||||
const auto Is256Bit = IROp->Size == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
auto ScalarEmitEQ = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
switch (SubRegSize.Scalar) {
|
||||
case ARMEmitter::ScalarRegSize::i16Bit: {
|
||||
fcmeq(Dst.H(), Src1.H(), Src2.H());
|
||||
break;
|
||||
}
|
||||
case ARMEmitter::ScalarRegSize::i32Bit:
|
||||
case ARMEmitter::ScalarRegSize::i64Bit:
|
||||
fcmeq(SubRegSize.Scalar, Dst, Src1, Src2);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
};
|
||||
auto ScalarEmitLT = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
switch (SubRegSize.Scalar) {
|
||||
case ARMEmitter::ScalarRegSize::i16Bit: {
|
||||
fcmgt(Dst.H(), Src2.H(), Src1.H());
|
||||
break;
|
||||
}
|
||||
case ARMEmitter::ScalarRegSize::i32Bit:
|
||||
case ARMEmitter::ScalarRegSize::i64Bit:
|
||||
fcmgt(SubRegSize.Scalar, Dst, Src2, Src1);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
};
|
||||
auto ScalarEmitLE = [this, SubRegSize](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
switch (SubRegSize.Scalar) {
|
||||
case ARMEmitter::ScalarRegSize::i16Bit: {
|
||||
fcmge(Dst.H(), Src2.H(), Src1.H());
|
||||
break;
|
||||
}
|
||||
case ARMEmitter::ScalarRegSize::i32Bit:
|
||||
case ARMEmitter::ScalarRegSize::i64Bit:
|
||||
fcmge(SubRegSize.Scalar, Dst, Src2, Src1);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
};
|
||||
auto ScalarEmitUNO = [this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
switch (SubRegSize.Scalar) {
|
||||
case ARMEmitter::ScalarRegSize::i16Bit: {
|
||||
fcmge(VTMP1.H(), Src1.H(), Src2.H());
|
||||
fcmgt(VTMP2.H(), Src2.H(), Src1.H());
|
||||
break;
|
||||
}
|
||||
case ARMEmitter::ScalarRegSize::i32Bit:
|
||||
case ARMEmitter::ScalarRegSize::i64Bit:
|
||||
fcmge(SubRegSize.Scalar, VTMP1, Src1, Src2);
|
||||
fcmgt(SubRegSize.Scalar, VTMP2, Src2, Src1);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
// If the destination is a temporary then it is going to do an insert after the operation.
|
||||
// This means this operation can avoid a redundant insert in this case.
|
||||
const bool DstIsTemp = Dst == VTMP1;
|
||||
|
||||
// Combine results and invert directly in VTMP1.
|
||||
orr(VTMP1.D(), VTMP1.D(), VTMP2.D());
|
||||
mvn(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
|
||||
if (!DstIsTemp) {
|
||||
// If the destination doesn't overlap VTMP1, then we need to insert the final result.
|
||||
// This only happens in the case that the host supports AFP.
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
constexpr auto Predicate = ARMEmitter::PReg::p0;
|
||||
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
|
||||
}
|
||||
else {
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
};
|
||||
auto ScalarEmitNEQ = [this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
switch (SubRegSize.Scalar) {
|
||||
case ARMEmitter::ScalarRegSize::i16Bit: {
|
||||
fcmeq(VTMP1.H(), Src1.H(), Src2.H());
|
||||
break;
|
||||
}
|
||||
case ARMEmitter::ScalarRegSize::i32Bit:
|
||||
case ARMEmitter::ScalarRegSize::i64Bit:
|
||||
fcmeq(SubRegSize.Scalar, VTMP1, Src1, Src2);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
// If the destination is a temporary then it is going to do an insert after the operation.
|
||||
// This means this operation can avoid a redundant insert in this case.
|
||||
const bool DstIsTemp = Dst == VTMP1;
|
||||
|
||||
// Invert directly in VTMP1.
|
||||
mvn(ARMEmitter::SubRegSize::i8Bit, VTMP1.D(), VTMP1.D());
|
||||
|
||||
if (!DstIsTemp) {
|
||||
// If the destination doesn't overlap VTMP1, then we need to insert the final result.
|
||||
// This only happens in the case that the host supports AFP.
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
constexpr auto Predicate = ARMEmitter::PReg::p0;
|
||||
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
|
||||
}
|
||||
else {
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
};
|
||||
auto ScalarEmitORD = [this, SubRegSize, ZeroUpperBits, Is256Bit](ARMEmitter::VRegister Dst, ARMEmitter::VRegister Src1, ARMEmitter::VRegister Src2) {
|
||||
switch (SubRegSize.Scalar) {
|
||||
case ARMEmitter::ScalarRegSize::i16Bit: {
|
||||
fcmge(VTMP1.H(), Src1.H(), Src2.H());
|
||||
fcmgt(VTMP2.H(), Src2.H(), Src1.H());
|
||||
break;
|
||||
}
|
||||
case ARMEmitter::ScalarRegSize::i32Bit:
|
||||
case ARMEmitter::ScalarRegSize::i64Bit:
|
||||
fcmge(SubRegSize.Scalar, VTMP1, Src1, Src2);
|
||||
fcmgt(SubRegSize.Scalar, VTMP2, Src2, Src1);
|
||||
break;
|
||||
default:
|
||||
break;
|
||||
}
|
||||
// If the destination is a temporary then it is going to do an insert after the operation.
|
||||
// This means this operation can avoid a redundant insert in this case.
|
||||
const bool DstIsTemp = Dst == VTMP1;
|
||||
|
||||
// Combine results directly in VTMP1.
|
||||
orr(VTMP1.D(), VTMP1.D(), VTMP2.D());
|
||||
|
||||
if (!DstIsTemp) {
|
||||
// If the destination doesn't overlap VTMP1, then we need to insert the final result.
|
||||
// This only happens in the case that the host supports AFP.
|
||||
if (!ZeroUpperBits && Is256Bit) {
|
||||
constexpr auto Predicate = ARMEmitter::PReg::p0;
|
||||
ptrue(SubRegSize.Vector, Predicate, ARMEmitter::PredicatePattern::SVE_VL1);
|
||||
mov(SubRegSize.Vector, Dst.Z(), Predicate.Merging(), VTMP1.Z());
|
||||
}
|
||||
else {
|
||||
ins(SubRegSize.Vector, Dst.Q(), 0, VTMP1.Q(), 0);
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
std::array<ScalarBinaryOpCaller, 6> Funcs = {{
|
||||
ScalarEmitEQ,
|
||||
ScalarEmitLT,
|
||||
ScalarEmitLE,
|
||||
ScalarEmitUNO,
|
||||
ScalarEmitNEQ,
|
||||
ScalarEmitORD,
|
||||
}};
|
||||
|
||||
// Bit of a tricky detail.
|
||||
// The upper bits of the destination comes from the first source.
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector1 = GetVReg(Op->Vector1.ID());
|
||||
const auto Vector2 = GetVReg(Op->Vector2.ID());
|
||||
|
||||
VFScalarOperation(IROp->Size, ElementSize, Op->ZeroUpperBits, Funcs[FEXCore::ToUnderlying(Op->Op)], Dst, Vector1, Vector2);
|
||||
}
|
||||
|
||||
DEF_OP(VectorZero) {
|
||||
const auto OpSize = IROp->Size;
|
||||
const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
|
||||
@@ -5030,7 +5030,7 @@ OrderedNode *OpDispatchBuilder::LoadSource_WithOpSize(FEXCore::IR::RegisterClass
|
||||
// Now extract the subregister if it was a partial load /smaller/ than SSE size
|
||||
// TODO: Instead of doing the VMov implicitly on load, hunt down all use cases that require partial loads and do it after load.
|
||||
// We don't have information here to know if the operation needs zero upper bits or can contain data.
|
||||
if (OpSize < Core::CPUState::XMM_SSE_REG_SIZE) {
|
||||
if (!AllowUpperGarbage && OpSize < Core::CPUState::XMM_SSE_REG_SIZE) {
|
||||
Src = _VMov(OpSize, Src);
|
||||
}
|
||||
}
|
||||
@@ -5908,8 +5908,8 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b00, 0x29), 1, &OpDispatchBuilder::MOVAPS_MOVAPDOp},
|
||||
{OPD(1, 0b01, 0x29), 1, &OpDispatchBuilder::MOVAPS_MOVAPDOp},
|
||||
|
||||
{OPD(1, 0b10, 0x2A), 1, &OpDispatchBuilder::AVXCVTGPR_To_FPR<4>},
|
||||
{OPD(1, 0b11, 0x2A), 1, &OpDispatchBuilder::AVXCVTGPR_To_FPR<8>},
|
||||
{OPD(1, 0b10, 0x2A), 1, &OpDispatchBuilder::AVXInsertCVTGPR_To_FPR<4>},
|
||||
{OPD(1, 0b11, 0x2A), 1, &OpDispatchBuilder::AVXInsertCVTGPR_To_FPR<8>},
|
||||
|
||||
{OPD(1, 0b00, 0x2B), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{OPD(1, 0b01, 0x2B), 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
@@ -5928,16 +5928,16 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b00, 0x50), 1, &OpDispatchBuilder::MOVMSKOp<4>},
|
||||
{OPD(1, 0b01, 0x50), 1, &OpDispatchBuilder::MOVMSKOp<8>},
|
||||
|
||||
{OPD(1, 0b00, 0x51), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 4, false>},
|
||||
{OPD(1, 0b01, 0x51), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 8, false>},
|
||||
{OPD(1, 0b10, 0x51), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 4, true>},
|
||||
{OPD(1, 0b11, 0x51), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 8, true>},
|
||||
{OPD(1, 0b00, 0x51), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 4>},
|
||||
{OPD(1, 0b01, 0x51), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 8>},
|
||||
{OPD(1, 0b10, 0x51), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 4>},
|
||||
{OPD(1, 0b11, 0x51), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 8>},
|
||||
|
||||
{OPD(1, 0b00, 0x52), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRSQRT, 4, false>},
|
||||
{OPD(1, 0b10, 0x52), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRSQRT, 4, true>},
|
||||
{OPD(1, 0b00, 0x52), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRSQRT, 4>},
|
||||
{OPD(1, 0b10, 0x52), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 4>},
|
||||
|
||||
{OPD(1, 0b00, 0x53), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRECP, 4, false>},
|
||||
{OPD(1, 0b10, 0x53), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRECP, 4, true>},
|
||||
{OPD(1, 0b00, 0x53), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRECP, 4>},
|
||||
{OPD(1, 0b10, 0x53), 1, &OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 4>},
|
||||
|
||||
{OPD(1, 0b00, 0x54), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VAND, 16>},
|
||||
{OPD(1, 0b01, 0x54), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VAND, 16>},
|
||||
@@ -5953,18 +5953,18 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
|
||||
{OPD(1, 0b00, 0x58), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFADD, 4>},
|
||||
{OPD(1, 0b01, 0x58), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFADD, 8>},
|
||||
{OPD(1, 0b10, 0x58), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFADD, 4>},
|
||||
{OPD(1, 0b11, 0x58), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFADD, 8>},
|
||||
{OPD(1, 0b10, 0x58), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 4>},
|
||||
{OPD(1, 0b11, 0x58), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 8>},
|
||||
|
||||
{OPD(1, 0b00, 0x59), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFMUL, 4>},
|
||||
{OPD(1, 0b01, 0x59), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFMUL, 8>},
|
||||
{OPD(1, 0b10, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMUL, 4>},
|
||||
{OPD(1, 0b11, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMUL, 8>},
|
||||
{OPD(1, 0b10, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 4>},
|
||||
{OPD(1, 0b11, 0x59), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 8>},
|
||||
|
||||
{OPD(1, 0b00, 0x5A), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Float<8, 4>},
|
||||
{OPD(1, 0b01, 0x5A), 1, &OpDispatchBuilder::Vector_CVT_Float_To_Float<4, 8>},
|
||||
{OPD(1, 0b10, 0x5A), 1, &OpDispatchBuilder::AVXScalar_CVT_Float_To_Float<8, 4>},
|
||||
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::AVXScalar_CVT_Float_To_Float<4, 8>},
|
||||
{OPD(1, 0b10, 0x5A), 1, &OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<8, 4>},
|
||||
{OPD(1, 0b11, 0x5A), 1, &OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<4, 8>},
|
||||
|
||||
{OPD(1, 0b00, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Int_To_Float<4, false>},
|
||||
{OPD(1, 0b01, 0x5B), 1, &OpDispatchBuilder::AVXVector_CVT_Float_To_Int<4, false, true>},
|
||||
@@ -5972,23 +5972,23 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
|
||||
{OPD(1, 0b00, 0x5C), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFSUB, 4>},
|
||||
{OPD(1, 0b01, 0x5C), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFSUB, 8>},
|
||||
{OPD(1, 0b10, 0x5C), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFSUB, 4>},
|
||||
{OPD(1, 0b11, 0x5C), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFSUB, 8>},
|
||||
{OPD(1, 0b10, 0x5C), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 4>},
|
||||
{OPD(1, 0b11, 0x5C), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 8>},
|
||||
|
||||
{OPD(1, 0b00, 0x5D), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFMIN, 4>},
|
||||
{OPD(1, 0b01, 0x5D), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFMIN, 8>},
|
||||
{OPD(1, 0b10, 0x5D), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMIN, 4>},
|
||||
{OPD(1, 0b11, 0x5D), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMIN, 8>},
|
||||
{OPD(1, 0b10, 0x5D), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 4>},
|
||||
{OPD(1, 0b11, 0x5D), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 8>},
|
||||
|
||||
{OPD(1, 0b00, 0x5E), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFDIV, 4>},
|
||||
{OPD(1, 0b01, 0x5E), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFDIV, 8>},
|
||||
{OPD(1, 0b10, 0x5E), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFDIV, 4>},
|
||||
{OPD(1, 0b11, 0x5E), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFDIV, 8>},
|
||||
{OPD(1, 0b10, 0x5E), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 4>},
|
||||
{OPD(1, 0b11, 0x5E), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 8>},
|
||||
|
||||
{OPD(1, 0b00, 0x5F), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFMAX, 4>},
|
||||
{OPD(1, 0b01, 0x5F), 1, &OpDispatchBuilder::AVXVectorALUOp<IR::OP_VFMAX, 8>},
|
||||
{OPD(1, 0b10, 0x5F), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMAX, 4>},
|
||||
{OPD(1, 0b11, 0x5F), 1, &OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMAX, 8>},
|
||||
{OPD(1, 0b10, 0x5F), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 4>},
|
||||
{OPD(1, 0b11, 0x5F), 1, &OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 8>},
|
||||
|
||||
{OPD(1, 0b01, 0x60), 1, &OpDispatchBuilder::VPUNPCKLOp<1>},
|
||||
{OPD(1, 0b01, 0x61), 1, &OpDispatchBuilder::VPUNPCKLOp<2>},
|
||||
@@ -6030,10 +6030,10 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(1, 0b01, 0x7F), 1, &OpDispatchBuilder::MOVAPS_MOVAPDOp},
|
||||
{OPD(1, 0b10, 0x7F), 1, &OpDispatchBuilder::MOVUPS_MOVUPDOp},
|
||||
|
||||
{OPD(1, 0b00, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<4, false>},
|
||||
{OPD(1, 0b01, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<8, false>},
|
||||
{OPD(1, 0b10, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<4, true>},
|
||||
{OPD(1, 0b11, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<8, true>},
|
||||
{OPD(1, 0b00, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<4>},
|
||||
{OPD(1, 0b01, 0xC2), 1, &OpDispatchBuilder::AVXVFCMPOp<8>},
|
||||
{OPD(1, 0b10, 0xC2), 1, &OpDispatchBuilder::AVXInsertScalarFCMPOp<4>},
|
||||
{OPD(1, 0b11, 0xC2), 1, &OpDispatchBuilder::AVXInsertScalarFCMPOp<8>},
|
||||
|
||||
{OPD(1, 0b01, 0xC4), 1, &OpDispatchBuilder::VPINSRWOp},
|
||||
{OPD(1, 0b01, 0xC5), 1, &OpDispatchBuilder::PExtrOp<2>},
|
||||
@@ -6124,9 +6124,9 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(2, 0b01, 0x18), 1, &OpDispatchBuilder::VBROADCASTOp<4>},
|
||||
{OPD(2, 0b01, 0x19), 1, &OpDispatchBuilder::VBROADCASTOp<8>},
|
||||
{OPD(2, 0b01, 0x1A), 1, &OpDispatchBuilder::VBROADCASTOp<16>},
|
||||
{OPD(2, 0b01, 0x1C), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 1, false>},
|
||||
{OPD(2, 0b01, 0x1D), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 2, false>},
|
||||
{OPD(2, 0b01, 0x1E), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 4, false>},
|
||||
{OPD(2, 0b01, 0x1C), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 1>},
|
||||
{OPD(2, 0b01, 0x1D), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 2>},
|
||||
{OPD(2, 0b01, 0x1E), 1, &OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 4>},
|
||||
|
||||
{OPD(2, 0b01, 0x20), 1, &OpDispatchBuilder::ExtendVectorElements<1, 2, true>},
|
||||
{OPD(2, 0b01, 0x21), 1, &OpDispatchBuilder::ExtendVectorElements<1, 4, true>},
|
||||
@@ -6190,10 +6190,10 @@ void OpDispatchBuilder::InstallHostSpecificOpcodeHandlers() {
|
||||
{OPD(3, 0b01, 0x04), 1, &OpDispatchBuilder::VPERMILImmOp<4>},
|
||||
{OPD(3, 0b01, 0x05), 1, &OpDispatchBuilder::VPERMILImmOp<8>},
|
||||
{OPD(3, 0b01, 0x06), 1, &OpDispatchBuilder::VPERM2Op},
|
||||
{OPD(3, 0b01, 0x08), 1, &OpDispatchBuilder::AVXVectorRound<4, false>},
|
||||
{OPD(3, 0b01, 0x09), 1, &OpDispatchBuilder::AVXVectorRound<8, false>},
|
||||
{OPD(3, 0b01, 0x0A), 1, &OpDispatchBuilder::AVXVectorRound<4, true>},
|
||||
{OPD(3, 0b01, 0x0B), 1, &OpDispatchBuilder::AVXVectorRound<8, true>},
|
||||
{OPD(3, 0b01, 0x08), 1, &OpDispatchBuilder::AVXVectorRound<4>},
|
||||
{OPD(3, 0b01, 0x09), 1, &OpDispatchBuilder::AVXVectorRound<8>},
|
||||
{OPD(3, 0b01, 0x0A), 1, &OpDispatchBuilder::AVXInsertScalarRound<4>},
|
||||
{OPD(3, 0b01, 0x0B), 1, &OpDispatchBuilder::AVXInsertScalarRound<8>},
|
||||
{OPD(3, 0b01, 0x0C), 1, &OpDispatchBuilder::VPBLENDDOp},
|
||||
{OPD(3, 0b01, 0x0D), 1, &OpDispatchBuilder::VBLENDPDOp},
|
||||
{OPD(3, 0b01, 0x0E), 1, &OpDispatchBuilder::VPBLENDWOp},
|
||||
@@ -6441,15 +6441,15 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x15, 1, &OpDispatchBuilder::PUNPCKHOp<4>},
|
||||
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVAPS_MOVAPDOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, false>},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertMMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<4, false, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<4, false, true>},
|
||||
{0x2E, 2, &OpDispatchBuilder::UCOMISxOp<4>},
|
||||
{0x50, 1, &OpDispatchBuilder::MOVMSKOp<4>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4, false>},
|
||||
{0x52, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4, false>},
|
||||
{0x53, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4, false>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4>},
|
||||
{0x52, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4>},
|
||||
{0x53, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4>},
|
||||
{0x54, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VAND, 16>},
|
||||
{0x55, 1, &OpDispatchBuilder::VectorALUROp<IR::OP_VBIC, 8>},
|
||||
{0x56, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VOR, 16>},
|
||||
@@ -6481,7 +6481,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x76, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VCMPEQ, 4>},
|
||||
{0x77, 1, &OpDispatchBuilder::X87EMMS},
|
||||
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<4, false>},
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<4>},
|
||||
{0xC6, 1, &OpDispatchBuilder::SHUFOp<4>},
|
||||
|
||||
{0xD1, 1, &OpDispatchBuilder::PSRLDOp<2>},
|
||||
@@ -6668,21 +6668,21 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x12, 1, &OpDispatchBuilder::VMOVSLDUPOp},
|
||||
{0x16, 1, &OpDispatchBuilder::VMOVSHDUPOp},
|
||||
{0x19, 7, &OpDispatchBuilder::NOPOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::CVTGPR_To_FPR<4>},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<4>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<4, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<4, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4, true>},
|
||||
{0x52, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4, true>},
|
||||
{0x53, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4, true>},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFADD, 4>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMUL, 4>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Scalar_CVT_Float_To_Float<8, 4>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 4>},
|
||||
{0x52, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 4>},
|
||||
{0x53, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 4>},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 4>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 4>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<8, 4>},
|
||||
{0x5B, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFSUB, 4>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMIN, 4>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFDIV, 4>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMAX, 4>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 4>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 4>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 4>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 4>},
|
||||
{0x6F, 1, &OpDispatchBuilder::MOVUPS_MOVUPDOp},
|
||||
{0x70, 1, &OpDispatchBuilder::PSHUFWOp<false>},
|
||||
{0x7E, 1, &OpDispatchBuilder::MOVQOp},
|
||||
@@ -6690,7 +6690,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0xB8, 1, &OpDispatchBuilder::PopcountOp},
|
||||
{0xBC, 1, &OpDispatchBuilder::TZCNT},
|
||||
{0xBD, 1, &OpDispatchBuilder::LZCNT},
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<4, true>},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<4>},
|
||||
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<true>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Int_To_Float<4, true>},
|
||||
};
|
||||
@@ -6699,25 +6699,25 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x10, 2, &OpDispatchBuilder::MOVSDOp},
|
||||
{0x12, 1, &OpDispatchBuilder::MOVDDUPOp},
|
||||
{0x19, 7, &OpDispatchBuilder::NOPOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::CVTGPR_To_FPR<8>},
|
||||
{0x2A, 1, &OpDispatchBuilder::InsertCVTGPR_To_FPR<8>},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::CVTFPR_To_GPR<8, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::CVTFPR_To_GPR<8, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8, true>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 8>},
|
||||
//x52 = Invalid
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFADD, 8>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMUL, 8>},
|
||||
{0x5A, 1, &OpDispatchBuilder::Scalar_CVT_Float_To_Float<4, 8>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFSUB, 8>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMIN, 8>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFDIV, 8>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMAX, 8>},
|
||||
{0x58, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 8>},
|
||||
{0x59, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 8>},
|
||||
{0x5A, 1, &OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<4, 8>},
|
||||
{0x5C, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 8>},
|
||||
{0x5D, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 8>},
|
||||
{0x5E, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 8>},
|
||||
{0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 8>},
|
||||
{0x70, 1, &OpDispatchBuilder::PSHUFWOp<true>},
|
||||
{0x7C, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VFADDP, 4>},
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<4>},
|
||||
{0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<4>},
|
||||
{0xD6, 1, &OpDispatchBuilder::MOVQ2DQ<false>},
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<8, true>},
|
||||
{0xC2, 1, &OpDispatchBuilder::InsertScalarFCMPOp<8>},
|
||||
{0xE6, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, true>},
|
||||
{0xF0, 1, &OpDispatchBuilder::MOVVectorOp},
|
||||
};
|
||||
@@ -6730,7 +6730,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x16, 2, &OpDispatchBuilder::MOVHPDOp},
|
||||
{0x19, 7, &OpDispatchBuilder::NOPOp},
|
||||
{0x28, 2, &OpDispatchBuilder::MOVAPS_MOVAPDOp},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, true>},
|
||||
{0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float},
|
||||
{0x2B, 1, &OpDispatchBuilder::MOVVectorNTOp},
|
||||
{0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, false>},
|
||||
{0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, true>},
|
||||
@@ -6738,7 +6738,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
|
||||
{0x40, 16, &OpDispatchBuilder::CMOVOp},
|
||||
{0x50, 1, &OpDispatchBuilder::MOVMSKOp<8>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8, false>},
|
||||
{0x51, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8>},
|
||||
{0x54, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VAND, 16>},
|
||||
{0x55, 1, &OpDispatchBuilder::VectorALUROp<IR::OP_VBIC, 8>},
|
||||
{0x56, 1, &OpDispatchBuilder::VectorALUOp<IR::OP_VOR, 16>},
|
||||
@@ -6777,7 +6777,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) {
|
||||
{0x7D, 1, &OpDispatchBuilder::HSUBP<8>},
|
||||
{0x7E, 1, &OpDispatchBuilder::MOVBetweenGPR_FPR},
|
||||
{0x7F, 1, &OpDispatchBuilder::MOVUPS_MOVUPDOp},
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<8, false>},
|
||||
{0xC2, 1, &OpDispatchBuilder::VFCMPOp<8>},
|
||||
{0xC4, 1, &OpDispatchBuilder::PINSROp<2>},
|
||||
{0xC5, 1, &OpDispatchBuilder::PExtrOp<2>},
|
||||
{0xC6, 1, &OpDispatchBuilder::SHUFOp<8>},
|
||||
@@ -7444,12 +7444,12 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{OPD(PF_38_66, 0x14), 1, &OpDispatchBuilder::VectorVariableBlend<4>},
|
||||
{OPD(PF_38_66, 0x15), 1, &OpDispatchBuilder::VectorVariableBlend<8>},
|
||||
{OPD(PF_38_66, 0x17), 1, &OpDispatchBuilder::PTestOp},
|
||||
{OPD(PF_38_NONE, 0x1C), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 1, false>},
|
||||
{OPD(PF_38_66, 0x1C), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 1, false>},
|
||||
{OPD(PF_38_NONE, 0x1D), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2, false>},
|
||||
{OPD(PF_38_66, 0x1D), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2, false>},
|
||||
{OPD(PF_38_NONE, 0x1E), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4, false>},
|
||||
{OPD(PF_38_66, 0x1E), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4, false>},
|
||||
{OPD(PF_38_NONE, 0x1C), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 1>},
|
||||
{OPD(PF_38_66, 0x1C), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 1>},
|
||||
{OPD(PF_38_NONE, 0x1D), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2>},
|
||||
{OPD(PF_38_66, 0x1D), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2>},
|
||||
{OPD(PF_38_NONE, 0x1E), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4>},
|
||||
{OPD(PF_38_66, 0x1E), 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4>},
|
||||
{OPD(PF_38_66, 0x20), 1, &OpDispatchBuilder::ExtendVectorElements<1, 2, true>},
|
||||
{OPD(PF_38_66, 0x21), 1, &OpDispatchBuilder::ExtendVectorElements<1, 4, true>},
|
||||
{OPD(PF_38_66, 0x22), 1, &OpDispatchBuilder::ExtendVectorElements<1, 8, true>},
|
||||
@@ -7491,10 +7491,10 @@ constexpr uint16_t PF_F2 = 3;
|
||||
#define PF_3A_NONE 0
|
||||
#define PF_3A_66 1
|
||||
constexpr std::tuple<uint16_t, uint8_t, FEXCore::X86Tables::OpDispatchPtr> H0F3ATable[] = {
|
||||
{OPD(0, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<4, false>},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<8, false>},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::VectorRound<4, true>},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::VectorRound<8, true>},
|
||||
{OPD(0, PF_3A_66, 0x08), 1, &OpDispatchBuilder::VectorRound<4>},
|
||||
{OPD(0, PF_3A_66, 0x09), 1, &OpDispatchBuilder::VectorRound<8>},
|
||||
{OPD(0, PF_3A_66, 0x0A), 1, &OpDispatchBuilder::InsertScalarRound<4>},
|
||||
{OPD(0, PF_3A_66, 0x0B), 1, &OpDispatchBuilder::InsertScalarRound<8>},
|
||||
{OPD(0, PF_3A_66, 0x0C), 1, &OpDispatchBuilder::VectorBlend<4>},
|
||||
{OPD(0, PF_3A_66, 0x0D), 1, &OpDispatchBuilder::VectorBlend<8>},
|
||||
{OPD(0, PF_3A_66, 0x0E), 1, &OpDispatchBuilder::VectorBlend<2>},
|
||||
@@ -7535,8 +7535,8 @@ constexpr uint16_t PF_F2 = 3;
|
||||
{0x1C, 1, &OpDispatchBuilder::PF2IWOp},
|
||||
{0x1D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>},
|
||||
|
||||
{0x86, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4, false>},
|
||||
{0x87, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4, false>},
|
||||
{0x86, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4>},
|
||||
{0x87, 1, &OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4>},
|
||||
|
||||
{0x8A, 1, &OpDispatchBuilder::PFNACCOp},
|
||||
{0x8E, 1, &OpDispatchBuilder::PFPNACCOp},
|
||||
|
||||
@@ -333,8 +333,6 @@ public:
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorALUROp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorScalarALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize, bool Scalar>
|
||||
void VectorUnaryOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorUnaryDuplicateOp(OpcodeArgs);
|
||||
@@ -379,7 +377,6 @@ public:
|
||||
void Vector_CVT_Float_To_Float(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Widen>
|
||||
void MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs);
|
||||
@@ -387,7 +384,7 @@ public:
|
||||
void MOVBetweenGPR_FPR(OpcodeArgs);
|
||||
void TZCNT(OpcodeArgs);
|
||||
void LZCNT(OpcodeArgs);
|
||||
template<size_t ElementSize, bool Scalar>
|
||||
template<size_t ElementSize>
|
||||
void VFCMPOp(OpcodeArgs);
|
||||
template<size_t ElementSize>
|
||||
void SHUFOp(OpcodeArgs);
|
||||
@@ -424,11 +421,9 @@ public:
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void AVXVectorALUOp(OpcodeArgs);
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void AVXVectorScalarALUOp(OpcodeArgs);
|
||||
template <IROps IROp, size_t ElementSize, bool Scalar>
|
||||
void AVXVectorUnaryOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Scalar>
|
||||
template <size_t ElementSize>
|
||||
void AVXVectorRound(OpcodeArgs);
|
||||
|
||||
template <size_t DstElementSize, size_t SrcElementSize>
|
||||
@@ -440,10 +435,41 @@ public:
|
||||
template <size_t SrcElementSize, bool Widen>
|
||||
void AVXVector_CVT_Int_To_Float(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorScalarInsertALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void AVXVectorScalarInsertALUOp(OpcodeArgs);
|
||||
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void VectorScalarUnaryInsertALUOp(OpcodeArgs);
|
||||
template<FEXCore::IR::IROps IROp, size_t ElementSize>
|
||||
void AVXVectorScalarUnaryInsertALUOp(OpcodeArgs);
|
||||
|
||||
void InsertMMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs);
|
||||
template<size_t DstElementSize>
|
||||
void InsertCVTGPR_To_FPR(OpcodeArgs);
|
||||
template <size_t DstElementSize>
|
||||
void AVXInsertCVTGPR_To_FPR(OpcodeArgs);
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void InsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void InsertScalarRound(OpcodeArgs);
|
||||
template <size_t ElementSize>
|
||||
void AVXInsertScalarRound(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void InsertScalarFCMPOp(OpcodeArgs);
|
||||
template <size_t ElementSize>
|
||||
void AVXInsertScalarFCMPOp(OpcodeArgs);
|
||||
|
||||
template <size_t DstElementSize>
|
||||
void AVXCVTGPR_To_FPR(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Scalar>
|
||||
template <size_t ElementSize>
|
||||
void AVXVFCMPOp(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize>
|
||||
@@ -787,7 +813,7 @@ public:
|
||||
|
||||
template<size_t ElementSize, size_t DstElementSize, bool Signed>
|
||||
void ExtendVectorElements(OpcodeArgs);
|
||||
template<size_t ElementSize, bool Scalar>
|
||||
template<size_t ElementSize>
|
||||
void VectorRound(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
@@ -889,8 +915,7 @@ private:
|
||||
OrderedNode *Src1, OrderedNode *Src2);
|
||||
|
||||
void AVXVectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVXVectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar);
|
||||
void AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
|
||||
template <size_t ElementSize>
|
||||
void AVXVectorVariableBlend(OpcodeArgs);
|
||||
@@ -997,19 +1022,63 @@ private:
|
||||
void MOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
void VMOVScalarOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
OrderedNode* VFCMPOpImpl(OpcodeArgs, size_t ElementSize, bool Scalar,
|
||||
OrderedNode* VFCMPOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src1, OrderedNode *Src2, uint8_t CompType);
|
||||
|
||||
void VTESTOpImpl(OpcodeArgs, size_t ElementSize);
|
||||
|
||||
void VectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorALUROpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar);
|
||||
void VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
void VectorUnaryDuplicateOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize);
|
||||
|
||||
// x86 ALU scalar operations operate in three different ways
|
||||
// - AVX512: Writemask shenanigans that we don't care about.
|
||||
// - AVX/VEX: Two source
|
||||
// - Example 32bit VADDSS Dest, Src1, Src2
|
||||
// - Dest[31:0] = Src1[31:0] + Src2[31:0]
|
||||
// - Dest[127:32] = Src1[127:32]
|
||||
// - SSE: Scalar operation inserts in to the low bits, upper bits completely unaffected.
|
||||
// - Example 32bit ADDSS Dest, Src
|
||||
// - Dest[31:0] = Dest[31:0] + Src[31:0]
|
||||
// - Dest[{256,128}:32] = (Unmodified)
|
||||
OrderedNode* VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IROp,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IROps IROp,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* InsertCVTGPR_To_FPRImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t DstElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* InsertScalar_CVT_Float_To_FloatImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t DstElementSize, size_t SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits);
|
||||
OrderedNode* InsertScalarRoundImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
uint64_t Mode, bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* InsertScalarFCMPOpImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
uint8_t CompType, bool ZeroUpperBits);
|
||||
|
||||
OrderedNode* VectorRoundImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src, uint64_t Mode, bool IsScalar);
|
||||
OrderedNode *Src, uint64_t Mode);
|
||||
|
||||
OrderedNode* Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
@@ -1067,6 +1136,10 @@ private:
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, FEXCore::X86Tables::DecodedOperand const& Operand, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, OrderedNode *const Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::ACCESS_DEFAULT);
|
||||
|
||||
constexpr OpSize GetGuestVectorLength() const {
|
||||
return CTX->HostFeatures.SupportsAVX ? OpSize::i256Bit : OpSize::i128Bit;
|
||||
}
|
||||
|
||||
[[nodiscard]] static uint32_t GPROffset(X86State::X86Reg reg) {
|
||||
LOGMAN_THROW_AA_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used");
|
||||
return static_cast<uint32_t>(offsetof(Core::CPUState, gregs[static_cast<size_t>(reg)]));
|
||||
|
||||
@@ -500,236 +500,487 @@ void OpDispatchBuilder::VectorALUROp<IR::OP_VFSUB, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorALUROp<IR::OP_VFSUB, 8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::VectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize) {
|
||||
OrderedNode* OpDispatchBuilder::VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IROp,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits) {
|
||||
// We load the full vector width when dealing with a source vector,
|
||||
// so that we don't do any unnecessary zero extension to the scalar
|
||||
// element that we're going to operate on.
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? 16U : GetSrcSize(Op);
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, -1, true, false, MemoryAccessType::ACCESS_DEFAULT, true);
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
auto ALUOp = _VFAddScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, ZeroUpperBits);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
|
||||
return ALUOp;
|
||||
}
|
||||
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp(OpcodeArgs) {
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Dest, Op->Src[0], false);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 8>(OpcodeArgs);
|
||||
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp(OpcodeArgs) {
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Src[0], Op->Src[1], true);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFADDSCALARINSERT, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFSUBSCALARINSERT, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMULSCALARINSERT, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFDIVSCALARINSERT, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMINSCALARINSERT, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarInsertALUOp<IR::OP_VFMAXSCALARINSERT, 8>(OpcodeArgs);
|
||||
|
||||
OrderedNode* OpDispatchBuilder::VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IROps IROp,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits) {
|
||||
// We load the full vector width when dealing with a source vector,
|
||||
// so that we don't do any unnecessary zero extension to the scalar
|
||||
// element that we're going to operate on.
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, -1, true, false, MemoryAccessType::ACCESS_DEFAULT, true);
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
auto ALUOp = _VFSqrtScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, ZeroUpperBits);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
|
||||
return ALUOp;
|
||||
}
|
||||
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp(OpcodeArgs) {
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Dest, Op->Src[0], false);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 8>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 8>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 8>(OpcodeArgs);
|
||||
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp(OpcodeArgs) {
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Src[0], Op->Src[1], true);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFSQRTSCALARINSERT, 8>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRSQRTSCALARINSERT, 8>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp<IR::OP_VFRECPSCALARINSERT, 8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::InsertMMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
// We load the full vector width when dealing with a source vector,
|
||||
// so that we don't do any unnecessary zero extension to the scalar
|
||||
// element that we're going to operate on.
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? 8 : GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1);
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
auto ALUOp = _VAdd(ElementSize, ElementSize, Dest, Src);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
// Always 32-bit.
|
||||
const size_t ElementSize = 4;
|
||||
// Always signed
|
||||
Dest = _VSToFVectorInsert(IR::SizeToOpSize(DstSize), ElementSize, ElementSize, Dest, Src, true, false);
|
||||
|
||||
OrderedNode* Result = ALUOp;
|
||||
|
||||
if (DstSize != ElementSize) {
|
||||
// Insert the lower bits
|
||||
Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, ALUOp);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Dest, DstSize, -1);
|
||||
}
|
||||
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs) {
|
||||
VectorScalarALUOpImpl(Op, IROp, ElementSize);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFADD, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFADD, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFSUB, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFSUB, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMUL, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMUL, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFDIV, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFDIV, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMIN, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMIN, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMAX, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorScalarALUOp<IR::OP_VFMAX, 8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize) {
|
||||
OrderedNode* OpDispatchBuilder::InsertCVTGPR_To_FPRImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t DstElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits) {
|
||||
// We load the full vector width when dealing with a source vector,
|
||||
// so that we don't do any unnecessary zero extension to the scalar
|
||||
// element that we're going to operate on.
|
||||
const auto SrcSize = Op->Src[1].IsGPR() ? 16U : GetSrcSize(Op);
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags, -1);
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags, -1);
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
auto ALUOp = _VAdd(ElementSize, ElementSize, Src1, Src2);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
|
||||
OrderedNode* Result = ALUOp;
|
||||
|
||||
if (DstSize != ElementSize) {
|
||||
// Insert the lower bits
|
||||
Result = _VInsElement(DstSize, ElementSize, 0, 0, Src1, ALUOp);
|
||||
if (Src2Op.IsGPR()) {
|
||||
// If the source is a GPR then convert directly from the GPR.
|
||||
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, CTX->GetGPRSize(), Op->Flags, -1);
|
||||
return _VSToFGPRInsert(IR::SizeToOpSize(DstSize), DstElementSize, SrcSize, Src1, Src2, ZeroUpperBits);
|
||||
}
|
||||
else if (SrcSize != DstElementSize) {
|
||||
// If the source is from memory but the Source size and destination size aren't the same,
|
||||
// then it is more optimal to load in to a GPR and convert between GPR->FPR.
|
||||
// ARM GPR->FPR conversion supports different size source and destinations while FPR->FPR doesn't.
|
||||
auto Src2 = LoadSource(GPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
return _VSToFGPRInsert(IR::SizeToOpSize(DstSize), DstElementSize, SrcSize, Src1, Src2, ZeroUpperBits);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
// In the case of cvtsi2s{s,d} where the source and destination are the same size,
|
||||
// then it is more optimal to load in to the FPR register directly and convert there.
|
||||
auto Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags, -1);
|
||||
// Always signed
|
||||
return _VSToFVectorInsert(IR::SizeToOpSize(DstSize), DstElementSize, DstElementSize, Src1, Src2, false, ZeroUpperBits);
|
||||
}
|
||||
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp(OpcodeArgs) {
|
||||
AVXVectorScalarALUOpImpl(Op, IROp, ElementSize);
|
||||
template<size_t DstElementSize>
|
||||
void OpDispatchBuilder::InsertCVTGPR_To_FPR(OpcodeArgs) {
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
auto Result = InsertCVTGPR_To_FPRImpl(Op, DstSize, DstElementSize, Op->Dest, Op->Src[0], false);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFADD, 4>(OpcodeArgs);
|
||||
void OpDispatchBuilder::InsertCVTGPR_To_FPR<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFADD, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFDIV, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFDIV, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMAX, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMAX, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMIN, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMIN, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMUL, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFMUL, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFSUB, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorScalarALUOp<IR::OP_VFSUB, 8>(OpcodeArgs);
|
||||
void OpDispatchBuilder::InsertCVTGPR_To_FPR<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar) {
|
||||
template <size_t DstElementSize>
|
||||
void OpDispatchBuilder::AVXInsertCVTGPR_To_FPR(OpcodeArgs) {
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
OrderedNode *Result = InsertCVTGPR_To_FPRImpl(Op, DstSize, DstElementSize, Op->Src[0], Op->Src[1], true);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
|
||||
}
|
||||
template
|
||||
void OpDispatchBuilder::AVXInsertCVTGPR_To_FPR<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXInsertCVTGPR_To_FPR<8>(OpcodeArgs);
|
||||
|
||||
OrderedNode* OpDispatchBuilder::InsertScalar_CVT_Float_To_FloatImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t DstElementSize, size_t SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
bool ZeroUpperBits) {
|
||||
|
||||
// We load the full vector width when dealing with a source vector,
|
||||
// so that we don't do any unnecessary zero extension to the scalar
|
||||
// element that we're going to operate on.
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, -1, true, false, MemoryAccessType::ACCESS_DEFAULT, true);
|
||||
|
||||
return _VFToFScalarInsert(IR::SizeToOpSize(DstSize), DstElementSize, SrcElementSize, Src1, Src2, ZeroUpperBits);
|
||||
}
|
||||
|
||||
template<size_t DstElementSize, size_t SrcElementSize>
|
||||
void OpDispatchBuilder::InsertScalar_CVT_Float_To_Float(OpcodeArgs) {
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
OrderedNode *Result = InsertScalar_CVT_Float_To_FloatImpl(Op, DstSize, DstElementSize, SrcElementSize, Op->Dest, Op->Src[0], false);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<4, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::InsertScalar_CVT_Float_To_Float<8, 4>(OpcodeArgs);
|
||||
|
||||
template <size_t DstElementSize, size_t SrcElementSize>
|
||||
void OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs) {
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
OrderedNode *Result = InsertScalar_CVT_Float_To_FloatImpl(Op, DstSize, DstElementSize, SrcElementSize, Op->Src[0], Op->Src[1], true);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<4, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float<8, 4>(OpcodeArgs);
|
||||
|
||||
OrderedNode* OpDispatchBuilder::InsertScalarRoundImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
uint64_t Mode, bool ZeroUpperBits) {
|
||||
// We load the full vector width when dealing with a source vector,
|
||||
// so that we don't do any unnecessary zero extension to the scalar
|
||||
// element that we're going to operate on.
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, -1, true, false, MemoryAccessType::ACCESS_DEFAULT, true);
|
||||
|
||||
const uint64_t RoundControlSource = (Mode >> 2) & 1;
|
||||
uint64_t RoundControl = Mode & 0b11;
|
||||
|
||||
static constexpr std::array SourceModes = {
|
||||
FEXCore::IR::Round_Nearest,
|
||||
FEXCore::IR::Round_Negative_Infinity,
|
||||
FEXCore::IR::Round_Positive_Infinity,
|
||||
FEXCore::IR::Round_Towards_Zero,
|
||||
};
|
||||
|
||||
const auto SourceMode = RoundControlSource ? Round_Host : SourceModes[RoundControl];
|
||||
|
||||
auto ALUOp = _VFToIScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, SourceMode, ZeroUpperBits);
|
||||
|
||||
return ALUOp;
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::InsertScalarRound(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
const uint64_t Mode = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
OrderedNode *Result = InsertScalarRoundImpl(Op, DstSize, ElementSize, Op->Dest, Op->Src[0], Mode, false);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::InsertScalarRound<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::InsertScalarRound<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::AVXInsertScalarRound(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[2].IsLiteral(), "Src1 needs to be literal here");
|
||||
const uint64_t Mode = Op->Src[2].Data.Literal.Value;
|
||||
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
OrderedNode *Result = InsertScalarRoundImpl(Op, DstSize, ElementSize, Op->Dest, Op->Src[0], Mode, true);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXInsertScalarRound<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXInsertScalarRound<8>(OpcodeArgs);
|
||||
|
||||
|
||||
OrderedNode* OpDispatchBuilder::InsertScalarFCMPOpImpl(OpcodeArgs,
|
||||
size_t DstSize, size_t ElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op,
|
||||
const X86Tables::DecodedOperand& Src2Op,
|
||||
uint8_t CompType, bool ZeroUpperBits) {
|
||||
// We load the full vector width when dealing with a source vector,
|
||||
// so that we don't do any unnecessary zero extension to the scalar
|
||||
// element that we're going to operate on.
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, -1, true, false, MemoryAccessType::ACCESS_DEFAULT, true);
|
||||
|
||||
switch (CompType) {
|
||||
case 0x00: case 0x08: case 0x10: case 0x18: // EQ
|
||||
return _VFCMPScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, FloatCompareOp::EQ, ZeroUpperBits);
|
||||
case 0x01: case 0x09: case 0x11: case 0x19: // LT, GT(Swapped operand)
|
||||
return _VFCMPScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, FloatCompareOp::LT, ZeroUpperBits);
|
||||
case 0x02: case 0x0A: case 0x12: case 0x1A: // LE, GE(Swapped operand)
|
||||
return _VFCMPScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, FloatCompareOp::LE, ZeroUpperBits);
|
||||
case 0x03: case 0x0B: case 0x13: case 0x1B: // Unordered
|
||||
return _VFCMPScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, FloatCompareOp::UNO, ZeroUpperBits);
|
||||
case 0x04: case 0x0C: case 0x14: case 0x1C: // NEQ
|
||||
return _VFCMPScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, FloatCompareOp::NEQ, ZeroUpperBits);
|
||||
case 0x05: case 0x0D: case 0x15: case 0x1D: { // NLT, NGT(Swapped operand)
|
||||
OrderedNode *Result = _VFCMPLT(ElementSize, ElementSize, Src1, Src2);
|
||||
Result = _VNot(ElementSize, ElementSize, Result);
|
||||
// Insert the lower bits
|
||||
return _VInsElement(GetDstSize(Op), ElementSize, 0, 0, Src1, Result);
|
||||
}
|
||||
case 0x06: case 0x0E: case 0x16: case 0x1E: { // NLE, NGE(Swapped operand)
|
||||
OrderedNode *Result = _VFCMPLE(ElementSize, ElementSize, Src1, Src2);
|
||||
Result = _VNot(ElementSize, ElementSize, Result);
|
||||
// Insert the lower bits
|
||||
return _VInsElement(GetDstSize(Op), ElementSize, 0, 0, Src1, Result);
|
||||
}
|
||||
case 0x07: case 0x0F: case 0x17: case 0x1F: // Ordered
|
||||
return _VFCMPScalarInsert(IR::SizeToOpSize(DstSize), ElementSize, Src1, Src2, FloatCompareOp::ORD, ZeroUpperBits);
|
||||
default:
|
||||
LOGMAN_MSG_A_FMT("Unknown Comparison type: {}", CompType);
|
||||
break;
|
||||
}
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::InsertScalarFCMPOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src[2] needs to be literal");
|
||||
const uint8_t CompType = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
OrderedNode *Result = InsertScalarFCMPOpImpl(Op, DstSize, ElementSize, Op->Dest, Op->Src[0], CompType, false);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::InsertScalarFCMPOp<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::InsertScalarFCMPOp<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::AVXInsertScalarFCMPOp(OpcodeArgs) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[2].IsLiteral(), "Src[2] needs to be literal");
|
||||
const uint8_t CompType = Op->Src[2].Data.Literal.Value;
|
||||
|
||||
const auto DstSize = GetGuestVectorLength();
|
||||
OrderedNode *Result = InsertScalarFCMPOpImpl(Op, DstSize, ElementSize, Op->Src[0], Op->Src[1], CompType, true);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXInsertScalarFCMPOp<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXInsertScalarFCMPOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize) {
|
||||
// In the event of a scalar operation and a vector source, then
|
||||
// we can specify the entire vector length in order to avoid
|
||||
// unnecessary sign extension on the element to be operated on.
|
||||
// In the event of a memory operand, we load the exact element size.
|
||||
const auto SrcSize = Scalar && Op->Src[0].IsGPR() ? 16U : GetSrcSize(Op);
|
||||
const auto OpSize = Scalar ? ElementSize : GetSrcSize(Op);
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto OpSize = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
|
||||
auto ALUOp = _VFSqrt(OpSize, ElementSize, Src);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
|
||||
if (Scalar) {
|
||||
// Insert the lower bits
|
||||
auto Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, ALUOp);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
} else {
|
||||
StoreResult(FPRClass, Op, ALUOp, -1);
|
||||
}
|
||||
StoreResult(FPRClass, Op, ALUOp, -1);
|
||||
}
|
||||
|
||||
template <IROps IROp, size_t ElementSize, bool Scalar>
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs) {
|
||||
VectorUnaryOpImpl(Op, IROp, ElementSize, Scalar);
|
||||
VectorUnaryOpImpl(Op, IROp, ElementSize);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 4, true>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRSQRT, 4, true>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFRECP, 4, true>(OpcodeArgs);
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 1>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VFSQRT, 8, true>(OpcodeArgs);
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 1, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 2, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorUnaryOp<IR::OP_VABS, 4, false>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize, bool Scalar) {
|
||||
void OpDispatchBuilder::AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize) {
|
||||
// In the event of a scalar operation and a vector source, then
|
||||
// we can specify the entire vector length in order to avoid
|
||||
// unnecessary sign extension on the element to be operated on.
|
||||
// In the event of a memory operand, we load the exact element size.
|
||||
const auto SrcSize = Scalar && Op->Src[1].IsGPR() ? 16U : GetSrcSize(Op);
|
||||
const auto OpSize = Scalar ? ElementSize : GetSrcSize(Op);
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto OpSize = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src = [&] {
|
||||
const auto SrcIndex = Scalar ? 1 : 0;
|
||||
return LoadSource_WithOpSize(FPRClass, Op, Op->Src[SrcIndex], SrcSize, Op->Flags, -1);
|
||||
}();
|
||||
OrderedNode *Dest = [&] {
|
||||
const auto& Operand = Scalar ? Op->Src[0] : Op->Dest;
|
||||
return LoadSource_WithOpSize(FPRClass, Op, Operand, DstSize, Op->Flags, -1);
|
||||
}();
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1);
|
||||
|
||||
auto ALUOp = _VFSqrt(OpSize, ElementSize, Src);
|
||||
// Overwrite our IR's op type
|
||||
ALUOp.first->Header.Op = IROp;
|
||||
|
||||
OrderedNode* Result = ALUOp;
|
||||
if (Scalar) {
|
||||
// Insert the lower bits
|
||||
Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, Result);
|
||||
}
|
||||
|
||||
// NOTE: We don't need to clear the upper lanes here, since the
|
||||
// IR ops make use of 128-bit AdvSimd for 128-bit cases,
|
||||
// which, on hardware with SVE, zero-extends as part of
|
||||
// storing into the destination.
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
StoreResult(FPRClass, Op, ALUOp, -1);
|
||||
}
|
||||
|
||||
template <IROps IROp, size_t ElementSize, bool Scalar>
|
||||
template <IROps IROp, size_t ElementSize>
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp(OpcodeArgs) {
|
||||
AVXVectorUnaryOpImpl(Op, IROp, ElementSize, Scalar);
|
||||
AVXVectorUnaryOpImpl(Op, IROp, ElementSize);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 1, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 1>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 2, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 2>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 4, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VABS, 4>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRECP, 4, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRECP, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRECP, 4, true>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 4, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 4, true>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 8, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFSQRT, 8, true>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRSQRT, 4, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRSQRT, 4, true>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXVectorUnaryOp<IR::OP_VFRSQRT, 4>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::VectorUnaryDuplicateOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
@@ -2411,38 +2662,22 @@ void OpDispatchBuilder::Vector_CVT_Float_To_Float<4, 8>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Float<8, 4>(OpcodeArgs);
|
||||
|
||||
template<size_t SrcElementSize, bool Widen>
|
||||
void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
size_t ElementSize = SrcElementSize;
|
||||
// Always 32-bit.
|
||||
size_t ElementSize = 4;
|
||||
size_t DstSize = GetDstSize(Op);
|
||||
if constexpr (Widen) {
|
||||
Src = _VSXTL(DstSize, ElementSize, Src);
|
||||
ElementSize <<= 1;
|
||||
}
|
||||
|
||||
Src = _VSXTL(DstSize, ElementSize, Src);
|
||||
ElementSize <<= 1;
|
||||
|
||||
// Always signed
|
||||
Src = _Vector_SToF(DstSize, ElementSize, Src);
|
||||
|
||||
OrderedNode *Dest{};
|
||||
if constexpr (Widen) {
|
||||
Dest = Src;
|
||||
}
|
||||
else {
|
||||
Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
// Insert the lower bits
|
||||
Dest = _VInsElement(GetDstSize(Op), 8, 0, 0, Dest, Src);
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Dest, -1);
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, true>(OpcodeArgs);
|
||||
|
||||
template<size_t SrcElementSize, bool Narrow, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
// If loading a vector, use the full size, so we don't
|
||||
@@ -2578,7 +2813,7 @@ void OpDispatchBuilder::MOVBetweenGPR_FPR(OpcodeArgs) {
|
||||
}
|
||||
}
|
||||
|
||||
OrderedNode* OpDispatchBuilder::VFCMPOpImpl(OpcodeArgs, size_t ElementSize, bool Scalar,
|
||||
OrderedNode* OpDispatchBuilder::VFCMPOpImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src1, OrderedNode *Src2, uint8_t CompType) {
|
||||
const auto Size = GetSrcSize(Op);
|
||||
|
||||
@@ -2615,44 +2850,35 @@ OrderedNode* OpDispatchBuilder::VFCMPOpImpl(OpcodeArgs, size_t ElementSize, bool
|
||||
break;
|
||||
}
|
||||
|
||||
if (Scalar) {
|
||||
// Insert the lower bits
|
||||
Result = _VInsElement(GetDstSize(Op), ElementSize, 0, 0, Src1, Result);
|
||||
}
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<size_t ElementSize, bool Scalar>
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VFCMPOp(OpcodeArgs) {
|
||||
// No need for zero-extending in the scalar case, since
|
||||
// all we need is an insert at the end of the operation.
|
||||
const auto SrcSize = Scalar && Op->Src[0].IsGPR() ? 16U : GetSrcSize(Op);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1);
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
const uint8_t CompType = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
OrderedNode* Result = VFCMPOpImpl(Op, ElementSize, Scalar, Dest, Src, CompType);
|
||||
OrderedNode* Result = VFCMPOpImpl(Op, ElementSize, Dest, Src, CompType);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::VFCMPOp<4, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::VFCMPOp<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VFCMPOp<4, true>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VFCMPOp<8, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VFCMPOp<8, true>(OpcodeArgs);
|
||||
void OpDispatchBuilder::VFCMPOp<8>(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Scalar>
|
||||
template <size_t ElementSize>
|
||||
void OpDispatchBuilder::AVXVFCMPOp(OpcodeArgs) {
|
||||
// No need for zero-extending in the scalar case, since
|
||||
// all we need is an insert at the end of the operation.
|
||||
const auto SrcSize = Scalar && Op->Src[1].IsGPR() ? 16U : GetSrcSize(Op);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Op->Src[2].IsLiteral(), "Src[2] needs to be literal");
|
||||
@@ -2660,19 +2886,15 @@ void OpDispatchBuilder::AVXVFCMPOp(OpcodeArgs) {
|
||||
|
||||
OrderedNode *Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags, -1);
|
||||
OrderedNode *Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags, -1);
|
||||
OrderedNode *Result = VFCMPOpImpl(Op, ElementSize, Scalar, Src1, Src2, CompType);
|
||||
OrderedNode *Result = VFCMPOpImpl(Op, ElementSize, Src1, Src2, CompType);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVFCMPOp<4, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXVFCMPOp<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVFCMPOp<4, true>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVFCMPOp<8, false>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVFCMPOp<8, true>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXVFCMPOp<8>(OpcodeArgs);
|
||||
|
||||
void OpDispatchBuilder::FXSaveOp(OpcodeArgs) {
|
||||
OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false);
|
||||
@@ -3928,8 +4150,7 @@ template
|
||||
void OpDispatchBuilder::ExtendVectorElements<4, 8, true>(OpcodeArgs);
|
||||
|
||||
OrderedNode* OpDispatchBuilder::VectorRoundImpl(OpcodeArgs, size_t ElementSize,
|
||||
OrderedNode *Src, uint64_t Mode,
|
||||
bool IsScalar) {
|
||||
OrderedNode *Src, uint64_t Mode) {
|
||||
const auto Size = GetDstSize(Op);
|
||||
const uint64_t RoundControlSource = (Mode >> 2) & 1;
|
||||
uint64_t RoundControl = Mode & 0b11;
|
||||
@@ -3947,82 +4168,48 @@ OrderedNode* OpDispatchBuilder::VectorRoundImpl(OpcodeArgs, size_t ElementSize,
|
||||
};
|
||||
|
||||
const auto SourceMode = SourceModes[(RoundControlSource << 2) | RoundControl];
|
||||
const auto OpSize = IsScalar ? ElementSize : Size;
|
||||
return _Vector_FToI(OpSize, ElementSize, Src, SourceMode);
|
||||
return _Vector_FToI(Size, ElementSize, Src, SourceMode);
|
||||
}
|
||||
|
||||
template<size_t ElementSize, bool Scalar>
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VectorRound(OpcodeArgs) {
|
||||
// No need to zero extend the vector in the event we have a
|
||||
// scalar source, especially since it's only inserted into another vector.
|
||||
const auto SrcSize = Scalar && Op->Src[0].IsGPR() ? 16U : GetSrcSize(Op);
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1);
|
||||
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
const uint64_t Mode = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
Src = VectorRoundImpl(Op, ElementSize, Src, Mode, Scalar);
|
||||
Src = VectorRoundImpl(Op, ElementSize, Src, Mode);
|
||||
|
||||
if constexpr (Scalar) {
|
||||
// Insert the lower bits
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1);
|
||||
auto Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, Src);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
} else {
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
}
|
||||
StoreResult(FPRClass, Op, Src, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::VectorRound<4, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::VectorRound<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorRound<8, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::VectorRound<8>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::VectorRound<4, true>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::VectorRound<8, true>(OpcodeArgs);
|
||||
|
||||
template <size_t ElementSize, bool Scalar>
|
||||
template <size_t ElementSize>
|
||||
void OpDispatchBuilder::AVXVectorRound(OpcodeArgs) {
|
||||
const auto GetMode = [&] {
|
||||
if constexpr (Scalar) {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[2].IsLiteral(), "Src2 needs to be literal here");
|
||||
return Op->Src[2].Data.Literal.Value;
|
||||
} else {
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
return Op->Src[1].Data.Literal.Value;
|
||||
}
|
||||
};
|
||||
LOGMAN_THROW_A_FMT(Op->Src[1].IsLiteral(), "Src1 needs to be literal here");
|
||||
const auto Mode = Op->Src[1].Data.Literal.Value;
|
||||
|
||||
// No need to zero extend the vector in the event we have a
|
||||
// scalar source, especially since it's only inserted into another vector.
|
||||
const auto SrcIdx = Scalar ? 1 : 0;
|
||||
const auto SrcSize = Scalar && Op->Src[SrcIdx].IsGPR() ? 16U : GetSrcSize(Op);
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto SrcSize = GetSrcSize(Op);
|
||||
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[SrcIdx], SrcSize, Op->Flags, -1);
|
||||
OrderedNode *Result = VectorRoundImpl(Op, ElementSize, Src, GetMode(), Scalar);
|
||||
|
||||
if constexpr (Scalar) {
|
||||
// Insert the lower bits
|
||||
OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags, -1);
|
||||
Result = _VInsElement(DstSize, ElementSize, 0, 0, Dest, Result);
|
||||
}
|
||||
OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, -1);
|
||||
OrderedNode *Result = VectorRoundImpl(Op, ElementSize, Src, Mode);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorRound<4, false>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXVectorRound<4>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorRound<8, false>(OpcodeArgs);
|
||||
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorRound<4, true>(OpcodeArgs);
|
||||
template
|
||||
void OpDispatchBuilder::AVXVectorRound<8, true>(OpcodeArgs);
|
||||
void OpDispatchBuilder::AVXVectorRound<8>(OpcodeArgs);
|
||||
|
||||
template<size_t ElementSize>
|
||||
void OpDispatchBuilder::VectorBlend(OpcodeArgs) {
|
||||
|
||||
@@ -156,6 +156,7 @@
|
||||
"MemOffsetType": "MemOffsetType",
|
||||
"BreakDefinition": "BreakDefinition",
|
||||
"RoundType": "RoundType",
|
||||
"FloatCompareOp": "FloatCompareOp",
|
||||
"NamedVectorConstant": "FEXCore::IR::NamedVectorConstant",
|
||||
"IndexNamedVectorConstant": "FEXCore::IR::IndexNamedVectorConstant"
|
||||
},
|
||||
@@ -1268,6 +1269,158 @@
|
||||
"DestSize": "4"
|
||||
}
|
||||
},
|
||||
"VectorScalar": {
|
||||
"FPR = VFAddScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'add' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFSubScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'sub' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMulScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'mul' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFDivScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'div' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMinScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'min' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics.",
|
||||
"Additionally matches x86 zero and NaN semantics",
|
||||
"If both source operands are zero, return the second operand (in the case of negative and positive zero)",
|
||||
"If either source operand is NaN then return the second operand."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMaxScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'max' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics.",
|
||||
"Additionally matches x86 zero and NaN semantics",
|
||||
"If both source operands are zero, return the second operand (in the case of negative and positive zero)",
|
||||
"If either source operand is NaN then return the second operand."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFSqrtScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'sqrt' on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFRSqrtScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'rsqrt' on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFRecpScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'recip' on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFToFScalarInsert OpSize:#RegisterSize, u8:#DstElementSize, u8:$SrcElementSize, FPR:$Vector1, FPR:$Vector2, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'cvt' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DstElementSize"
|
||||
},
|
||||
"FPR = VSToFVectorInsert OpSize:#RegisterSize, u8:#DstElementSize, u8:$SrcElementSize, FPR:$Vector1, FPR:$Vector2, i8:$HasTwoElements, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a Vector 'scvt' between Vector1 and Vector2.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics.",
|
||||
"HasTwoElements is slightly different than most of these scalar operations.",
|
||||
"Handles the edge case of cvtpi2ps xmm0, mm0 which is two elements in the lower 64-bits"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DstElementSize"
|
||||
},
|
||||
"FPR = VSToFGPRInsert OpSize:#RegisterSize, u8:#DstElementSize, u8:$SrcElementSize, FPR:$Vector, GPR:$Src, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'cvt' between Vector1 and GPR.",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / DstElementSize"
|
||||
},
|
||||
"FPR = VFToIScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, RoundType:$Round, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar round float to integral on Vector2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Rounding mode determined by argument",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFCMPScalarInsert OpSize:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2, FloatCompareOp:$Op, i1:$ZeroUpperBits": {
|
||||
"Desc": ["Does a scalar 'cmp' between Vector1 and Vecto2, inserting in to Vector1 and storing in to the destination.",
|
||||
"Compare op determined by argument",
|
||||
"Inserting the result in to the lower element of Vector1 and returning the results.",
|
||||
"If ZeroUpperBits is set then in a 256-bit wide operation it will zero the upper 128-bits of the destination.",
|
||||
"For 128-bit operation this matches SSE insert semantics.",
|
||||
"For 256-bit operation with ZeroUpperBits, this matches AVX insert semantics."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
}
|
||||
},
|
||||
"Vector": {
|
||||
"FPR = VMov u8:#RegisterSize, FPR:$Source": {
|
||||
"Desc" : ["Copy vector register",
|
||||
|
||||
@@ -238,6 +238,18 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::FloatCompareOp Arg) {
|
||||
switch (Arg) {
|
||||
case FloatCompareOp::EQ: *out << "FEQ"; break;
|
||||
case FloatCompareOp::LT: *out << "FLT"; break;
|
||||
case FloatCompareOp::LE: *out << "FLE"; break;
|
||||
case FloatCompareOp::UNO: *out << "UNO"; break;
|
||||
case FloatCompareOp::NEQ: *out << "NEQ"; break;
|
||||
case FloatCompareOp::ORD: *out << "ORD"; break;
|
||||
default: *out << "<Unknown OpSize Type>"; break;
|
||||
}
|
||||
}
|
||||
|
||||
static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::BreakDefinition Arg) {
|
||||
*out << "{" << Arg.ErrorRegister << ".";
|
||||
*out << static_cast<uint32_t>(Arg.Signal) << ".";
|
||||
|
||||
@@ -555,6 +555,15 @@ enum OpSize : uint8_t {
|
||||
i256Bit = 32,
|
||||
};
|
||||
|
||||
enum class FloatCompareOp : uint8_t {
|
||||
EQ = 0,
|
||||
LT,
|
||||
LE,
|
||||
UNO,
|
||||
NEQ,
|
||||
ORD,
|
||||
};
|
||||
|
||||
// Converts a size stored as an integer in to an OpSize enum.
|
||||
// This is a nop operation and will be eliminated by the compiler.
|
||||
static inline OpSize SizeToOpSize(uint8_t Size) {
|
||||
@@ -568,6 +577,7 @@ static inline OpSize SizeToOpSize(uint8_t Size) {
|
||||
default: FEX_UNREACHABLE;
|
||||
}
|
||||
}
|
||||
|
||||
#define IROP_ENUM
|
||||
#define IROP_STRUCTS
|
||||
#define IROP_SIZES
|
||||
|
||||
@@ -0,0 +1,124 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"AFP"
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
"roundss xmm0, xmm1, 00000000b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Nearest rounding",
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintn s16, s17"
|
||||
]
|
||||
},
|
||||
"roundss xmm0, xmm1, 00000001b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"-inf rounding",
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintm s16, s17"
|
||||
]
|
||||
},
|
||||
"roundss xmm0, xmm1, 00000010b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"+inf rounding",
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintp s16, s17"
|
||||
]
|
||||
},
|
||||
"roundss xmm0, xmm1, 00000011b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"truncate rounding",
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintz s16, s17"
|
||||
]
|
||||
},
|
||||
"roundss xmm0, xmm1, 00000100b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"host rounding mode rounding",
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frinti s16, s17"
|
||||
]
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000000b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Nearest rounding",
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintn d16, d17"
|
||||
]
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000001b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"-inf rounding",
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintm d16, d17"
|
||||
]
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000010b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"+inf rounding",
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintp d16, d17"
|
||||
]
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000011b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"truncate rounding",
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintz d16, d17"
|
||||
]
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000100b": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"host rounding mode rounding",
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frinti d16, d17"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,35 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"AFP"
|
||||
],
|
||||
"DisabledHostFeatures": []
|
||||
},
|
||||
"Instructions": {
|
||||
"cvtpi2ps xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x4]",
|
||||
"scvtf v16.2s, v2.2s"
|
||||
]
|
||||
},
|
||||
"cvtpi2ps xmm0, mm0": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #752]",
|
||||
"scvtf v16.2s, v2.2s"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,251 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"AFP"
|
||||
],
|
||||
"DisabledHostFeatures": []
|
||||
},
|
||||
"Instructions": {
|
||||
"cvtsi2ss xmm0, eax": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf s16, w4"
|
||||
]
|
||||
},
|
||||
"cvtsi2ss xmm0, dword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr s2, [x4]",
|
||||
"scvtf s16, s2"
|
||||
]
|
||||
},
|
||||
"cvtsi2ss xmm0, qword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr x20, [x4]",
|
||||
"scvtf s16, x20"
|
||||
]
|
||||
},
|
||||
"sqrtss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x51",
|
||||
"ExpectedArm64ASM": [
|
||||
"fsqrt s16, s17"
|
||||
]
|
||||
},
|
||||
"rsqrtss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"FEAT_FPRES could make this more optimal",
|
||||
"0xf3 0x0f 0x52"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmov s0, #0x70 (1.0000)",
|
||||
"fsqrt s1, s17",
|
||||
"fdiv s16, s0, s1"
|
||||
]
|
||||
},
|
||||
"rcpss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"FEAT_FPRES could make this more optimal",
|
||||
"0xf3 0x0f 0x53"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmov s0, #0x70 (1.0000)",
|
||||
"fdiv s16, s0, s17"
|
||||
]
|
||||
},
|
||||
"addss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x58"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fadd s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"mulss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x59"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmul s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"cvtss2sd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x5a",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcvt d16, s17"
|
||||
]
|
||||
},
|
||||
"cvtss2sd xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x5a",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x4]",
|
||||
"fcvt d16, s2"
|
||||
]
|
||||
},
|
||||
"subss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5c"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsub s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"minss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5d"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmin s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"divss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5e"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fdiv s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"maxss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5f"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmax s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 0": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt s16, s17, s16"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 2": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s16, s17, s16"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 3": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s0, s16, s17",
|
||||
"fcmgt s1, s17, s16",
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"ptrue p0.s, vl1",
|
||||
"mov z16.s, p0/m, z0.s"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 4": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq s0, s16, s17",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"ptrue p0.s, vl1",
|
||||
"mov z16.s, p0/m, z0.s"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 5": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt s2, s17, s16",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 6": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s2, s17, s16",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 7": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s0, s16, s17",
|
||||
"fcmgt s1, s17, s16",
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"ptrue p0.s, vl1",
|
||||
"mov z16.s, p0/m, z0.s"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,242 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"AFP"
|
||||
],
|
||||
"DisabledHostFeatures": []
|
||||
},
|
||||
"Instructions": {
|
||||
"cvtsi2sd xmm0, eax": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf d16, w4"
|
||||
]
|
||||
},
|
||||
"cvtsi2sd xmm0, dword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr w20, [x4]",
|
||||
"scvtf d16, w20"
|
||||
]
|
||||
},
|
||||
"cvtsi2sd xmm0, rax": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf d16, x4"
|
||||
]
|
||||
},
|
||||
"cvtsi2sd xmm0, qword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x4]",
|
||||
"scvtf d16, d2"
|
||||
]
|
||||
},
|
||||
"sqrtsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x51"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsqrt d16, d17"
|
||||
]
|
||||
},
|
||||
"addsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x58"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fadd d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"mulsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x59"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmul d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"cvtsd2ss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcvt s16, d17"
|
||||
]
|
||||
},
|
||||
"cvtsd2ss xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x4]",
|
||||
"fcvt s16, d2"
|
||||
]
|
||||
},
|
||||
"subsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5c"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsub d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"minsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5d"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmin d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"divsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5e"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fdiv d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"maxsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5f"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmax d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 0": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt d16, d17, d16"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 2": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d16, d17, d16"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 3": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d0, d16, d17",
|
||||
"fcmgt d1, d17, d16",
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"ptrue p0.d, vl1",
|
||||
"mov z16.d, p0/m, z0.d"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 4": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq d0, d16, d17",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"ptrue p0.d, vl1",
|
||||
"mov z16.d, p0/m, z0.d"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 5": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt d2, d17, d16",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 6": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d2, d17, d16",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 7": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d0, d16, d17",
|
||||
"fcmgt d1, d17, d16",
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"ptrue p0.d, vl1",
|
||||
"mov z16.d, p0/m, z0.d"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,36 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"AFP"
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
"cvtpi2ps xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x4]",
|
||||
"scvtf v16.2s, v2.2s"
|
||||
]
|
||||
},
|
||||
"cvtpi2ps xmm0, mm0": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #752]",
|
||||
"scvtf v16.2s, v2.2s"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,249 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"AFP"
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
"cvtsi2ss xmm0, eax": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf s16, w4"
|
||||
]
|
||||
},
|
||||
"cvtsi2ss xmm0, dword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr s2, [x4]",
|
||||
"scvtf s16, s2"
|
||||
]
|
||||
},
|
||||
"cvtsi2ss xmm0, qword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr x20, [x4]",
|
||||
"scvtf s16, x20"
|
||||
]
|
||||
},
|
||||
"sqrtss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x51",
|
||||
"ExpectedArm64ASM": [
|
||||
"fsqrt s16, s17"
|
||||
]
|
||||
},
|
||||
"rsqrtss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"FEAT_FPRES could make this more optimal",
|
||||
"0xf3 0x0f 0x52"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmov s0, #0x70 (1.0000)",
|
||||
"fsqrt s1, s17",
|
||||
"fdiv s16, s0, s1"
|
||||
]
|
||||
},
|
||||
"rcpss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"FEAT_FPRES could make this more optimal",
|
||||
"0xf3 0x0f 0x53"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmov s0, #0x70 (1.0000)",
|
||||
"fdiv s16, s0, s17"
|
||||
]
|
||||
},
|
||||
"addss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x58"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fadd s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"mulss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x59"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmul s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"cvtss2sd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x5a",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcvt d16, s17"
|
||||
]
|
||||
},
|
||||
"cvtss2sd xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x5a",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x4]",
|
||||
"fcvt d16, s2"
|
||||
]
|
||||
},
|
||||
"subss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5c"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsub s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"minss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5d"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmin s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"divss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5e"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fdiv s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"maxss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x5f"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmax s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 0": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq s16, s16, s17"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt s16, s17, s16"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 2": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s16, s17, s16"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 3": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s0, s16, s17",
|
||||
"fcmgt s1, s17, s16",
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 4": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq s0, s16, s17",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 5": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt s2, s17, s16",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 6": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s2, s17, s16",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 7": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s0, s16, s17",
|
||||
"fcmgt s1, s17, s16",
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,240 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"AFP"
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
"cvtsi2sd xmm0, eax": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf d16, w4"
|
||||
]
|
||||
},
|
||||
"cvtsi2sd xmm0, dword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr w20, [x4]",
|
||||
"scvtf d16, w20"
|
||||
]
|
||||
},
|
||||
"cvtsi2sd xmm0, rax": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf d16, x4"
|
||||
]
|
||||
},
|
||||
"cvtsi2sd xmm0, qword [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x4]",
|
||||
"scvtf d16, d2"
|
||||
]
|
||||
},
|
||||
"sqrtsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x51"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsqrt d16, d17"
|
||||
]
|
||||
},
|
||||
"addsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x58"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fadd d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"mulsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x59"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmul d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"cvtsd2ss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcvt s16, d17"
|
||||
]
|
||||
},
|
||||
"cvtsd2ss xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x4]",
|
||||
"fcvt s16, d2"
|
||||
]
|
||||
},
|
||||
"subsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5c"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsub d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"minsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5d"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmin d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"divsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5e"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fdiv d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"maxsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0x5f"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmax d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 0": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq d16, d16, d17"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt d16, d17, d16"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 2": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d16, d17, d16"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 3": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d0, d16, d17",
|
||||
"fcmgt d1, d17, d16",
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 4": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq d0, d16, d17",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 5": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt d2, d17, d16",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 6": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d2, d17, d16",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 7": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d0, d16, d17",
|
||||
"fcmgt d1, d17, d16",
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,440 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"AFP"
|
||||
],
|
||||
"DisabledHostFeatures": []
|
||||
},
|
||||
"Instructions": {
|
||||
"vsqrtss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x51 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fsqrt s16, s18"
|
||||
]
|
||||
},
|
||||
"vsqrtsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x51 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fsqrt d16, d18"
|
||||
]
|
||||
},
|
||||
"vrsqrtss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"FEAT_FPRES could make this more optimal",
|
||||
"Map 1 0b10 0x52 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fmov s0, #0x70 (1.0000)",
|
||||
"fsqrt s1, s18",
|
||||
"fdiv s16, s0, s1"
|
||||
]
|
||||
},
|
||||
"vrcpss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"FEAT_FPRES could make this more optimal",
|
||||
"Map 1 0b10 0x53 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fmov s0, #0x70 (1.0000)",
|
||||
"fdiv s16, s0, s18"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x00": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmeq s16, s17, s18"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x01": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmgt s16, s18, s17"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x02": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmge s16, s18, s17"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x03": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmge s0, s17, s18",
|
||||
"fcmgt s1, s18, s17",
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x04": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmeq s0, s17, s18",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x05": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt s2, s18, s17",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v16.16b, v17.16b",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x06": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s2, s18, s17",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v16.16b, v17.16b",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x07": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmge s0, s17, s18",
|
||||
"fcmgt s1, s18, s17",
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x00": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmeq d16, d17, d18"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x01": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmgt d16, d18, d17"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x02": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmge d16, d18, d17"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x03": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmge d0, d17, d18",
|
||||
"fcmgt d1, d18, d17",
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x04": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmeq d0, d17, d18",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x05": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt d2, d18, d17",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v16.16b, v17.16b",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x06": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d2, d18, d17",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v16.16b, v17.16b",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x07": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmge d0, d17, d18",
|
||||
"fcmgt d1, d18, d17",
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vcvtsi2ss xmm0, xmm1, eax": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x2A 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"scvtf s16, w4"
|
||||
]
|
||||
},
|
||||
"vcvtsi2ss xmm0, xmm1, rax": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x2A 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"scvtf s16, x4"
|
||||
]
|
||||
},
|
||||
"vcvtsi2sd xmm0, xmm1, eax": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x2A 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"scvtf d16, w4"
|
||||
]
|
||||
},
|
||||
"vcvtsi2sd xmm0, xmm1, rax": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x2A 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"scvtf d16, x4"
|
||||
]
|
||||
},
|
||||
"vmulss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x59 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fmul s16, s17, s18"
|
||||
]
|
||||
},
|
||||
"vmulsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x59 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fmul d16, d17, d18"
|
||||
]
|
||||
},
|
||||
"vcvtss2sd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcvt d16, s18"
|
||||
]
|
||||
},
|
||||
"vcvtsd2ss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcvt s16, d18"
|
||||
]
|
||||
},
|
||||
"vsubss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5c 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fsub s16, s17, s18"
|
||||
]
|
||||
},
|
||||
"vsubsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5c 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fsub d16, d17, d18"
|
||||
]
|
||||
},
|
||||
"vminss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5d 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fmin s16, s17, s18"
|
||||
]
|
||||
},
|
||||
"vminsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5d 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fmin d16, d17, d18"
|
||||
]
|
||||
},
|
||||
"vdivss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5e 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fdiv s16, s17, s18"
|
||||
]
|
||||
},
|
||||
"vdivsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5e 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fdiv d16, d17, d18"
|
||||
]
|
||||
},
|
||||
"vmaxss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5f 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fmax s16, s17, s18"
|
||||
]
|
||||
},
|
||||
"vmaxsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5f 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fmax d16, d17, d18"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,133 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"AFP"
|
||||
],
|
||||
"DisabledHostFeatures": []
|
||||
},
|
||||
"Instructions": {
|
||||
"vroundss xmm0, xmm1, 00000000b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"nearest rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintn s16, s16"
|
||||
]
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000001b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"-inf rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintm s16, s16"
|
||||
]
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000010b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"+inf rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintp s16, s16"
|
||||
]
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000011b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"truncate rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintz s16, s16"
|
||||
]
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000100b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"host mode rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v16.16b",
|
||||
"frinti s16, s16"
|
||||
]
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000000b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"nearest rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintn d16, d16"
|
||||
]
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000001b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"-inf rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintm d16, d16"
|
||||
]
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000010b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"+inf rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintp d16, d16"
|
||||
]
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000011b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"truncate rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintz d16, d16"
|
||||
]
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000100b": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"host mode rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v16.16b",
|
||||
"frinti d16, d16"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -4,7 +4,8 @@
|
||||
"EnabledHostFeatures": [],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
"SVE256",
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Comment": [
|
||||
@@ -167,8 +168,8 @@
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintn s2, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"frintn s0, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"roundss xmm0, xmm1, 00000001b": {
|
||||
@@ -179,8 +180,8 @@
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintm s2, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"frintm s0, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"roundss xmm0, xmm1, 00000010b": {
|
||||
@@ -191,8 +192,8 @@
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintp s2, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"frintp s0, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"roundss xmm0, xmm1, 00000011b": {
|
||||
@@ -203,8 +204,8 @@
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintz s2, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"frintz s0, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"roundss xmm0, xmm1, 00000100b": {
|
||||
@@ -215,8 +216,8 @@
|
||||
"0x66 0x0f 0x3a 0x0a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frinti s2, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"frinti s0, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000000b": {
|
||||
@@ -227,8 +228,8 @@
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintn d2, d17",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"frintn d0, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000001b": {
|
||||
@@ -239,8 +240,8 @@
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintm d2, d17",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"frintm d0, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000010b": {
|
||||
@@ -251,8 +252,8 @@
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintp d2, d17",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"frintp d0, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000011b": {
|
||||
@@ -263,8 +264,8 @@
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintz d2, d17",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"frintz d0, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"roundsd xmm0, xmm1, 00000100b": {
|
||||
@@ -275,8 +276,8 @@
|
||||
"0x66 0x0f 0x3a 0x0b"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frinti d2, d17",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"frinti d0, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"blendps xmm0, xmm1, 0000b": {
|
||||
|
||||
@@ -6,7 +6,8 @@
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
"SVE256",
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
@@ -18,8 +19,8 @@
|
||||
"0xf3 0x0f 0x52"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frsqrte s2, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"frecpe s0, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"rcpss xmm0, xmm1": {
|
||||
@@ -30,8 +31,8 @@
|
||||
"0xf3 0x0f 0x53"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frecpe s2, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"frsqrte s0, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,37 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"RPRES",
|
||||
"AFP"
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
"rsqrtss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"AFP can make this more optimal",
|
||||
"0xf3 0x0f 0x52"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frecpe s16, s17"
|
||||
]
|
||||
},
|
||||
"rcpss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"AFP can make this more optimal",
|
||||
"0xf3 0x0f 0x53"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frsqrte s16, s17"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -6,7 +6,9 @@
|
||||
"SVE256",
|
||||
"RPRES"
|
||||
],
|
||||
"DisabledHostFeatures": []
|
||||
"DisabledHostFeatures": [
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
"vrsqrtps xmm0, xmm1": {
|
||||
@@ -31,18 +33,16 @@
|
||||
]
|
||||
},
|
||||
"vrsqrtss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"AFP can make this more optimal",
|
||||
"Map 1 0b10 0x52 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frsqrte s2, s18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"frecpe s0, s18",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vrcpps xmm0, xmm1": {
|
||||
@@ -67,17 +67,15 @@
|
||||
]
|
||||
},
|
||||
"vrcpss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x53 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frecpe s2, s18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"frsqrte s0, s18",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
}
|
||||
}
|
||||
|
||||
@@ -0,0 +1,79 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"RPRES",
|
||||
"AFP"
|
||||
],
|
||||
"DisabledHostFeatures": []
|
||||
},
|
||||
"Instructions": {
|
||||
"vrsqrtps xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 1 0b00 0x52 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frsqrte v2.4s, v17.4s",
|
||||
"mov v16.16b, v2.16b"
|
||||
]
|
||||
},
|
||||
"vrsqrtps ymm0, ymm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b00 0x52 256-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frsqrte z16.s, z17.s"
|
||||
]
|
||||
},
|
||||
"vrsqrtss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"AFP can make this more optimal",
|
||||
"Map 1 0b10 0x52 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"frecpe s16, s18"
|
||||
]
|
||||
},
|
||||
"vrcpps xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 1 0b00 0x53 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frecpe v2.4s, v17.4s",
|
||||
"mov v16.16b, v2.16b"
|
||||
]
|
||||
},
|
||||
"vrcpps ymm0, ymm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b00 0x53 256-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frecpe z16.s, z17.s"
|
||||
]
|
||||
},
|
||||
"vrcpss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x53 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"frsqrte s16, s18"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,11 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
]
|
||||
},
|
||||
"Instructions": {}
|
||||
}
|
||||
@@ -5,7 +5,8 @@
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"RPRES"
|
||||
"RPRES",
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Comment": [
|
||||
@@ -130,26 +131,24 @@
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x4]",
|
||||
"scvtf v2.4s, v2.4s",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"scvtf v0.2s, v2.2s",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cvtpi2ps xmm0, mm0": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #752]",
|
||||
"scvtf v2.4s, v2.4s",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"scvtf v0.2s, v2.2s",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"movntps [rax], xmm0": {
|
||||
|
||||
@@ -6,7 +6,8 @@
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
"SVE256",
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
|
||||
@@ -5,7 +5,8 @@
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"RPRES"
|
||||
"RPRES",
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
@@ -71,25 +72,23 @@
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf s2, w4",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"scvtf s0, w4",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"cvtsi2ss xmm0, dword [rax]": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr s2, [x4]",
|
||||
"scvtf s2, s2",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"scvtf s0, s2",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"cvtsi2ss xmm0, rax": {
|
||||
@@ -97,21 +96,20 @@
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x2a",
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf s2, x4",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"scvtf s0, x4",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"cvtsi2ss xmm0, qword [rax]": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr x20, [x4]",
|
||||
"scvtf s2, x20",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"scvtf s0, x20",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"movntss [rax], xmm0": {
|
||||
@@ -196,60 +194,58 @@
|
||||
},
|
||||
"sqrtss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x51",
|
||||
"ExpectedArm64ASM": [
|
||||
"fsqrt s2, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fsqrt s0, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"rsqrtss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x52"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmov s0, #0x70 (1.0000)",
|
||||
"fsqrt s1, s17",
|
||||
"fdiv s2, s0, s1",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fdiv s0, s0, s1",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"rcpss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0xf3 0x0f 0x53"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmov s0, #0x70 (1.0000)",
|
||||
"fdiv s2, s0, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fdiv s0, s0, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"addss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0x58"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fadd s2, s16, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fadd s0, s16, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"mulss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0x59"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmul s2, s16, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fmul s0, s16, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"cvtss2sd xmm0, xmm1": {
|
||||
@@ -257,8 +253,8 @@
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x5a",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcvt d2, s17",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"fcvt d0, s17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cvtss2sd xmm0, [rax]": {
|
||||
@@ -266,9 +262,9 @@
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0xf3 0x0f 0x5a",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr s2, [x4]",
|
||||
"fcvt d2, s2",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"ldr d2, [x4]",
|
||||
"fcvt d0, s2",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cvttps2dq xmm0, xmm1": {
|
||||
@@ -283,50 +279,46 @@
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0x5c"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsub s2, s16, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fsub s0, s16, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"minss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0x5d"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"fcsel s2, s16, s17, mi",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fcsel s0, s16, s17, mi",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"divss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0x5e"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fdiv s2, s16, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fdiv s0, s16, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"maxss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0x5f"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"fcsel s2, s17, s16, mi",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fcsel s0, s17, s16, mi",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"movdqu xmm0, xmm1": {
|
||||
@@ -588,73 +580,67 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 0": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq s2, s16, s17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fcmeq s0, s16, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt s2, s17, s16",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fcmgt s0, s17, s16",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s2, s17, s16",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fcmge s0, s17, s16",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 3": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s0, s16, s17",
|
||||
"fcmgt s1, s17, s16",
|
||||
"orr v2.8b, v0.8b, v1.8b",
|
||||
"mvn v2.8b, v2.8b",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 4": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq s2, s16, s17",
|
||||
"mvn v2.8b, v2.8b",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fcmeq s0, s16, s17",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"cmpss xmm0, xmm1, 5": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
@@ -665,9 +651,8 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 6": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
@@ -678,16 +663,15 @@
|
||||
},
|
||||
"cmpss xmm0, xmm1, 7": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf3 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s0, s16, s17",
|
||||
"fcmgt s1, s17, s16",
|
||||
"orr v2.8b, v0.8b, v1.8b",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"movq2dq xmm0, mm0": {
|
||||
|
||||
@@ -5,7 +5,8 @@
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"FCMA"
|
||||
"FCMA",
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
@@ -54,50 +55,46 @@
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf d2, w4",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"scvtf d0, w4",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cvtsi2sd xmm0, dword [rax]": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr w20, [x4]",
|
||||
"scvtf d2, w20",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"scvtf d0, w20",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cvtsi2sd xmm0, rax": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf d2, x4",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"scvtf d0, x4",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cvtsi2sd xmm0, qword [rax]": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x2a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x4]",
|
||||
"scvtf d2, d2",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"scvtf d0, d2",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"movntsd [rax], xmm0": {
|
||||
@@ -182,113 +179,104 @@
|
||||
},
|
||||
"sqrtsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x51"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsqrt d2, d17",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"fsqrt d0, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"addsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x58"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fadd d2, d16, d17",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"fadd d0, d16, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"mulsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x59"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmul d2, d16, d17",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"fmul d0, d16, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cvtsd2ss xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x5a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcvt s2, d17",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"fcvt s0, d17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"cvtsd2ss xmm0, [rax]": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x5a"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x4]",
|
||||
"fcvt s2, d2",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
"ldr q2, [x4]",
|
||||
"fcvt s0, d2",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"subsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x5c"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsub d2, d16, d17",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"fsub d0, d16, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"minsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x5d"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"fcsel d2, d16, d17, mi",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"fcsel d0, d16, d17, mi",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"divsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x5e"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fdiv d2, d16, d17",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"fdiv d0, d16, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"maxsd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0x5f"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"fcsel d2, d17, d16, mi",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"fcsel d0, d17, d16, mi",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"pshuflw xmm0, xmm1, 0": {
|
||||
@@ -408,73 +396,67 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 0": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq d2, d16, d17",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"fcmeq d0, d16, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt d2, d17, d16",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"fcmgt d0, d17, d16",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 2": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d2, d17, d16",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"fcmge d0, d17, d16",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 3": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d0, d16, d17",
|
||||
"fcmgt d1, d17, d16",
|
||||
"orr v2.8b, v0.8b, v1.8b",
|
||||
"mvn v2.8b, v2.8b",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 4": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq d2, d16, d17",
|
||||
"mvn v2.8b, v2.8b",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"fcmeq d0, d16, d17",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 5": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
@@ -485,9 +467,8 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 6": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
@@ -498,16 +479,15 @@
|
||||
},
|
||||
"cmpsd xmm0, xmm1, 7": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"With AFP mode FEX can remove an insert after the operation.",
|
||||
"0xf2 0x0f 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d0, d16, d17",
|
||||
"fcmgt d1, d17, d16",
|
||||
"orr v2.8b, v0.8b, v1.8b",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"addsubps xmm0, xmm1": {
|
||||
|
||||
@@ -0,0 +1,27 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
"push ax, bx": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x50",
|
||||
"x86Insts": [
|
||||
"push ax",
|
||||
"push bx"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"uxth w20, w4",
|
||||
"strh w20, [x8, #-2]!",
|
||||
"uxth w20, w7",
|
||||
"strh w20, [x8, #-2]!"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -7,7 +7,8 @@
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"FCMA",
|
||||
"RPRES"
|
||||
"RPRES",
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
@@ -672,33 +673,27 @@
|
||||
]
|
||||
},
|
||||
"vsqrtss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Insert in to first element could be more optimal, which is the common case.",
|
||||
"Map 1 0b10 0x51 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsqrt s2, s18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fsqrt s0, s18",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vsqrtsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Insert in to first element could be more optimal, which is the common case.",
|
||||
"Map 1 0b11 0x51 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsqrt d2, d18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fsqrt d0, d18",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vrsqrtps xmm0, xmm1": {
|
||||
@@ -727,19 +722,17 @@
|
||||
]
|
||||
},
|
||||
"vrsqrtss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x52 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fmov s0, #0x70 (1.0000)",
|
||||
"fsqrt s1, s18",
|
||||
"fdiv s2, s0, s1",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"fdiv s0, s0, s1",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vrcpps xmm0, xmm1": {
|
||||
@@ -767,18 +760,16 @@
|
||||
]
|
||||
},
|
||||
"vrcpss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x53 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fmov s0, #0x70 (1.0000)",
|
||||
"fdiv s2, s0, s18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"fdiv s0, s0, s18",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vandps xmm0, xmm1": {
|
||||
@@ -2397,243 +2388,211 @@
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x00": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq s2, s17, s18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmeq s0, s17, s18",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x01": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt s2, s18, s17",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmgt s0, s18, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x02": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s2, s18, s17",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmge s0, s18, s17",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x03": {
|
||||
"ExpectedInstructionCount": 8,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmge s0, s17, s18",
|
||||
"fcmgt s1, s18, s17",
|
||||
"orr v2.8b, v0.8b, v1.8b",
|
||||
"mvn v2.8b, v2.8b",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x04": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq s2, s17, s18",
|
||||
"mvn v2.8b, v2.8b",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmeq s0, s17, s18",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x05": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt s2, s18, s17",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x06": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge s2, s18, s17",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"mov v16.s[0], v2.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpss xmm0, xmm1, xmm2, 0x07": {
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmge s0, s17, s18",
|
||||
"fcmgt s1, s18, s17",
|
||||
"orr v2.8b, v0.8b, v1.8b",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x00": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq d2, d17, d18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmeq d0, d17, d18",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x01": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt d2, d18, d17",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmgt d0, d18, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x02": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d2, d18, d17",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmge d0, d18, d17",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x03": {
|
||||
"ExpectedInstructionCount": 8,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmge d0, d17, d18",
|
||||
"fcmgt d1, d18, d17",
|
||||
"orr v2.8b, v0.8b, v1.8b",
|
||||
"mvn v2.8b, v2.8b",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x04": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmeq d2, d17, d18",
|
||||
"mvn v2.8b, v2.8b",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmeq d0, d17, d18",
|
||||
"mvn v0.8b, v0.8b",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x05": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmgt d2, d18, d17",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x06": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmge d2, d18, d17",
|
||||
"mvn v2.16b, v2.16b",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"mov v16.d[0], v2.d[0]"
|
||||
]
|
||||
},
|
||||
"vcmpsd xmm0, xmm1, xmm2, 0x07": {
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0xC2 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmge d0, d17, d18",
|
||||
"fcmgt d1, d18, d17",
|
||||
"orr v2.8b, v0.8b, v1.8b",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"orr v0.8b, v0.8b, v1.8b",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vpinsrw xmm0, xmm1, eax, 000b": {
|
||||
@@ -3153,59 +3112,51 @@
|
||||
]
|
||||
},
|
||||
"vcvtsi2ss xmm0, xmm1, eax": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x2A 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf s2, w4",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"scvtf s0, w4",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vcvtsi2ss xmm0, xmm1, rax": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x2A 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf s2, x4",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"scvtf s0, x4",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vcvtsi2sd xmm0, xmm1, eax": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x2A 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf d2, w4",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"scvtf d0, w4",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vcvtsi2sd xmm0, xmm1, rax": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x2A 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"scvtf d2, x4",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"scvtf d0, x4",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vmovntps [rax], xmm0": {
|
||||
@@ -3496,31 +3447,27 @@
|
||||
]
|
||||
},
|
||||
"vaddss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x58 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fadd s2, s17, s18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fadd s0, s17, s18",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vaddsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x58 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fadd d2, d17, d18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fadd d0, d17, d18",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vmulps xmm0, xmm1, xmm2": {
|
||||
@@ -3566,31 +3513,27 @@
|
||||
]
|
||||
},
|
||||
"vmulss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x59 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmul s2, s17, s18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fmul s0, s17, s18",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vmulsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x59 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fmul d2, d17, d18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fmul d0, d17, d18",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vcvtps2pd xmm0, xmm1": {
|
||||
@@ -3616,31 +3559,27 @@
|
||||
]
|
||||
},
|
||||
"vcvtss2sd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcvt d2, s18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcvt d0, s18",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vcvtsd2ss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcvt s2, d18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcvt s0, d18",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vcvtdq2ps xmm0, xmm1": {
|
||||
@@ -3751,31 +3690,27 @@
|
||||
]
|
||||
},
|
||||
"vsubss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5c 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsub s2, s17, s18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fsub s0, s17, s18",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vsubsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5c 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fsub d2, d17, d18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fsub d0, d17, d18",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vminps xmm0, xmm1, xmm2": {
|
||||
@@ -3833,33 +3768,29 @@
|
||||
]
|
||||
},
|
||||
"vminss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5d 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmp s17, s18",
|
||||
"fcsel s2, s17, s18, mi",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"fcsel s0, s17, s18, mi",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vminsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5d 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmp d17, d18",
|
||||
"fcsel d2, d17, d18, mi",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"fcsel d0, d17, d18, mi",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vdivps xmm0, xmm1, xmm2": {
|
||||
@@ -3909,31 +3840,27 @@
|
||||
]
|
||||
},
|
||||
"vdivss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5e 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fdiv s2, s17, s18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fdiv s0, s17, s18",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vdivsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5e 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fdiv d2, d17, d18",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v17.16b",
|
||||
"fdiv d0, d17, d18",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vmaxps xmm0, xmm1, xmm2": {
|
||||
@@ -3989,33 +3916,29 @@
|
||||
]
|
||||
},
|
||||
"vmaxss xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b10 0x5f 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmp s17, s18",
|
||||
"fcsel s2, s18, s17, mi",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"fcsel s0, s18, s17, mi",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vmaxsd xmm0, xmm1, xmm2": {
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 1 0b11 0x5f 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v16.16b, v17.16b",
|
||||
"fcmp d17, d18",
|
||||
"fcsel d2, d18, d17, mi",
|
||||
"mov v0.16b, v17.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"fcsel d0, d18, d17, mi",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vpunpckhbw xmm0, xmm1, xmm2": {
|
||||
|
||||
@@ -4,7 +4,9 @@
|
||||
"EnabledHostFeatures": [
|
||||
"SVE256"
|
||||
],
|
||||
"DisabledHostFeatures": []
|
||||
"DisabledHostFeatures": [
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
"vpshufb xmm0, xmm1, xmm2": {
|
||||
|
||||
@@ -4,7 +4,9 @@
|
||||
"EnabledHostFeatures": [
|
||||
"SVE256"
|
||||
],
|
||||
"DisabledHostFeatures": []
|
||||
"DisabledHostFeatures": [
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
"vpermq ymm0, ymm1, 00000000b": {
|
||||
@@ -1745,153 +1747,133 @@
|
||||
]
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000000b": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"nearest rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintn s2, s17",
|
||||
"mov v0.16b, v16.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintn s0, s16",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000001b": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"-inf rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintm s2, s17",
|
||||
"mov v0.16b, v16.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintm s0, s16",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000010b": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"+inf rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintp s2, s17",
|
||||
"mov v0.16b, v16.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintp s0, s16",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000011b": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"truncate rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintz s2, s17",
|
||||
"mov v0.16b, v16.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintz s0, s16",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vroundss xmm0, xmm1, 00000100b": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"host mode rounding",
|
||||
"Map 3 0b01 0x0a 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frinti s2, s17",
|
||||
"mov v0.16b, v16.16b",
|
||||
"mov v0.s[0], v2.s[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v16.16b",
|
||||
"frinti s0, s16",
|
||||
"mov v16.s[0], v0.s[0]"
|
||||
]
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000000b": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"nearest rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintn d2, d17",
|
||||
"mov v0.16b, v16.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintn d0, d16",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000001b": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"-inf rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintm d2, d17",
|
||||
"mov v0.16b, v16.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintm d0, d16",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000010b": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"+inf rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintp d2, d17",
|
||||
"mov v0.16b, v16.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintp d0, d16",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000011b": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"truncate rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frintz d2, d17",
|
||||
"mov v0.16b, v16.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v16.16b",
|
||||
"frintz d0, d16",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vroundsd xmm0, xmm1, 00000100b": {
|
||||
"ExpectedInstructionCount": 5,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"host mode rounding",
|
||||
"Map 3 0b01 0x0b 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"frinti d2, d17",
|
||||
"mov v0.16b, v16.16b",
|
||||
"mov v0.d[0], v2.d[0]",
|
||||
"mov v2.16b, v0.16b",
|
||||
"mov v16.16b, v2.16b"
|
||||
"mov v16.16b, v16.16b",
|
||||
"frinti d0, d16",
|
||||
"mov v16.d[0], v0.d[0]"
|
||||
]
|
||||
},
|
||||
"vblendps xmm0, xmm1, xmm2, 0000b": {
|
||||
|
||||
@@ -5,7 +5,8 @@
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"CSSC"
|
||||
"CSSC",
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
|
||||
@@ -7,7 +7,8 @@
|
||||
"EnabledHostFeatures": [],
|
||||
"DisabledHostFeatures": [
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
"SVE256",
|
||||
"AFP"
|
||||
]
|
||||
},
|
||||
"Instructions": {
|
||||
|
||||
Reference in new issue
Block a user