mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 08:00:15 +02:00
AVX-VNNI
This commit is contained in:
1 parent
794a11833d
commit
89a13cd5fd
28 files changed
+1690
-44
No files matched your search
@@ -91,7 +91,11 @@
|
||||
"ENABLESSE4A": "enablesse4a",
|
||||
"DISABLESSE4A": "disablesse4a",
|
||||
"ENABLEMOPS": "enablemops",
|
||||
"DISABLEMOPS": "disablemops"
|
||||
"DISABLEMOPS": "disablemops",
|
||||
"ENABLEI8MM": "enablei8mm",
|
||||
"DISABLEI8MM": "disablei8mm",
|
||||
"ENABLEDOTPROD": "enabledotprod",
|
||||
"DISABLEDOTPROD": "disabledotprod"
|
||||
},
|
||||
"Desc": [
|
||||
"Allows controlling of the CPU features in the JIT.",
|
||||
@@ -116,7 +120,9 @@
|
||||
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
|
||||
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
|
||||
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it",
|
||||
"\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it"
|
||||
"\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it",
|
||||
"\t{enable,disable}i8mm: Will force enable or disable i8mm even if the host doesn't support it",
|
||||
"\t{enable,disable}dotprod: Will force enable or disable dotprod even if the host doesn't support it"
|
||||
]
|
||||
},
|
||||
"SmallTSCScale": {
|
||||
|
||||
@@ -648,6 +648,12 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h(uint32_t Leaf) const {
|
||||
|
||||
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
FEXCore::CPUID::FunctionResults Res {};
|
||||
|
||||
// AVX-VNNI is only advertised when the CPU supports I8MM or Dot Product.
|
||||
// Without these features the implementation is so slow that it is likely
|
||||
// to harm performance.
|
||||
const uint32_t SupportsAVXVNNI = SupportsAVX() && (CTX->HostFeatures.SupportsI8MM || CTX->HostFeatures.SupportsDotProd);
|
||||
|
||||
if (Leaf == 0) {
|
||||
#ifndef _WIN32
|
||||
constexpr uint32_t SUPPORTS_RDPID = 1;
|
||||
@@ -665,7 +671,9 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
const uint32_t SupportsWFXT = CTX->HostFeatures.SupportsWFXT;
|
||||
|
||||
// Number of subfunctions
|
||||
Res.eax = 0x0;
|
||||
// TODO: For now, subfunction 1 only exposes AVX-VNNI so we make it conditional
|
||||
// on AVX-VNNI support. We should revisit this if/when we add more to this leaf.
|
||||
Res.eax = SupportsAVXVNNI;
|
||||
Res.ebx = (1 << 0) | // FS/GS support
|
||||
(0 << 1) | // TSC adjust MSR
|
||||
(0 << 2) | // SGX
|
||||
@@ -765,38 +773,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
|
||||
(0 << 30) | // Arch capabilities - MSR module specific
|
||||
(0 << 31); // SSBD - Speculative Store Bypass Disable
|
||||
} else if (Leaf == 1) {
|
||||
Res.eax = (0U << 0) | // SHA512
|
||||
(0U << 1) | // SM3
|
||||
(0U << 2) | // SM4
|
||||
(0U << 3) | // RAO_INT
|
||||
(0U << 4) | // AVX_VNNI
|
||||
(0U << 5) | // AVX512_BF16
|
||||
(0U << 6) | // LASS (Linear Address Space Separation)
|
||||
(0U << 7) | // CMPCCXADD
|
||||
(0U << 8) | // ARCH_PERFMON_EXT
|
||||
(0U << 9) | // Reserved
|
||||
(0U << 10) | // FAST_REP_MOVSB
|
||||
(0U << 11) | // FAST_REP_STOSB
|
||||
(0U << 12) | // FAST_REP_CMPSB_SCASB
|
||||
(0U << 13) | // Reserved
|
||||
(0U << 14) | // Reserved
|
||||
(0U << 15) | // Reserved
|
||||
(0U << 16) | // Reserved
|
||||
(0U << 17) | // FRED (Flexible Return and Event Delivery)
|
||||
(0U << 18) | // LKGS (Load into Kernel GS Base)
|
||||
(0U << 19) | // WRMSRNS
|
||||
(0U << 20) | // NMI_SRC
|
||||
(0U << 21) | // AMX_FP16
|
||||
(0U << 22) | // HRESET
|
||||
(0U << 23) | // AVX_IFMA
|
||||
(0U << 24) | // Reserved
|
||||
(0U << 25) | // Reserved
|
||||
(0U << 26) | // LAM (Linear Address Masking)
|
||||
(0U << 27) | // MSRLIST
|
||||
(0U << 28) | // Reserved
|
||||
(0U << 29) | // Reserved
|
||||
(0U << 30) | // INVD_DISABLE_POST_BIOS_DONE
|
||||
(0U << 31); // MOVRS
|
||||
Res.eax = (0U << 0) | // SHA512
|
||||
(0U << 1) | // SM3
|
||||
(0U << 2) | // SM4
|
||||
(0U << 3) | // RAO_INT
|
||||
(SupportsAVXVNNI << 4) | // AVX_VNNI
|
||||
(0U << 5) | // AVX512_BF16
|
||||
(0U << 6) | // LASS (Linear Address Space Separation)
|
||||
(0U << 7) | // CMPCCXADD
|
||||
(0U << 8) | // ARCH_PERFMON_EXT
|
||||
(0U << 9) | // Reserved
|
||||
(0U << 10) | // FAST_REP_MOVSB
|
||||
(0U << 11) | // FAST_REP_STOSB
|
||||
(0U << 12) | // FAST_REP_CMPSB_SCASB
|
||||
(0U << 13) | // Reserved
|
||||
(0U << 14) | // Reserved
|
||||
(0U << 15) | // Reserved
|
||||
(0U << 16) | // Reserved
|
||||
(0U << 17) | // FRED (Flexible Return and Event Delivery)
|
||||
(0U << 18) | // LKGS (Load into Kernel GS Base)
|
||||
(0U << 19) | // WRMSRNS
|
||||
(0U << 20) | // NMI_SRC
|
||||
(0U << 21) | // AMX_FP16
|
||||
(0U << 22) | // HRESET
|
||||
(0U << 23) | // AVX_IFMA
|
||||
(0U << 24) | // Reserved
|
||||
(0U << 25) | // Reserved
|
||||
(0U << 26) | // LAM (Linear Address Masking)
|
||||
(0U << 27) | // MSRLIST
|
||||
(0U << 28) | // Reserved
|
||||
(0U << 29) | // Reserved
|
||||
(0U << 30) | // INVD_DISABLE_POST_BIOS_DONE
|
||||
(0U << 31); // MOVRS
|
||||
|
||||
// Bits 4-31 currently reserved.
|
||||
Res.ebx = (0U << 0) | // PPIN
|
||||
|
||||
@@ -1112,8 +1112,10 @@ DEF_OP(VAddP) {
|
||||
// pairwise addition, the SVE version actually interleaves the
|
||||
// results of the pairwise addition (gross!), so we need to undo that.
|
||||
addp(SubRegSize, LHS.Z(), Pred, LHS.Z(), VectorUpper.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
|
||||
|
||||
// Extract the upper half first, since Dst may alias the LHS.
|
||||
uzp2(SubRegSize, VTMP2.Z(), LHS.Z(), LHS.Z());
|
||||
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
|
||||
|
||||
// Merge upper half with lower half.
|
||||
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z());
|
||||
@@ -3833,6 +3835,171 @@ DEF_OP(VMul) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUSDot) {
|
||||
///< Dest = Acc + dot(Vector1 (unsigned 8-bit), Vector2 (signed 8-bit))
|
||||
// Matches:
|
||||
// - SVE - USDOT
|
||||
// - ASIMD - USDOT
|
||||
const auto Op = IROp->C<IR::IROp_VUSDot>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Acc = GetVReg(Op->Acc);
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
// USDOT accumulates in to its destination register,
|
||||
// so we need to emit a move if Acc != Dst
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != Acc) {
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
DestTmp = Dst;
|
||||
} else {
|
||||
DestTmp = VTMP1;
|
||||
}
|
||||
}
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Z(), Acc.Z());
|
||||
}
|
||||
|
||||
usdot(DestTmp.Z(), Vector1.Z(), Vector2.Z());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Q(), Acc.Q());
|
||||
}
|
||||
|
||||
usdot(DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Q(), DestTmp.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSDot) {
|
||||
///< Dest = Acc + dot(Vector1 (signed 8-bit), Vector2 (signed 8-bit))
|
||||
// Matches:
|
||||
// - SVE - SDOT
|
||||
// - ASIMD - SDOT
|
||||
const auto Op = IROp->C<IR::IROp_VSDot>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Acc = GetVReg(Op->Acc);
|
||||
const auto Vector1 = GetVReg(Op->Vector1);
|
||||
const auto Vector2 = GetVReg(Op->Vector2);
|
||||
|
||||
// SDOT accumulates in to its destination register,
|
||||
// so we need to emit a move if Acc != Dst
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != Acc) {
|
||||
if (Dst != Vector1 && Dst != Vector2) {
|
||||
DestTmp = Dst;
|
||||
} else {
|
||||
DestTmp = VTMP1;
|
||||
}
|
||||
}
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Z(), Acc.Z());
|
||||
}
|
||||
|
||||
sdot(ARMEmitter::SubRegSize::i32Bit, DestTmp.Z(), Vector1.Z(), Vector2.Z());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Q(), Acc.Q());
|
||||
}
|
||||
|
||||
sdot(ARMEmitter::SubRegSize::i32Bit, DestTmp.Q(), Vector1.Q(), Vector2.Q());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Q(), DestTmp.Q());
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSAddLP) {
|
||||
const auto Op = IROp->C<IR::IROp_VSAddLP>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
// SVE only has the accumulating form, so accumulate in to a zeroed register.
|
||||
// Zero a temporary instead if Dst aliases the source.
|
||||
const auto DestTmp = Dst == Vector ? VTMP1 : Dst;
|
||||
dup_imm(SubRegSize, DestTmp.Z(), 0);
|
||||
sadalp(SubRegSize, DestTmp.Z(), PRED_TMP_32B.Merging(), Vector.Z());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
saddlp(SubRegSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VSAdALP) {
|
||||
const auto Op = IROp->C<IR::IROp_VSAdALP>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
const auto SubRegSize = ConvertSubRegSize248(IROp);
|
||||
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
|
||||
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
const auto Acc = GetVReg(Op->Acc);
|
||||
const auto Vector = GetVReg(Op->Vector);
|
||||
|
||||
// SADALP accumulates in to its destination register,
|
||||
// so we need to emit a move if Acc != Dst
|
||||
ARMEmitter::VRegister DestTmp = Dst;
|
||||
if (Dst != Acc) {
|
||||
DestTmp = Dst != Vector ? Dst : VTMP1;
|
||||
}
|
||||
|
||||
if (HostSupportsSVE256 && Is256Bit) {
|
||||
if (Dst != Acc) {
|
||||
mov(DestTmp.Z(), Acc.Z());
|
||||
}
|
||||
|
||||
sadalp(SubRegSize, DestTmp.Z(), PRED_TMP_32B.Merging(), Vector.Z());
|
||||
if (Dst != DestTmp) {
|
||||
mov(Dst.Z(), DestTmp.Z());
|
||||
}
|
||||
} else {
|
||||
if (Dst == Vector) {
|
||||
// ASIMD has the non-accumulating form, which is cheaper than shuffling through a temporary.
|
||||
saddlp(SubRegSize, VTMP1.Q(), Vector.Q());
|
||||
add(SubRegSize, Dst.Q(), Acc.Q(), VTMP1.Q());
|
||||
return;
|
||||
}
|
||||
|
||||
if (Dst != Acc) {
|
||||
mov(Dst.Q(), Acc.Q());
|
||||
}
|
||||
|
||||
sadalp(SubRegSize, Dst.Q(), Vector.Q());
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(VUMull) {
|
||||
const auto Op = IROp->C<IR::IROp_VUMull>();
|
||||
const auto OpSize = IROp->Size;
|
||||
|
||||
@@ -665,6 +665,8 @@ public:
|
||||
|
||||
void VPMADDUBSWOp(OpcodeArgs);
|
||||
void VPMADDWDOp(OpcodeArgs);
|
||||
void VPDPBUSDOp(OpcodeArgs, bool Saturating);
|
||||
void VPDPWSSDOp(OpcodeArgs, bool Saturating);
|
||||
|
||||
void VPMASKMOVOp(OpcodeArgs, bool IsStore);
|
||||
|
||||
@@ -1036,6 +1038,9 @@ public:
|
||||
|
||||
void AVX128_VPMADDUBSW(OpcodeArgs);
|
||||
void AVX128_VPMADDWD(OpcodeArgs);
|
||||
void AVX128_VPDPImpl(OpcodeArgs, std::function<Ref(Ref Acc, Ref Src1, Ref Src2)> Helper);
|
||||
void AVX128_VPDPBUSD(OpcodeArgs, bool Saturating);
|
||||
void AVX128_VPDPWSSD(OpcodeArgs, bool Saturating);
|
||||
|
||||
void AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize);
|
||||
|
||||
@@ -1400,6 +1405,10 @@ private:
|
||||
|
||||
Ref PMADDUBSWOpImpl(IR::OpSize Size, Ref Src1, Ref Src2);
|
||||
|
||||
Ref VPDPBUSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating);
|
||||
|
||||
Ref VPDPWSSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating);
|
||||
|
||||
Ref PMULHRSWOpImpl(OpSize Size, Ref Src1, Ref Src2);
|
||||
|
||||
Ref PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2);
|
||||
|
||||
@@ -1470,6 +1470,35 @@ void OpDispatchBuilder::AVX128_VPMADDWD(OpcodeArgs) {
|
||||
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return PMADDWDOpImpl(OpSize::i128Bit, Src1, Src2); });
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPDPImpl(OpcodeArgs, std::function<Ref(Ref Acc, Ref Src1, Ref Src2)> Helper) {
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
const auto Is128Bit = Size == OpSize::i128Bit;
|
||||
|
||||
auto Acc = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, !Is128Bit);
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
|
||||
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit);
|
||||
|
||||
RefPair Result {};
|
||||
Result.Low = Helper(Acc.Low, Src1.Low, Src2.Low);
|
||||
if (Is128Bit) {
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
} else {
|
||||
Result.High = Helper(Acc.High, Src1.High, Src2.High);
|
||||
}
|
||||
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPDPBUSD(OpcodeArgs, bool Saturating) {
|
||||
AVX128_VPDPImpl(Op,
|
||||
[this, Saturating](Ref Acc, Ref Src1, Ref Src2) { return VPDPBUSDOpImpl(OpSize::i128Bit, Acc, Src1, Src2, Saturating); });
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VPDPWSSD(OpcodeArgs, bool Saturating) {
|
||||
AVX128_VPDPImpl(Op,
|
||||
[this, Saturating](Ref Acc, Ref Src1, Ref Src2) { return VPDPWSSDOpImpl(OpSize::i128Bit, Acc, Src1, Src2, Saturating); });
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto Is128Bit = SrcSize == OpSize::i128Bit;
|
||||
|
||||
@@ -3760,6 +3760,84 @@ void OpDispatchBuilder::VPMADDUBSWOp(OpcodeArgs) {
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::VPDPBUSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating) {
|
||||
// Does four 8-bit unsigned * signed byte multiplies per 32-bit element, sums them and accumulates in to the destination
|
||||
|
||||
if (CTX->HostFeatures.SupportsI8MM) {
|
||||
// The I8MM extension maps onto VPDP* almost directly.
|
||||
if (!Saturating) {
|
||||
return _VUSDot(Size, Acc, Src1, Src2);
|
||||
}
|
||||
|
||||
auto DotProduct = _VUSDot(Size, LoadZeroVector(Size), Src1, Src2);
|
||||
return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
|
||||
if (CTX->HostFeatures.SupportsDotProd) {
|
||||
// VSDOT assumes signed input, so we need to convert Src1 to signed, and then
|
||||
// perform correction afterwards using 0x40 to account for the unsigned input.
|
||||
auto Src1Signed = _VXor(Size, Src1, _VectorImm(Size, OpSize::i8Bit, 0x80));
|
||||
auto SixtyFour = _VectorImm(Size, OpSize::i8Bit, 0x40);
|
||||
|
||||
auto DotProduct = _VSDot(Size, Saturating ? LoadZeroVector(Size) : Acc, Src2, SixtyFour);
|
||||
DotProduct = _VSDot(Size, DotProduct, Src2, SixtyFour);
|
||||
DotProduct = _VSDot(Size, DotProduct, Src1Signed, Src2);
|
||||
if (Saturating) {
|
||||
return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
return DotProduct;
|
||||
}
|
||||
|
||||
// Naive software implementation.
|
||||
auto Even = _VUnZip(Size, OpSize::i16Bit, Src1, Src2);
|
||||
auto Even1_16b = _VUXTL(Size, OpSize::i8Bit, Even);
|
||||
auto Even2_16b = _VSXTL2(Size, OpSize::i8Bit, Even);
|
||||
auto ResMul_Even = _VMul(Size, OpSize::i16Bit, Even1_16b, Even2_16b);
|
||||
|
||||
auto Odd = _VUnZip2(Size, OpSize::i16Bit, Src1, Src2);
|
||||
auto Odd1_16b = _VUXTL(Size, OpSize::i8Bit, Odd);
|
||||
auto Odd2_16b = _VSXTL2(Size, OpSize::i8Bit, Odd);
|
||||
auto ResMul_Odd = _VMul(Size, OpSize::i16Bit, Odd1_16b, Odd2_16b);
|
||||
|
||||
auto DotProduct = _VSAdALP(Size, OpSize::i16Bit, _VSAddLP(Size, OpSize::i16Bit, ResMul_Even), ResMul_Odd);
|
||||
if (Saturating) {
|
||||
return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
return _VAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::VPDPWSSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating) {
|
||||
auto DotProduct = PMADDWDOpImpl(Size, Src1, Src2);
|
||||
if (!Saturating) {
|
||||
return _VAdd(Size, OpSize::i32Bit, Acc, DotProduct);
|
||||
}
|
||||
|
||||
auto NegDotProduct = _VNeg(Size, OpSize::i32Bit, DotProduct);
|
||||
return _VSQSub(Size, OpSize::i32Bit, Acc, NegDotProduct);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPDPBUSDOp(OpcodeArgs, bool Saturating) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Acc = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
|
||||
Ref Result = VPDPBUSDOpImpl(Size, Acc, Src1, Src2, Saturating);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPDPWSSDOp(OpcodeArgs, bool Saturating) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
Ref Acc = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
|
||||
Ref Result = VPDPWSSDOpImpl(Size, Acc, Src1, Src2, Saturating);
|
||||
StoreResultFPR(Op, Result);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
if (Signed) {
|
||||
|
||||
@@ -312,6 +312,11 @@ namespace AVX128 {
|
||||
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VSSHR>}, // VPSRAVD
|
||||
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VUSHL>}, // VPSLLV
|
||||
|
||||
{OPD(2, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPBUSD, false>},
|
||||
{OPD(2, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPBUSD, true>},
|
||||
{OPD(2, 0b01, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPWSSD, false>},
|
||||
{OPD(2, 0b01, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPWSSD, true>},
|
||||
|
||||
{OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i128Bit>},
|
||||
@@ -757,6 +762,11 @@ namespace AVX256 {
|
||||
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::VPSRAVDOp},
|
||||
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::VPSLLVOp},
|
||||
|
||||
{OPD(2, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPBUSDOp, false>},
|
||||
{OPD(2, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPBUSDOp, true>},
|
||||
{OPD(2, 0b01, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPWSSDOp, false>},
|
||||
{OPD(2, 0b01, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPWSSDOp, true>},
|
||||
|
||||
{OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i32Bit>},
|
||||
{OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i64Bit>},
|
||||
{OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i128Bit>},
|
||||
@@ -1207,6 +1217,11 @@ auto BaseTableLambda = [](const auto RuntimeTable) consteval {
|
||||
{OPD(2, 0b01, 0x46), 1, X86InstInfo{"VPSRAVD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x47), 1, X86InstInfo{"VPSLLV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(2, 0b01, 0x50), 1, X86InstInfo{"VPDPBUSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x51), 1, X86InstInfo{"VPDPBUSDS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x52), 1, X86InstInfo{"VPDPWSSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x53), 1, X86InstInfo{"VPDPWSSDS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
{OPD(2, 0b01, 0x58), 1, X86InstInfo{"VPBROADCASTD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x59), 1, X86InstInfo{"VPBROADCASTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
{OPD(2, 0b01, 0x5A), 1, X86InstInfo{"VBROADCASTI128", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_L_1 | FLAGS_SF_MOD_MEM_ONLY | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
|
||||
|
||||
@@ -2363,6 +2363,43 @@
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VUSDot OpSize:#RegisterSize, FPR:$Acc, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Unsigned by signed 8-bit dot product, accumulating four products in to each 32-bit element of Acc.",
|
||||
"Requires FEAT_I8MM."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "OpSize::i32Bit",
|
||||
"TiedSource": 0,
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
|
||||
]
|
||||
},
|
||||
"FPR = VSDot OpSize:#RegisterSize, FPR:$Acc, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Signed 8-bit dot product, accumulating four products in to each 32-bit element of Acc.",
|
||||
"Requires FEAT_DotProd."
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "OpSize::i32Bit",
|
||||
"TiedSource": 0,
|
||||
"EmitValidation": [
|
||||
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
|
||||
]
|
||||
},
|
||||
"FPR = VSAddLP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
|
||||
"Desc": ["Signed add long pairwise. Adds adjacent pairs of elements in to elements of twice the size.",
|
||||
"ElementSize is the source size"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1"
|
||||
},
|
||||
"FPR = VSAdALP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Acc, FPR:$Vector": {
|
||||
"Desc": ["Signed add and accumulate long pairwise. Adds adjacent pairs of elements in to the elements of Acc, which are twice the size.",
|
||||
"ElementSize is the source size"
|
||||
],
|
||||
"DestSize": "RegisterSize",
|
||||
"ElementSize": "ElementSize << 1",
|
||||
"TiedSource": 0
|
||||
},
|
||||
"FPR = VUMulH OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"Desc": ["Wide unsigned multiply returning the high results"],
|
||||
"DestSize": "RegisterSize",
|
||||
|
||||
@@ -71,13 +71,15 @@ struct HostFeatures {
|
||||
uint32_t Supports3DNow : 1 {};
|
||||
uint32_t SupportsSSE4a : 1 {};
|
||||
uint32_t SupportsMOPS : 1 {};
|
||||
uint32_t SupportsI8MM : 1 {};
|
||||
uint32_t SupportsDotProd : 1 {};
|
||||
uint32_t PreferZVAForVZero : 1 {};
|
||||
uint32_t SupportsAFP : 1 {};
|
||||
uint32_t SupportsFloatExceptions : 1 {};
|
||||
// Flag if this is InstCountCI
|
||||
uint32_t IsInstCountCI : 1 {};
|
||||
HostTypeEnum HostType : 2 {};
|
||||
uint32_t pad : 26 {};
|
||||
uint32_t pad : 24 {};
|
||||
|
||||
// MIDR information
|
||||
// Also used for determining number of CPU cores for CPUID
|
||||
|
||||
@@ -60,6 +60,8 @@ class HostFeatures(Flag) :
|
||||
FEATURE_LRCPC2 = (1 << 15)
|
||||
FEATURE_FRINTTS = (1 << 16)
|
||||
FEATURE_MOPS = (1 << 17)
|
||||
FEATURE_I8MM = (1 << 18)
|
||||
FEATURE_DOTPROD = (1 << 19)
|
||||
|
||||
HostFeaturesLookup = {
|
||||
"SVE128" : HostFeatures.FEATURE_SVE128,
|
||||
@@ -80,6 +82,8 @@ HostFeaturesLookup = {
|
||||
"LRCPC2" : HostFeatures.FEATURE_LRCPC2,
|
||||
"FRINTTS" : HostFeatures.FEATURE_FRINTTS,
|
||||
"MOPS" : HostFeatures.FEATURE_MOPS,
|
||||
"I8MM" : HostFeatures.FEATURE_I8MM,
|
||||
"DOTPROD" : HostFeatures.FEATURE_DOTPROD,
|
||||
}
|
||||
|
||||
def GetHostFeatures(data):
|
||||
|
||||
@@ -87,6 +87,7 @@ class HostFeatures(Flag) :
|
||||
FEATURE_CLFLOPT = (1 << 21)
|
||||
FEATURE_FSGSBASE = (1 << 22)
|
||||
FEATURE_EMMI = (1 << 23)
|
||||
FEATURE_AVX_VNNI = (1 << 24)
|
||||
|
||||
RegStringLookup = {
|
||||
"NONE": Regs.REG_NONE,
|
||||
@@ -173,6 +174,7 @@ HostFeaturesLookup = {
|
||||
"CLFLOPT" : HostFeatures.FEATURE_CLFLOPT,
|
||||
"FSGSBASE" : HostFeatures.FEATURE_FSGSBASE,
|
||||
"EMMI" : HostFeatures.FEATURE_EMMI,
|
||||
"AVX_VNNI" : HostFeatures.FEATURE_AVX_VNNI,
|
||||
}
|
||||
|
||||
def parse_hexstring(s):
|
||||
|
||||
@@ -374,6 +374,8 @@ static void OverrideFeatures(FEXCore::HostFeatures* Features, uint64_t ForceSVEW
|
||||
ENABLE_DISABLE_OPTION(Supports3DNow, 3DNOW, 3DNOW);
|
||||
ENABLE_DISABLE_OPTION(SupportsSSE4a, SSE4A, SSE4A);
|
||||
ENABLE_DISABLE_OPTION(SupportsMOPS, MOPS, MOPS);
|
||||
ENABLE_DISABLE_OPTION(SupportsI8MM, I8MM, I8MM);
|
||||
ENABLE_DISABLE_OPTION(SupportsDotProd, DOTPROD, DOTPROD);
|
||||
GET_SINGLE_OPTION(Crypto, CRYPTO);
|
||||
|
||||
#undef ENABLE_DISABLE_OPTION
|
||||
@@ -504,6 +506,8 @@ void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFe
|
||||
HostFeatures.SupportsSVEBitPerm = Features.Supports(CPUFeatures::Feature::SVE_BitPerm);
|
||||
HostFeatures.SupportsECV = Features.Supports(CPUFeatures::Feature::ECV);
|
||||
HostFeatures.SupportsWFXT = Features.Supports(CPUFeatures::Feature::WFxt);
|
||||
HostFeatures.SupportsI8MM = Features.Supports(CPUFeatures::Feature::I8MM);
|
||||
HostFeatures.SupportsDotProd = Features.Supports(CPUFeatures::Feature::DotProd);
|
||||
|
||||
#ifdef VIXL_SIMULATOR
|
||||
// Hardcode enable SVE with 256-bit wide registers.
|
||||
|
||||
@@ -40,6 +40,11 @@ public:
|
||||
Feat_vaes = data_7.ecx & (1U << 9);
|
||||
Feat_pclmulqdq = Feat_pclmulqdq && (data_7.ecx & (1U << 10));
|
||||
Feat_rdpid = data_7.ecx & (1U << 22);
|
||||
|
||||
if (data_7.eax >= 1) {
|
||||
auto data_7_1 = cpuid(0x7, 0x1);
|
||||
Feat_avx_vnni = Feat_avx && (data_7_1.eax & (1U << 4));
|
||||
}
|
||||
}
|
||||
|
||||
data = cpuid(0x8000'0000U);
|
||||
@@ -79,6 +84,7 @@ public:
|
||||
bool Feat_rdpid {};
|
||||
bool Feat_clflopt {};
|
||||
bool Feat_fsgsbase {};
|
||||
bool Feat_avx_vnni {};
|
||||
|
||||
private:
|
||||
struct cpuid_data {
|
||||
|
||||
@@ -557,6 +557,8 @@ int main(int argc, char** argv, char** const envp) {
|
||||
FEATURE_LRCPC2 = (1U << 15),
|
||||
FEATURE_FRINTTS = (1U << 16),
|
||||
FEATURE_MOPS = (1U << 17),
|
||||
FEATURE_I8MM = (1U << 18),
|
||||
FEATURE_DOTPROD = (1U << 19),
|
||||
};
|
||||
|
||||
uint64_t SVEWidth = 0;
|
||||
@@ -610,6 +612,12 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_MOPS) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEMOPS);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_I8MM) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEI8MM);
|
||||
}
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_DOTPROD) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEDOTPROD);
|
||||
}
|
||||
|
||||
if (TestHeaderData->EnabledHostFeatures & FEATURE_TSO) {
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "1");
|
||||
@@ -668,6 +676,12 @@ int main(int argc, char** argv, char** const envp) {
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_MOPS) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEMOPS);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_I8MM) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEI8MM);
|
||||
}
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_DOTPROD) {
|
||||
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEDOTPROD);
|
||||
}
|
||||
|
||||
if (TestHeaderData->DisabledHostFeatures & FEATURE_TSO) {
|
||||
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "0");
|
||||
|
||||
@@ -321,6 +321,7 @@ public:
|
||||
FEATURE_CLFLOPT = (1 << 21),
|
||||
FEATURE_FSGSBASE = (1 << 22),
|
||||
FEATURE_EMMI = (1 << 23),
|
||||
FEATURE_AVX_VNNI = (1 << 24),
|
||||
};
|
||||
|
||||
bool Requires3DNow() const {
|
||||
@@ -395,6 +396,9 @@ public:
|
||||
bool RequiresEMMI() const {
|
||||
return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_EMMI;
|
||||
}
|
||||
bool RequiresAVXVNNI() const {
|
||||
return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_AVX_VNNI;
|
||||
}
|
||||
|
||||
private:
|
||||
FEX_CONFIG_OPT(ConfigDumpGPRs, DUMPGPRS);
|
||||
@@ -602,6 +606,9 @@ public:
|
||||
bool RequiresEMMI() const {
|
||||
return Config.RequiresEMMI();
|
||||
}
|
||||
bool RequiresAVXVNNI() const {
|
||||
return Config.RequiresAVXVNNI();
|
||||
}
|
||||
|
||||
private:
|
||||
constexpr static uint64_t STACK_OFFSET = 0xc000'0000;
|
||||
|
||||
@@ -287,6 +287,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
const bool SupportsRDPID = Feature.Feat_rdpid;
|
||||
const bool SupportsCLFLOPT = Feature.Feat_clflopt;
|
||||
const bool SupportsFSGSBase = Feature.Feat_fsgsbase;
|
||||
const bool SupportsAVXVNNI = Feature.Feat_avx_vnni;
|
||||
|
||||
TestUnsupported |=
|
||||
(!Supports3DNow && Loader.Requires3DNow()) || (!SupportsSSE4A && Loader.RequiresSSE4A()) || (!SupportsBMI1 && Loader.RequiresBMI1()) ||
|
||||
@@ -295,6 +296,7 @@ int main(int argc, char** argv, char** const envp) {
|
||||
(!SupportsAES && Loader.RequiresAES()) || (!SupportsPCLMUL && Loader.RequiresPCLMUL()) || (!SupportsMOVBE && Loader.RequiresMOVBE()) ||
|
||||
(!SupportsADX && Loader.RequiresADX()) || (!SupportsXSAVE && Loader.RequiresXSAVE()) || (!SupportsRDPID && Loader.RequiresRDPID()) ||
|
||||
(!SupportsCLFLOPT && Loader.RequiresCLFLOPT()) || (!SupportsFSGSBase && Loader.RequiresFSGSBase()) || Loader.RequiresEMMI();
|
||||
TestUnsupported |= !SupportsAVXVNNI && Loader.RequiresAVXVNNI();
|
||||
#endif
|
||||
|
||||
#ifdef _WIN32
|
||||
|
||||
@@ -0,0 +1,59 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"HostFeatures": ["AVX", "AVX_VNNI"],
|
||||
"RegData": {
|
||||
"XMM4": ["0xFFFFFF3100000006", "0x7FFF07008001F904", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM5": ["0xFFFFFFFF00000020", "0x7FFFC08080013B03", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM6": ["0x02FE0140FC03FDF7", "0x807F86807F817983", "0xFCFDFD0101FF8000", "0x807E82807F817983"],
|
||||
"XMM7": ["0xFFFFFF3100000006", "0x7FFF07008001F904", "0x87654123123456F8", "0x7FFE02108001F9F4"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
vmovdqa ymm0, [rel src1]
|
||||
vmovdqa ymm1, [rel src2]
|
||||
vmovdqa ymm4, [rel accumulator]
|
||||
vmovdqa ymm5, [rel accumulator]
|
||||
vmovdqa ymm6, [rel src2]
|
||||
vmovdqa ymm7, [rel accumulator]
|
||||
lea rdx, [rel src2]
|
||||
|
||||
; Hand assembled instructions, since NASM will only output
|
||||
; the EVEX encodings for these instructions.
|
||||
db 0xC4, 0xE2, 0x79, 0x50, 0xE1 ; vpdpbusd xmm4, xmm0, xmm1
|
||||
db 0xC4, 0xE2, 0x51, 0x50, 0x2A ; vpdpbusd xmm5, xmm5, [rdx]
|
||||
db 0xC4, 0xE2, 0x7D, 0x50, 0xF6 ; vpdpbusd ymm6, ymm0, ymm6
|
||||
db 0xC4, 0xE2, 0x7D, 0x50, 0x3A ; vpdpbusd ymm7, ymm0, [rdx]
|
||||
|
||||
hlt
|
||||
|
||||
align 32
|
||||
accumulator:
|
||||
dd 0x00000010
|
||||
dd 0xFFFFFFF0
|
||||
dd 0x7FFFFF00
|
||||
dd 0x80000100
|
||||
dd 0x12345678
|
||||
dd 0x87654321
|
||||
dd 0x7FFFFFF0
|
||||
dd 0x80000010
|
||||
|
||||
src1:
|
||||
db 1, 2, 3, 4
|
||||
db 255, 128, 64, 32
|
||||
db 255, 255, 255, 255
|
||||
db 200, 150, 100, 50
|
||||
db 0, 1, 254, 255
|
||||
db 17, 34, 51, 68
|
||||
db 255, 255, 255, 255
|
||||
db 255, 255, 255, 255
|
||||
|
||||
src2:
|
||||
db 1, -2, 3, -4
|
||||
db -1, 1, -2, 2
|
||||
db 127, 127, 127, 127
|
||||
db -128, -128, -128, -128
|
||||
db -128, 127, -1, 1
|
||||
db -1, -2, -3, -4
|
||||
db 127, 127, 127, 127
|
||||
db -128, -128, -128, -128
|
||||
@@ -0,0 +1,59 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"HostFeatures": ["AVX", "AVX_VNNI"],
|
||||
"RegData": {
|
||||
"XMM4": ["0xFFFFFF3100000006", "0x800000007FFFFFFF", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM5": ["0xFFFFFFFF00000020", "0x800000007FFFFFFF", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM6": ["0x02FE0140FC03FDF7", "0x807F86807F817983", "0xFCFDFD0101FF8000", "0x807E82807F817983"],
|
||||
"XMM7": ["0xFFFFFF3100000006", "0x800000007FFFFFFF", "0x87654123123456F8", "0x800000007FFFFFFF"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
vmovdqa ymm0, [rel src1]
|
||||
vmovdqa ymm1, [rel src2]
|
||||
vmovdqa ymm4, [rel accumulator]
|
||||
vmovdqa ymm5, [rel accumulator]
|
||||
vmovdqa ymm6, [rel src2]
|
||||
vmovdqa ymm7, [rel accumulator]
|
||||
lea rdx, [rel src2]
|
||||
|
||||
; Hand assembled instructions, since NASM will only output
|
||||
; the EVEX encodings for these instructions.
|
||||
db 0xC4, 0xE2, 0x79, 0x51, 0xE1 ; vpdpbusds xmm4, xmm0, xmm1
|
||||
db 0xC4, 0xE2, 0x51, 0x51, 0x2A ; vpdpbusds xmm5, xmm5, [rdx]
|
||||
db 0xC4, 0xE2, 0x7D, 0x51, 0xF6 ; vpdpbusds ymm6, ymm0, ymm6
|
||||
db 0xC4, 0xE2, 0x7D, 0x51, 0x3A ; vpdpbusds ymm7, ymm0, [rdx]
|
||||
|
||||
hlt
|
||||
|
||||
align 32
|
||||
accumulator:
|
||||
dd 0x00000010
|
||||
dd 0xFFFFFFF0
|
||||
dd 0x7FFFFF00
|
||||
dd 0x80000100
|
||||
dd 0x12345678
|
||||
dd 0x87654321
|
||||
dd 0x7FFFFFF0
|
||||
dd 0x80000010
|
||||
|
||||
src1:
|
||||
db 1, 2, 3, 4
|
||||
db 255, 128, 64, 32
|
||||
db 255, 255, 255, 255
|
||||
db 200, 150, 100, 50
|
||||
db 0, 1, 254, 255
|
||||
db 17, 34, 51, 68
|
||||
db 255, 255, 255, 255
|
||||
db 255, 255, 255, 255
|
||||
|
||||
src2:
|
||||
db 1, -2, 3, -4
|
||||
db -1, 1, -2, 2
|
||||
db 127, 127, 127, 127
|
||||
db -128, -128, -128, -128
|
||||
db -128, 127, -1, 1
|
||||
db -1, -2, -3, -4
|
||||
db 127, 127, 127, 127
|
||||
db -128, -128, -128, -128
|
||||
@@ -0,0 +1,59 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"HostFeatures": ["AVX", "AVX_VNNI"],
|
||||
"RegData": {
|
||||
"XMM4": ["0x800000000000000B", "0xFFFF000180000002", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM5": ["0x0000000000000040", "0x400080000002FFFE", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM6": ["0x00008000FFFBFFFE", "0xFFFE8001FFFD8001", "0x800100001C173F20", "0x0001800000008000"],
|
||||
"XMM7": ["0x800000000000000B", "0xFFFF000180000002", "0x8765C322E02B0AC8", "0x000100107FFFFFFE"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
vmovdqa ymm0, [rel src1]
|
||||
vmovdqa ymm1, [rel src2]
|
||||
vmovdqa ymm4, [rel accumulator]
|
||||
vmovdqa ymm5, [rel accumulator]
|
||||
vmovdqa ymm6, [rel src2]
|
||||
vmovdqa ymm7, [rel accumulator]
|
||||
lea rdx, [rel src2]
|
||||
|
||||
; Hand assembled instructions, since NASM will only output
|
||||
; the EVEX encodings for these instructions
|
||||
db 0xC4, 0xE2, 0x79, 0x52, 0xE1 ; vpdpwssd xmm4, xmm0, xmm1
|
||||
db 0xC4, 0xE2, 0x51, 0x52, 0x2A ; vpdpwssd xmm5, xmm5, [rdx]
|
||||
db 0xC4, 0xE2, 0x7D, 0x52, 0xF6 ; vpdpwssd ymm6, ymm0, ymm6
|
||||
db 0xC4, 0xE2, 0x7D, 0x52, 0x3A ; vpdpwssd ymm7, ymm0, [rdx]
|
||||
|
||||
hlt
|
||||
|
||||
align 32
|
||||
accumulator:
|
||||
dd 0x00000010
|
||||
dd 0x00000000
|
||||
dd 0x00020000
|
||||
dd 0x80000000
|
||||
dd 0x12345678
|
||||
dd 0x87654321
|
||||
dd 0xFFFFFFFE
|
||||
dd 0x80000010
|
||||
|
||||
src1:
|
||||
dw 1, 2
|
||||
dw -32768, -32768
|
||||
dw 32767, 32767
|
||||
dw -32768, 32767
|
||||
dw 12345, -23456
|
||||
dw -1, -2
|
||||
dw -32768, -32768
|
||||
dw 32767, 32767
|
||||
|
||||
src2:
|
||||
dw 3, -4
|
||||
dw -32768, -32768
|
||||
dw 32767, 32767
|
||||
dw -32768, 32767
|
||||
dw -30000, 20000
|
||||
dw 32767, -32768
|
||||
dw -32768, -32768
|
||||
dw -32768, -32768
|
||||
@@ -0,0 +1,59 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"HostFeatures": ["AVX", "AVX_VNNI"],
|
||||
"RegData": {
|
||||
"XMM4": ["0x7FFFFFFF0000000B", "0xFFFF00017FFFFFFF", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM5": ["0x0000000000000040", "0x800000000002FFFE", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM6": ["0x00008000FFFBFFFE", "0x7FFFFFFF7FFFFFFF", "0x800100001C173F20", "0x8000000000008000"],
|
||||
"XMM7": ["0x7FFFFFFF0000000B", "0xFFFF00017FFFFFFF", "0x8765C322E02B0AC8", "0x800000007FFFFFFE"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
vmovdqa ymm0, [rel src1]
|
||||
vmovdqa ymm1, [rel src2]
|
||||
vmovdqa ymm4, [rel accumulator]
|
||||
vmovdqa ymm5, [rel accumulator]
|
||||
vmovdqa ymm6, [rel src2]
|
||||
vmovdqa ymm7, [rel accumulator]
|
||||
lea rdx, [rel src2]
|
||||
|
||||
; Hand assembled instructions, since NASM will only output
|
||||
; the EVEX encodings for these instructions.
|
||||
db 0xC4, 0xE2, 0x79, 0x53, 0xE1 ; vpdpwssds xmm4, xmm0, xmm1
|
||||
db 0xC4, 0xE2, 0x51, 0x53, 0x2A ; vpdpwssds xmm5, xmm5, [rdx]
|
||||
db 0xC4, 0xE2, 0x7D, 0x53, 0xF6 ; vpdpwssds ymm6, ymm0, ymm6
|
||||
db 0xC4, 0xE2, 0x7D, 0x53, 0x3A ; vpdpwssds ymm7, ymm0, [rdx]
|
||||
|
||||
hlt
|
||||
|
||||
align 32
|
||||
accumulator:
|
||||
dd 0x00000010
|
||||
dd 0x00000000
|
||||
dd 0x00020000
|
||||
dd 0x80000000
|
||||
dd 0x12345678
|
||||
dd 0x87654321
|
||||
dd 0xFFFFFFFE
|
||||
dd 0x80000010
|
||||
|
||||
src1:
|
||||
dw 1, 2
|
||||
dw -32768, -32768
|
||||
dw 32767, 32767
|
||||
dw -32768, 32767
|
||||
dw 12345, -23456
|
||||
dw -1, -2
|
||||
dw -32768, -32768
|
||||
dw 32767, 32767
|
||||
|
||||
src2:
|
||||
dw 3, -4
|
||||
dw -32768, -32768
|
||||
dw 32767, 32767
|
||||
dw -32768, 32767
|
||||
dw -30000, 20000
|
||||
dw 32767, -32768
|
||||
dw -32768, -32768
|
||||
dw -32768, -32768
|
||||
@@ -7,7 +7,9 @@
|
||||
"FLAGM",
|
||||
"FLAGM2",
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
"SVE256",
|
||||
"I8MM",
|
||||
"DOTPROD"
|
||||
],
|
||||
"BinaryCacheVersion": 20
|
||||
},
|
||||
@@ -2075,6 +2077,231 @@
|
||||
"str q2, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpbusd xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 12,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x50 128-bit",
|
||||
"vpdpbusd xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x50, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"uzp1 v2.8h, v17.8h, v18.8h",
|
||||
"uxtl v3.8h, v2.8b",
|
||||
"sxtl2 v2.8h, v2.16b",
|
||||
"mul v2.8h, v3.8h, v2.8h",
|
||||
"uzp2 v3.8h, v17.8h, v18.8h",
|
||||
"uxtl v4.8h, v3.8b",
|
||||
"sxtl2 v3.8h, v3.16b",
|
||||
"mul v3.8h, v4.8h, v3.8h",
|
||||
"saddlp v2.4s, v2.8h",
|
||||
"sadalp v2.4s, v3.8h",
|
||||
"add v16.4s, v16.4s, v2.4s",
|
||||
"stp xzr, xzr, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpbusd ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 26,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x50 256-bit",
|
||||
"vpdpbusd ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x50, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #192]",
|
||||
"ldr q3, [x28, #208]",
|
||||
"ldr q4, [x28, #224]",
|
||||
"uzp1 v5.8h, v17.8h, v18.8h",
|
||||
"uxtl v6.8h, v5.8b",
|
||||
"sxtl2 v5.8h, v5.16b",
|
||||
"mul v5.8h, v6.8h, v5.8h",
|
||||
"uzp2 v6.8h, v17.8h, v18.8h",
|
||||
"uxtl v7.8h, v6.8b",
|
||||
"sxtl2 v6.8h, v6.16b",
|
||||
"mul v6.8h, v7.8h, v6.8h",
|
||||
"saddlp v5.4s, v5.8h",
|
||||
"sadalp v5.4s, v6.8h",
|
||||
"add v16.4s, v16.4s, v5.4s",
|
||||
"uzp1 v5.8h, v3.8h, v4.8h",
|
||||
"uxtl v6.8h, v5.8b",
|
||||
"sxtl2 v5.8h, v5.16b",
|
||||
"mul v5.8h, v6.8h, v5.8h",
|
||||
"uzp2 v3.8h, v3.8h, v4.8h",
|
||||
"uxtl v4.8h, v3.8b",
|
||||
"sxtl2 v3.8h, v3.16b",
|
||||
"mul v3.8h, v4.8h, v3.8h",
|
||||
"saddlp v4.4s, v5.8h",
|
||||
"sadalp v4.4s, v3.8h",
|
||||
"add v2.4s, v2.4s, v4.4s",
|
||||
"str q2, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpbusds xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 12,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x51 128-bit",
|
||||
"vpdpbusds xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x51, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"uzp1 v2.8h, v17.8h, v18.8h",
|
||||
"uxtl v3.8h, v2.8b",
|
||||
"sxtl2 v2.8h, v2.16b",
|
||||
"mul v2.8h, v3.8h, v2.8h",
|
||||
"uzp2 v3.8h, v17.8h, v18.8h",
|
||||
"uxtl v4.8h, v3.8b",
|
||||
"sxtl2 v3.8h, v3.16b",
|
||||
"mul v3.8h, v4.8h, v3.8h",
|
||||
"saddlp v2.4s, v2.8h",
|
||||
"sadalp v2.4s, v3.8h",
|
||||
"sqadd v16.4s, v16.4s, v2.4s",
|
||||
"stp xzr, xzr, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpbusds ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 26,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x51 256-bit",
|
||||
"vpdpbusds ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x51, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #192]",
|
||||
"ldr q3, [x28, #208]",
|
||||
"ldr q4, [x28, #224]",
|
||||
"uzp1 v5.8h, v17.8h, v18.8h",
|
||||
"uxtl v6.8h, v5.8b",
|
||||
"sxtl2 v5.8h, v5.16b",
|
||||
"mul v5.8h, v6.8h, v5.8h",
|
||||
"uzp2 v6.8h, v17.8h, v18.8h",
|
||||
"uxtl v7.8h, v6.8b",
|
||||
"sxtl2 v6.8h, v6.16b",
|
||||
"mul v6.8h, v7.8h, v6.8h",
|
||||
"saddlp v5.4s, v5.8h",
|
||||
"sadalp v5.4s, v6.8h",
|
||||
"sqadd v16.4s, v16.4s, v5.4s",
|
||||
"uzp1 v5.8h, v3.8h, v4.8h",
|
||||
"uxtl v6.8h, v5.8b",
|
||||
"sxtl2 v5.8h, v5.16b",
|
||||
"mul v5.8h, v6.8h, v5.8h",
|
||||
"uzp2 v3.8h, v3.8h, v4.8h",
|
||||
"uxtl v4.8h, v3.8b",
|
||||
"sxtl2 v3.8h, v3.16b",
|
||||
"mul v3.8h, v4.8h, v3.8h",
|
||||
"saddlp v4.4s, v5.8h",
|
||||
"sadalp v4.4s, v3.8h",
|
||||
"sqadd v2.4s, v2.4s, v4.4s",
|
||||
"str q2, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpwssd xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x52 128-bit",
|
||||
"vpdpwssd xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x52, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"smull v2.4s, v17.4h, v18.4h",
|
||||
"smull2 v3.4s, v17.8h, v18.8h",
|
||||
"addp v2.4s, v2.4s, v3.4s",
|
||||
"add v16.4s, v16.4s, v2.4s",
|
||||
"stp xzr, xzr, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpwssd ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 12,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x52 256-bit",
|
||||
"vpdpwssd ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x52, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #192]",
|
||||
"ldr q3, [x28, #208]",
|
||||
"ldr q4, [x28, #224]",
|
||||
"smull v5.4s, v17.4h, v18.4h",
|
||||
"smull2 v6.4s, v17.8h, v18.8h",
|
||||
"addp v5.4s, v5.4s, v6.4s",
|
||||
"add v16.4s, v16.4s, v5.4s",
|
||||
"smull v5.4s, v3.4h, v4.4h",
|
||||
"smull2 v3.4s, v3.8h, v4.8h",
|
||||
"addp v3.4s, v5.4s, v3.4s",
|
||||
"add v2.4s, v2.4s, v3.4s",
|
||||
"str q2, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpwssds xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x53 128-bit",
|
||||
"vpdpwssds xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x53, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"smull v2.4s, v17.4h, v18.4h",
|
||||
"smull2 v3.4s, v17.8h, v18.8h",
|
||||
"addp v2.4s, v2.4s, v3.4s",
|
||||
"neg v2.4s, v2.4s",
|
||||
"sqsub v16.4s, v16.4s, v2.4s",
|
||||
"stp xzr, xzr, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpwssds ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 14,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x53 256-bit",
|
||||
"vpdpwssds ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x53, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #192]",
|
||||
"ldr q3, [x28, #208]",
|
||||
"ldr q4, [x28, #224]",
|
||||
"smull v5.4s, v17.4h, v18.4h",
|
||||
"smull2 v6.4s, v17.8h, v18.8h",
|
||||
"addp v5.4s, v5.4s, v6.4s",
|
||||
"neg v5.4s, v5.4s",
|
||||
"sqsub v16.4s, v16.4s, v5.4s",
|
||||
"smull v5.4s, v3.4h, v4.4h",
|
||||
"smull2 v3.4s, v3.8h, v4.8h",
|
||||
"addp v3.4s, v5.4s, v3.4s",
|
||||
"neg v3.4s, v3.4s",
|
||||
"sqsub v2.4s, v2.4s, v3.4s",
|
||||
"str q2, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpbroadcastd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Comment": [
|
||||
|
||||
@@ -0,0 +1,131 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"DOTPROD"
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"AFP",
|
||||
"FLAGM",
|
||||
"FLAGM2",
|
||||
"SVE128",
|
||||
"SVE256",
|
||||
"I8MM"
|
||||
],
|
||||
"BinaryCacheVersion": 20
|
||||
},
|
||||
"Instructions": {
|
||||
"vpdpbusd xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 9,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x50 128-bit",
|
||||
"vpdpbusd xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x50, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"movi v2.16b, #0x80",
|
||||
"eor v2.16b, v17.16b, v2.16b",
|
||||
"movi v3.16b, #0x40",
|
||||
"mov v4.16b, v16.16b",
|
||||
"sdot v4.4s, v18.16b, v3.16b",
|
||||
"sdot v4.4s, v18.16b, v3.16b",
|
||||
"mov v16.16b, v4.16b",
|
||||
"sdot v16.4s, v2.16b, v18.16b",
|
||||
"stp xzr, xzr, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpbusd ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x50 256-bit",
|
||||
"vpdpbusd ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x50, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #192]",
|
||||
"ldr q3, [x28, #208]",
|
||||
"ldr q4, [x28, #224]",
|
||||
"movi v5.16b, #0x80",
|
||||
"eor v5.16b, v17.16b, v5.16b",
|
||||
"movi v6.16b, #0x40",
|
||||
"mov v7.16b, v16.16b",
|
||||
"sdot v7.4s, v18.16b, v6.16b",
|
||||
"sdot v7.4s, v18.16b, v6.16b",
|
||||
"mov v16.16b, v7.16b",
|
||||
"sdot v16.4s, v5.16b, v18.16b",
|
||||
"movi v5.16b, #0x80",
|
||||
"eor v3.16b, v3.16b, v5.16b",
|
||||
"movi v5.16b, #0x40",
|
||||
"sdot v2.4s, v4.16b, v5.16b",
|
||||
"sdot v2.4s, v4.16b, v5.16b",
|
||||
"sdot v2.4s, v3.16b, v4.16b",
|
||||
"str q2, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpbusds xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 9,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x51 128-bit",
|
||||
"vpdpbusds xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x51, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"movi v2.16b, #0x80",
|
||||
"eor v2.16b, v17.16b, v2.16b",
|
||||
"movi v3.16b, #0x40",
|
||||
"movi v4.2d, #0x0",
|
||||
"sdot v4.4s, v18.16b, v3.16b",
|
||||
"sdot v4.4s, v18.16b, v3.16b",
|
||||
"sdot v4.4s, v2.16b, v18.16b",
|
||||
"sqadd v16.4s, v16.4s, v4.4s",
|
||||
"stp xzr, xzr, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpbusds ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 20,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x51 256-bit",
|
||||
"vpdpbusds ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x51, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #192]",
|
||||
"ldr q3, [x28, #208]",
|
||||
"ldr q4, [x28, #224]",
|
||||
"movi v5.16b, #0x80",
|
||||
"eor v5.16b, v17.16b, v5.16b",
|
||||
"movi v6.16b, #0x40",
|
||||
"movi v7.2d, #0x0",
|
||||
"mov v8.16b, v7.16b",
|
||||
"sdot v8.4s, v18.16b, v6.16b",
|
||||
"sdot v8.4s, v18.16b, v6.16b",
|
||||
"sdot v8.4s, v5.16b, v18.16b",
|
||||
"sqadd v16.4s, v16.4s, v8.4s",
|
||||
"movi v5.16b, #0x80",
|
||||
"eor v3.16b, v3.16b, v5.16b",
|
||||
"movi v5.16b, #0x40",
|
||||
"sdot v7.4s, v4.16b, v5.16b",
|
||||
"sdot v7.4s, v4.16b, v5.16b",
|
||||
"sdot v7.4s, v3.16b, v4.16b",
|
||||
"sqadd v2.4s, v2.4s, v7.4s",
|
||||
"str q2, [x28, #192]"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,189 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"I8MM"
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"AFP",
|
||||
"FLAGM",
|
||||
"FLAGM2",
|
||||
"SVE128",
|
||||
"SVE256"
|
||||
],
|
||||
"BinaryCacheVersion": 20
|
||||
},
|
||||
"Instructions": {
|
||||
"vpdpbusd xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x50 128-bit",
|
||||
"vpdpbusd xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x50, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"usdot v16.4s, v17.16b, v18.16b",
|
||||
"stp xzr, xzr, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpbusd ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x50 256-bit",
|
||||
"vpdpbusd ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x50, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #192]",
|
||||
"ldr q3, [x28, #208]",
|
||||
"ldr q4, [x28, #224]",
|
||||
"usdot v16.4s, v17.16b, v18.16b",
|
||||
"usdot v2.4s, v3.16b, v4.16b",
|
||||
"str q2, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpbusds xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x51 128-bit",
|
||||
"vpdpbusds xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x51, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"movi v2.2d, #0x0",
|
||||
"usdot v2.4s, v17.16b, v18.16b",
|
||||
"sqadd v16.4s, v16.4s, v2.4s",
|
||||
"stp xzr, xzr, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpbusds ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 10,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x51 256-bit",
|
||||
"vpdpbusds ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x51, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #192]",
|
||||
"ldr q3, [x28, #208]",
|
||||
"ldr q4, [x28, #224]",
|
||||
"movi v5.2d, #0x0",
|
||||
"mov v6.16b, v5.16b",
|
||||
"usdot v6.4s, v17.16b, v18.16b",
|
||||
"sqadd v16.4s, v16.4s, v6.4s",
|
||||
"usdot v5.4s, v3.16b, v4.16b",
|
||||
"sqadd v2.4s, v2.4s, v5.4s",
|
||||
"str q2, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpwssd xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x52 128-bit",
|
||||
"vpdpwssd xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x52, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"smull v2.4s, v17.4h, v18.4h",
|
||||
"smull2 v3.4s, v17.8h, v18.8h",
|
||||
"addp v2.4s, v2.4s, v3.4s",
|
||||
"add v16.4s, v16.4s, v2.4s",
|
||||
"stp xzr, xzr, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpwssd ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 12,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x52 256-bit",
|
||||
"vpdpwssd ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x52, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #192]",
|
||||
"ldr q3, [x28, #208]",
|
||||
"ldr q4, [x28, #224]",
|
||||
"smull v5.4s, v17.4h, v18.4h",
|
||||
"smull2 v6.4s, v17.8h, v18.8h",
|
||||
"addp v5.4s, v5.4s, v6.4s",
|
||||
"add v16.4s, v16.4s, v5.4s",
|
||||
"smull v5.4s, v3.4h, v4.4h",
|
||||
"smull2 v3.4s, v3.8h, v4.8h",
|
||||
"addp v3.4s, v5.4s, v3.4s",
|
||||
"add v2.4s, v2.4s, v3.4s",
|
||||
"str q2, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpwssds xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x53 128-bit",
|
||||
"vpdpwssds xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x53, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"smull v2.4s, v17.4h, v18.4h",
|
||||
"smull2 v3.4s, v17.8h, v18.8h",
|
||||
"addp v2.4s, v2.4s, v3.4s",
|
||||
"neg v2.4s, v2.4s",
|
||||
"sqsub v16.4s, v16.4s, v2.4s",
|
||||
"stp xzr, xzr, [x28, #192]"
|
||||
]
|
||||
},
|
||||
"vpdpwssds ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 14,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x53 256-bit",
|
||||
"vpdpwssds ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x53, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #192]",
|
||||
"ldr q3, [x28, #208]",
|
||||
"ldr q4, [x28, #224]",
|
||||
"smull v5.4s, v17.4h, v18.4h",
|
||||
"smull2 v6.4s, v17.8h, v18.8h",
|
||||
"addp v5.4s, v5.4s, v6.4s",
|
||||
"neg v5.4s, v5.4s",
|
||||
"sqsub v16.4s, v16.4s, v5.4s",
|
||||
"smull v5.4s, v3.4h, v4.4h",
|
||||
"smull2 v3.4s, v3.8h, v4.8h",
|
||||
"addp v3.4s, v5.4s, v3.4s",
|
||||
"neg v3.4s, v3.4s",
|
||||
"sqsub v2.4s, v2.4s, v3.4s",
|
||||
"str q2, [x28, #192]"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -95,8 +95,8 @@
|
||||
"msr nzcv, x0",
|
||||
"and z2.d, z3.d, z2.d",
|
||||
"addp z2.b, p7/m, z2.b, z2.b",
|
||||
"uzp1 z2.b, z2.b, z2.b",
|
||||
"uzp2 z1.b, z2.b, z2.b",
|
||||
"uzp1 z2.b, z2.b, z2.b",
|
||||
"splice z2.d, p6, z2.d, z1.d",
|
||||
"addp v2.16b, v2.16b, v2.16b",
|
||||
"addp v2.8b, v2.8b, v2.8b",
|
||||
|
||||
@@ -4540,8 +4540,8 @@
|
||||
"msr nzcv, x0",
|
||||
"and z2.d, z3.d, z2.d",
|
||||
"addp z2.b, p7/m, z2.b, z2.b",
|
||||
"uzp1 z2.b, z2.b, z2.b",
|
||||
"uzp2 z1.b, z2.b, z2.b",
|
||||
"uzp1 z2.b, z2.b, z2.b",
|
||||
"splice z2.d, p6, z2.d, z1.d",
|
||||
"addp v2.16b, v2.16b, v2.16b",
|
||||
"addp v2.8b, v2.8b, v2.8b",
|
||||
@@ -5357,8 +5357,8 @@
|
||||
"zip2 z3.s, z0.s, z1.s",
|
||||
"movprfx z0, z2",
|
||||
"addp z0.s, p7/m, z0.s, z3.s",
|
||||
"uzp1 z16.s, z0.s, z0.s",
|
||||
"uzp2 z1.s, z0.s, z0.s",
|
||||
"uzp1 z16.s, z0.s, z0.s",
|
||||
"splice z16.d, p6, z16.d, z1.d"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -9,7 +9,9 @@
|
||||
"AFP",
|
||||
"FLAGM",
|
||||
"FLAGM2",
|
||||
"SVEBITPERM"
|
||||
"SVEBITPERM",
|
||||
"I8MM",
|
||||
"DOTPROD"
|
||||
],
|
||||
"BinaryCacheVersion": 20
|
||||
},
|
||||
@@ -59,8 +61,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"movprfx z0, z17",
|
||||
"addp z0.h, p7/m, z0.h, z18.h",
|
||||
"uzp1 z2.h, z0.h, z0.h",
|
||||
"uzp2 z1.h, z0.h, z0.h",
|
||||
"uzp1 z2.h, z0.h, z0.h",
|
||||
"splice z2.d, p6, z2.d, z1.d",
|
||||
"ldr x0, [x28, #2552]",
|
||||
"ld1b {z3.b}, p7/z, [x0]",
|
||||
@@ -84,8 +86,8 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"movprfx z0, z17",
|
||||
"addp z0.s, p7/m, z0.s, z18.s",
|
||||
"uzp1 z2.s, z0.s, z0.s",
|
||||
"uzp2 z1.s, z0.s, z0.s",
|
||||
"uzp1 z2.s, z0.s, z0.s",
|
||||
"splice z2.d, p6, z2.d, z1.d",
|
||||
"ldr x0, [x28, #2552]",
|
||||
"ld1b {z3.b}, p7/z, [x0]",
|
||||
@@ -1639,6 +1641,198 @@
|
||||
"lsl z16.d, p7/m, z16.d, z1.d"
|
||||
]
|
||||
},
|
||||
"vpdpbusd xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 11,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x50 128-bit",
|
||||
"vpdpbusd xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x50, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"uzp1 v2.8h, v17.8h, v18.8h",
|
||||
"uxtl v3.8h, v2.8b",
|
||||
"sxtl2 v2.8h, v2.16b",
|
||||
"mul v2.8h, v3.8h, v2.8h",
|
||||
"uzp2 v3.8h, v17.8h, v18.8h",
|
||||
"uxtl v4.8h, v3.8b",
|
||||
"sxtl2 v3.8h, v3.16b",
|
||||
"mul v3.8h, v4.8h, v3.8h",
|
||||
"saddlp v2.4s, v2.8h",
|
||||
"sadalp v2.4s, v3.8h",
|
||||
"add v16.4s, v16.4s, v2.4s"
|
||||
]
|
||||
},
|
||||
"vpdpbusd ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 13,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x50 256-bit",
|
||||
"vpdpbusd ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x50, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"uzp1 z2.h, z17.h, z18.h",
|
||||
"uunpklo z3.h, z2.b",
|
||||
"sunpkhi z2.h, z2.b",
|
||||
"mul z2.h, z3.h, z2.h",
|
||||
"uzp2 z3.h, z17.h, z18.h",
|
||||
"uunpklo z4.h, z3.b",
|
||||
"sunpkhi z3.h, z3.b",
|
||||
"mul z3.h, z4.h, z3.h",
|
||||
"mov z0.s, #0",
|
||||
"sadalp z0.s, p7/m, z2.h",
|
||||
"mov z2.d, z0.d",
|
||||
"sadalp z2.s, p7/m, z3.h",
|
||||
"add z16.s, z16.s, z2.s"
|
||||
]
|
||||
},
|
||||
"vpdpbusds xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 11,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x51 128-bit",
|
||||
"vpdpbusds xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x51, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"uzp1 v2.8h, v17.8h, v18.8h",
|
||||
"uxtl v3.8h, v2.8b",
|
||||
"sxtl2 v2.8h, v2.16b",
|
||||
"mul v2.8h, v3.8h, v2.8h",
|
||||
"uzp2 v3.8h, v17.8h, v18.8h",
|
||||
"uxtl v4.8h, v3.8b",
|
||||
"sxtl2 v3.8h, v3.16b",
|
||||
"mul v3.8h, v4.8h, v3.8h",
|
||||
"saddlp v2.4s, v2.8h",
|
||||
"sadalp v2.4s, v3.8h",
|
||||
"sqadd v16.4s, v16.4s, v2.4s"
|
||||
]
|
||||
},
|
||||
"vpdpbusds ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 13,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x51 256-bit",
|
||||
"vpdpbusds ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x51, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"uzp1 z2.h, z17.h, z18.h",
|
||||
"uunpklo z3.h, z2.b",
|
||||
"sunpkhi z2.h, z2.b",
|
||||
"mul z2.h, z3.h, z2.h",
|
||||
"uzp2 z3.h, z17.h, z18.h",
|
||||
"uunpklo z4.h, z3.b",
|
||||
"sunpkhi z3.h, z3.b",
|
||||
"mul z3.h, z4.h, z3.h",
|
||||
"mov z0.s, #0",
|
||||
"sadalp z0.s, p7/m, z2.h",
|
||||
"mov z2.d, z0.d",
|
||||
"sadalp z2.s, p7/m, z3.h",
|
||||
"sqadd z16.s, z16.s, z2.s"
|
||||
]
|
||||
},
|
||||
"vpdpwssd xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x52 128-bit",
|
||||
"vpdpwssd xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x52, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"smull v2.4s, v17.4h, v18.4h",
|
||||
"smull2 v3.4s, v17.8h, v18.8h",
|
||||
"addp v2.4s, v2.4s, v3.4s",
|
||||
"add v16.4s, v16.4s, v2.4s"
|
||||
]
|
||||
},
|
||||
"vpdpwssd ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 11,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x52 256-bit",
|
||||
"vpdpwssd ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x52, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"smullb z0.s, z17.h, z18.h",
|
||||
"smullt z1.s, z17.h, z18.h",
|
||||
"zip1 z2.s, z0.s, z1.s",
|
||||
"smullb z0.s, z17.h, z18.h",
|
||||
"smullt z1.s, z17.h, z18.h",
|
||||
"zip2 z3.s, z0.s, z1.s",
|
||||
"addp z2.s, p7/m, z2.s, z3.s",
|
||||
"uzp2 z1.s, z2.s, z2.s",
|
||||
"uzp1 z2.s, z2.s, z2.s",
|
||||
"splice z2.d, p6, z2.d, z1.d",
|
||||
"add z16.s, z16.s, z2.s"
|
||||
]
|
||||
},
|
||||
"vpdpwssds xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x53 128-bit",
|
||||
"vpdpwssds xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x53, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"smull v2.4s, v17.4h, v18.4h",
|
||||
"smull2 v3.4s, v17.8h, v18.8h",
|
||||
"addp v2.4s, v2.4s, v3.4s",
|
||||
"neg v2.4s, v2.4s",
|
||||
"sqsub v16.4s, v16.4s, v2.4s"
|
||||
]
|
||||
},
|
||||
"vpdpwssds ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 12,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x53 256-bit",
|
||||
"vpdpwssds ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x53, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"smullb z0.s, z17.h, z18.h",
|
||||
"smullt z1.s, z17.h, z18.h",
|
||||
"zip1 z2.s, z0.s, z1.s",
|
||||
"smullb z0.s, z17.h, z18.h",
|
||||
"smullt z1.s, z17.h, z18.h",
|
||||
"zip2 z3.s, z0.s, z1.s",
|
||||
"addp z2.s, p7/m, z2.s, z3.s",
|
||||
"uzp2 z1.s, z2.s, z2.s",
|
||||
"uzp1 z2.s, z2.s, z2.s",
|
||||
"splice z2.d, p6, z2.d, z1.d",
|
||||
"neg z2.s, p7/m, z2.s",
|
||||
"sqsub z16.s, z16.s, z2.s"
|
||||
]
|
||||
},
|
||||
"vpbroadcastd xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
|
||||
@@ -0,0 +1,108 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"SVE256",
|
||||
"SVE128",
|
||||
"DOTPROD"
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"AFP",
|
||||
"FLAGM",
|
||||
"FLAGM2",
|
||||
"SVEBITPERM",
|
||||
"I8MM"
|
||||
],
|
||||
"BinaryCacheVersion": 20
|
||||
},
|
||||
"Instructions": {
|
||||
"vpdpbusd xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 8,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x50 128-bit",
|
||||
"vpdpbusd xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x50, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"movi v2.16b, #0x80",
|
||||
"eor v2.16b, v17.16b, v2.16b",
|
||||
"movi v3.16b, #0x40",
|
||||
"mov v4.16b, v16.16b",
|
||||
"sdot v4.4s, v18.16b, v3.16b",
|
||||
"sdot v4.4s, v18.16b, v3.16b",
|
||||
"mov v16.16b, v4.16b",
|
||||
"sdot v16.4s, v2.16b, v18.16b"
|
||||
]
|
||||
},
|
||||
"vpdpbusd ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 8,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x50 256-bit",
|
||||
"vpdpbusd ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x50, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov z2.b, #-128",
|
||||
"eor z2.d, z17.d, z2.d",
|
||||
"mov z3.b, #64",
|
||||
"mov z4.d, z16.d",
|
||||
"sdot z4.s, z18.b, z3.b",
|
||||
"sdot z4.s, z18.b, z3.b",
|
||||
"mov z16.d, z4.d",
|
||||
"sdot z16.s, z2.b, z18.b"
|
||||
]
|
||||
},
|
||||
"vpdpbusds xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 8,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x51 128-bit",
|
||||
"vpdpbusds xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x51, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"movi v2.16b, #0x80",
|
||||
"eor v2.16b, v17.16b, v2.16b",
|
||||
"movi v3.16b, #0x40",
|
||||
"movi v4.2d, #0x0",
|
||||
"sdot v4.4s, v18.16b, v3.16b",
|
||||
"sdot v4.4s, v18.16b, v3.16b",
|
||||
"sdot v4.4s, v2.16b, v18.16b",
|
||||
"sqadd v16.4s, v16.4s, v4.4s"
|
||||
]
|
||||
},
|
||||
"vpdpbusds ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 8,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x51 256-bit",
|
||||
"vpdpbusds ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x51, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"mov z2.b, #-128",
|
||||
"eor z2.d, z17.d, z2.d",
|
||||
"mov z3.b, #64",
|
||||
"movi v4.2d, #0x0",
|
||||
"sdot z4.s, z18.b, z3.b",
|
||||
"sdot z4.s, z18.b, z3.b",
|
||||
"sdot z4.s, z2.b, z18.b",
|
||||
"sqadd z16.s, z16.s, z4.s"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -0,0 +1,171 @@
|
||||
{
|
||||
"Features": {
|
||||
"Bitness": 64,
|
||||
"EnabledHostFeatures": [
|
||||
"SVE256",
|
||||
"SVE128",
|
||||
"I8MM"
|
||||
],
|
||||
"DisabledHostFeatures": [
|
||||
"AFP",
|
||||
"FLAGM",
|
||||
"FLAGM2",
|
||||
"SVEBITPERM"
|
||||
],
|
||||
"BinaryCacheVersion": 20
|
||||
},
|
||||
"Instructions": {
|
||||
"vpdpbusd xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x50 128-bit",
|
||||
"vpdpbusd xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x50, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"usdot v16.4s, v17.16b, v18.16b"
|
||||
]
|
||||
},
|
||||
"vpdpbusd ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x50 256-bit",
|
||||
"vpdpbusd ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x50, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"usdot z16.s, z17.b, z18.b"
|
||||
]
|
||||
},
|
||||
"vpdpbusds xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x51 128-bit",
|
||||
"vpdpbusds xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x51, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"movi v2.2d, #0x0",
|
||||
"usdot v2.4s, v17.16b, v18.16b",
|
||||
"sqadd v16.4s, v16.4s, v2.4s"
|
||||
]
|
||||
},
|
||||
"vpdpbusds ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x51 256-bit",
|
||||
"vpdpbusds ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x51, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"movi v2.2d, #0x0",
|
||||
"usdot z2.s, z17.b, z18.b",
|
||||
"sqadd z16.s, z16.s, z2.s"
|
||||
]
|
||||
},
|
||||
"vpdpwssd xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x52 128-bit",
|
||||
"vpdpwssd xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x52, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"smull v2.4s, v17.4h, v18.4h",
|
||||
"smull2 v3.4s, v17.8h, v18.8h",
|
||||
"addp v2.4s, v2.4s, v3.4s",
|
||||
"add v16.4s, v16.4s, v2.4s"
|
||||
]
|
||||
},
|
||||
"vpdpwssd ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 11,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x52 256-bit",
|
||||
"vpdpwssd ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x52, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"smullb z0.s, z17.h, z18.h",
|
||||
"smullt z1.s, z17.h, z18.h",
|
||||
"zip1 z2.s, z0.s, z1.s",
|
||||
"smullb z0.s, z17.h, z18.h",
|
||||
"smullt z1.s, z17.h, z18.h",
|
||||
"zip2 z3.s, z0.s, z1.s",
|
||||
"addp z2.s, p7/m, z2.s, z3.s",
|
||||
"uzp2 z1.s, z2.s, z2.s",
|
||||
"uzp1 z2.s, z2.s, z2.s",
|
||||
"splice z2.d, p6, z2.d, z1.d",
|
||||
"add z16.s, z16.s, z2.s"
|
||||
]
|
||||
},
|
||||
"vpdpwssds xmm0, xmm1, xmm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x53 128-bit",
|
||||
"vpdpwssds xmm0, xmm1, xmm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x71, 0x53, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"smull v2.4s, v17.4h, v18.4h",
|
||||
"smull2 v3.4s, v17.8h, v18.8h",
|
||||
"addp v2.4s, v2.4s, v3.4s",
|
||||
"neg v2.4s, v2.4s",
|
||||
"sqsub v16.4s, v16.4s, v2.4s"
|
||||
]
|
||||
},
|
||||
"vpdpwssds ymm0, ymm1, ymm2": {
|
||||
"x86InstructionCount": 1,
|
||||
"ExpectedInstructionCount": 12,
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x53 256-bit",
|
||||
"vpdpwssds ymm0, ymm1, ymm2",
|
||||
"NASM only emits the EVEX encoding, hand encode the VEX form"
|
||||
],
|
||||
"x86Insts": [
|
||||
"db 0xc4, 0xe2, 0x75, 0x53, 0xc2"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"smullb z0.s, z17.h, z18.h",
|
||||
"smullt z1.s, z17.h, z18.h",
|
||||
"zip1 z2.s, z0.s, z1.s",
|
||||
"smullb z0.s, z17.h, z18.h",
|
||||
"smullt z1.s, z17.h, z18.h",
|
||||
"zip2 z3.s, z0.s, z1.s",
|
||||
"addp z2.s, p7/m, z2.s, z3.s",
|
||||
"uzp2 z1.s, z2.s, z2.s",
|
||||
"uzp1 z2.s, z2.s, z2.s",
|
||||
"splice z2.d, p6, z2.d, z1.d",
|
||||
"neg z2.s, p7/m, z2.s",
|
||||
"sqsub z16.s, z16.s, z2.s"
|
||||
]
|
||||
}
|
||||
}
|
||||
}
|
||||
Reference in new issue
Block a user