This commit is contained in:
Justin Becker committed 2026-09-10 15:56:03 -07:00
1 parent 794a11833d
commit 89a13cd5fd
28 files changed
+1690 -44

No files matched your search

@@ -91,7 +91,11 @@
"ENABLESSE4A": "enablesse4a",
"DISABLESSE4A": "disablesse4a",
"ENABLEMOPS": "enablemops",
"DISABLEMOPS": "disablemops"
"DISABLEMOPS": "disablemops",
"ENABLEI8MM": "enablei8mm",
"DISABLEI8MM": "disablei8mm",
"ENABLEDOTPROD": "enabledotprod",
"DISABLEDOTPROD": "disabledotprod"
},
"Desc": [
"Allows controlling of the CPU features in the JIT.",
@@ -116,7 +120,9 @@
"\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it",
"\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it",
"\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it",
"\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it"
"\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it",
"\t{enable,disable}i8mm: Will force enable or disable i8mm even if the host doesn't support it",
"\t{enable,disable}dotprod: Will force enable or disable dotprod even if the host doesn't support it"
]
},
"SmallTSCScale": {
+41 -33
View File
@@ -648,6 +648,12 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h(uint32_t Leaf) const {
FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
FEXCore::CPUID::FunctionResults Res {};
// AVX-VNNI is only advertised when the CPU supports I8MM or Dot Product.
// Without these features the implementation is so slow that it is likely
// to harm performance.
const uint32_t SupportsAVXVNNI = SupportsAVX() && (CTX->HostFeatures.SupportsI8MM || CTX->HostFeatures.SupportsDotProd);
if (Leaf == 0) {
#ifndef _WIN32
constexpr uint32_t SUPPORTS_RDPID = 1;
@@ -665,7 +671,9 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
const uint32_t SupportsWFXT = CTX->HostFeatures.SupportsWFXT;
// Number of subfunctions
Res.eax = 0x0;
// TODO: For now, subfunction 1 only exposes AVX-VNNI so we make it conditional
// on AVX-VNNI support. We should revisit this if/when we add more to this leaf.
Res.eax = SupportsAVXVNNI;
Res.ebx = (1 << 0) | // FS/GS support
(0 << 1) | // TSC adjust MSR
(0 << 2) | // SGX
@@ -765,38 +773,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
(0 << 30) | // Arch capabilities - MSR module specific
(0 << 31); // SSBD - Speculative Store Bypass Disable
} else if (Leaf == 1) {
Res.eax = (0U << 0) | // SHA512
(0U << 1) | // SM3
(0U << 2) | // SM4
(0U << 3) | // RAO_INT
(0U << 4) | // AVX_VNNI
(0U << 5) | // AVX512_BF16
(0U << 6) | // LASS (Linear Address Space Separation)
(0U << 7) | // CMPCCXADD
(0U << 8) | // ARCH_PERFMON_EXT
(0U << 9) | // Reserved
(0U << 10) | // FAST_REP_MOVSB
(0U << 11) | // FAST_REP_STOSB
(0U << 12) | // FAST_REP_CMPSB_SCASB
(0U << 13) | // Reserved
(0U << 14) | // Reserved
(0U << 15) | // Reserved
(0U << 16) | // Reserved
(0U << 17) | // FRED (Flexible Return and Event Delivery)
(0U << 18) | // LKGS (Load into Kernel GS Base)
(0U << 19) | // WRMSRNS
(0U << 20) | // NMI_SRC
(0U << 21) | // AMX_FP16
(0U << 22) | // HRESET
(0U << 23) | // AVX_IFMA
(0U << 24) | // Reserved
(0U << 25) | // Reserved
(0U << 26) | // LAM (Linear Address Masking)
(0U << 27) | // MSRLIST
(0U << 28) | // Reserved
(0U << 29) | // Reserved
(0U << 30) | // INVD_DISABLE_POST_BIOS_DONE
(0U << 31); // MOVRS
Res.eax = (0U << 0) | // SHA512
(0U << 1) | // SM3
(0U << 2) | // SM4
(0U << 3) | // RAO_INT
(SupportsAVXVNNI << 4) | // AVX_VNNI
(0U << 5) | // AVX512_BF16
(0U << 6) | // LASS (Linear Address Space Separation)
(0U << 7) | // CMPCCXADD
(0U << 8) | // ARCH_PERFMON_EXT
(0U << 9) | // Reserved
(0U << 10) | // FAST_REP_MOVSB
(0U << 11) | // FAST_REP_STOSB
(0U << 12) | // FAST_REP_CMPSB_SCASB
(0U << 13) | // Reserved
(0U << 14) | // Reserved
(0U << 15) | // Reserved
(0U << 16) | // Reserved
(0U << 17) | // FRED (Flexible Return and Event Delivery)
(0U << 18) | // LKGS (Load into Kernel GS Base)
(0U << 19) | // WRMSRNS
(0U << 20) | // NMI_SRC
(0U << 21) | // AMX_FP16
(0U << 22) | // HRESET
(0U << 23) | // AVX_IFMA
(0U << 24) | // Reserved
(0U << 25) | // Reserved
(0U << 26) | // LAM (Linear Address Masking)
(0U << 27) | // MSRLIST
(0U << 28) | // Reserved
(0U << 29) | // Reserved
(0U << 30) | // INVD_DISABLE_POST_BIOS_DONE
(0U << 31); // MOVRS
// Bits 4-31 currently reserved.
Res.ebx = (0U << 0) | // PPIN
+168 -1
View File
@@ -1112,8 +1112,10 @@ DEF_OP(VAddP) {
// pairwise addition, the SVE version actually interleaves the
// results of the pairwise addition (gross!), so we need to undo that.
addp(SubRegSize, LHS.Z(), Pred, LHS.Z(), VectorUpper.Z());
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
// Extract the upper half first, since Dst may alias the LHS.
uzp2(SubRegSize, VTMP2.Z(), LHS.Z(), LHS.Z());
uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z());
// Merge upper half with lower half.
splice<ARMEmitter::OpType::Destructive>(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z());
@@ -3833,6 +3835,171 @@ DEF_OP(VMul) {
}
}
DEF_OP(VUSDot) {
///< Dest = Acc + dot(Vector1 (unsigned 8-bit), Vector2 (signed 8-bit))
// Matches:
// - SVE - USDOT
// - ASIMD - USDOT
const auto Op = IROp->C<IR::IROp_VUSDot>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Acc = GetVReg(Op->Acc);
const auto Vector1 = GetVReg(Op->Vector1);
const auto Vector2 = GetVReg(Op->Vector2);
// USDOT accumulates in to its destination register,
// so we need to emit a move if Acc != Dst
ARMEmitter::VRegister DestTmp = Dst;
if (Dst != Acc) {
if (Dst != Vector1 && Dst != Vector2) {
DestTmp = Dst;
} else {
DestTmp = VTMP1;
}
}
if (HostSupportsSVE256 && Is256Bit) {
if (Dst != Acc) {
mov(DestTmp.Z(), Acc.Z());
}
usdot(DestTmp.Z(), Vector1.Z(), Vector2.Z());
if (Dst != DestTmp) {
mov(Dst.Z(), DestTmp.Z());
}
} else {
if (Dst != Acc) {
mov(DestTmp.Q(), Acc.Q());
}
usdot(DestTmp.Q(), Vector1.Q(), Vector2.Q());
if (Dst != DestTmp) {
mov(Dst.Q(), DestTmp.Q());
}
}
}
DEF_OP(VSDot) {
///< Dest = Acc + dot(Vector1 (signed 8-bit), Vector2 (signed 8-bit))
// Matches:
// - SVE - SDOT
// - ASIMD - SDOT
const auto Op = IROp->C<IR::IROp_VSDot>();
const auto OpSize = IROp->Size;
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Acc = GetVReg(Op->Acc);
const auto Vector1 = GetVReg(Op->Vector1);
const auto Vector2 = GetVReg(Op->Vector2);
// SDOT accumulates in to its destination register,
// so we need to emit a move if Acc != Dst
ARMEmitter::VRegister DestTmp = Dst;
if (Dst != Acc) {
if (Dst != Vector1 && Dst != Vector2) {
DestTmp = Dst;
} else {
DestTmp = VTMP1;
}
}
if (HostSupportsSVE256 && Is256Bit) {
if (Dst != Acc) {
mov(DestTmp.Z(), Acc.Z());
}
sdot(ARMEmitter::SubRegSize::i32Bit, DestTmp.Z(), Vector1.Z(), Vector2.Z());
if (Dst != DestTmp) {
mov(Dst.Z(), DestTmp.Z());
}
} else {
if (Dst != Acc) {
mov(DestTmp.Q(), Acc.Q());
}
sdot(ARMEmitter::SubRegSize::i32Bit, DestTmp.Q(), Vector1.Q(), Vector2.Q());
if (Dst != DestTmp) {
mov(Dst.Q(), DestTmp.Q());
}
}
}
DEF_OP(VSAddLP) {
const auto Op = IROp->C<IR::IROp_VSAddLP>();
const auto OpSize = IROp->Size;
const auto SubRegSize = ConvertSubRegSize248(IROp);
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Vector = GetVReg(Op->Vector);
if (HostSupportsSVE256 && Is256Bit) {
// SVE only has the accumulating form, so accumulate in to a zeroed register.
// Zero a temporary instead if Dst aliases the source.
const auto DestTmp = Dst == Vector ? VTMP1 : Dst;
dup_imm(SubRegSize, DestTmp.Z(), 0);
sadalp(SubRegSize, DestTmp.Z(), PRED_TMP_32B.Merging(), Vector.Z());
if (Dst != DestTmp) {
mov(Dst.Z(), DestTmp.Z());
}
} else {
saddlp(SubRegSize, Dst.Q(), Vector.Q());
}
}
DEF_OP(VSAdALP) {
const auto Op = IROp->C<IR::IROp_VSAdALP>();
const auto OpSize = IROp->Size;
const auto SubRegSize = ConvertSubRegSize248(IROp);
const auto Is256Bit = OpSize == IR::OpSize::i256Bit;
LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__);
const auto Dst = GetVReg(Node);
const auto Acc = GetVReg(Op->Acc);
const auto Vector = GetVReg(Op->Vector);
// SADALP accumulates in to its destination register,
// so we need to emit a move if Acc != Dst
ARMEmitter::VRegister DestTmp = Dst;
if (Dst != Acc) {
DestTmp = Dst != Vector ? Dst : VTMP1;
}
if (HostSupportsSVE256 && Is256Bit) {
if (Dst != Acc) {
mov(DestTmp.Z(), Acc.Z());
}
sadalp(SubRegSize, DestTmp.Z(), PRED_TMP_32B.Merging(), Vector.Z());
if (Dst != DestTmp) {
mov(Dst.Z(), DestTmp.Z());
}
} else {
if (Dst == Vector) {
// ASIMD has the non-accumulating form, which is cheaper than shuffling through a temporary.
saddlp(SubRegSize, VTMP1.Q(), Vector.Q());
add(SubRegSize, Dst.Q(), Acc.Q(), VTMP1.Q());
return;
}
if (Dst != Acc) {
mov(Dst.Q(), Acc.Q());
}
sadalp(SubRegSize, Dst.Q(), Vector.Q());
}
}
DEF_OP(VUMull) {
const auto Op = IROp->C<IR::IROp_VUMull>();
const auto OpSize = IROp->Size;
@@ -665,6 +665,8 @@ public:
void VPMADDUBSWOp(OpcodeArgs);
void VPMADDWDOp(OpcodeArgs);
void VPDPBUSDOp(OpcodeArgs, bool Saturating);
void VPDPWSSDOp(OpcodeArgs, bool Saturating);
void VPMASKMOVOp(OpcodeArgs, bool IsStore);
@@ -1036,6 +1038,9 @@ public:
void AVX128_VPMADDUBSW(OpcodeArgs);
void AVX128_VPMADDWD(OpcodeArgs);
void AVX128_VPDPImpl(OpcodeArgs, std::function<Ref(Ref Acc, Ref Src1, Ref Src2)> Helper);
void AVX128_VPDPBUSD(OpcodeArgs, bool Saturating);
void AVX128_VPDPWSSD(OpcodeArgs, bool Saturating);
void AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize);
@@ -1400,6 +1405,10 @@ private:
Ref PMADDUBSWOpImpl(IR::OpSize Size, Ref Src1, Ref Src2);
Ref VPDPBUSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating);
Ref VPDPWSSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating);
Ref PMULHRSWOpImpl(OpSize Size, Ref Src1, Ref Src2);
Ref PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2);
@@ -1470,6 +1470,35 @@ void OpDispatchBuilder::AVX128_VPMADDWD(OpcodeArgs) {
[this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return PMADDWDOpImpl(OpSize::i128Bit, Src1, Src2); });
}
void OpDispatchBuilder::AVX128_VPDPImpl(OpcodeArgs, std::function<Ref(Ref Acc, Ref Src1, Ref Src2)> Helper) {
const auto Size = OpSizeFromDst(Op);
const auto Is128Bit = Size == OpSize::i128Bit;
auto Acc = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, !Is128Bit);
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit);
RefPair Result {};
Result.Low = Helper(Acc.Low, Src1.Low, Src2.Low);
if (Is128Bit) {
Result.High = LoadZeroVector(OpSize::i128Bit);
} else {
Result.High = Helper(Acc.High, Src1.High, Src2.High);
}
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
}
void OpDispatchBuilder::AVX128_VPDPBUSD(OpcodeArgs, bool Saturating) {
AVX128_VPDPImpl(Op,
[this, Saturating](Ref Acc, Ref Src1, Ref Src2) { return VPDPBUSDOpImpl(OpSize::i128Bit, Acc, Src1, Src2, Saturating); });
}
void OpDispatchBuilder::AVX128_VPDPWSSD(OpcodeArgs, bool Saturating) {
AVX128_VPDPImpl(Op,
[this, Saturating](Ref Acc, Ref Src1, Ref Src2) { return VPDPWSSDOpImpl(OpSize::i128Bit, Acc, Src1, Src2, Saturating); });
}
void OpDispatchBuilder::AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize) {
const auto SrcSize = OpSizeFromSrc(Op);
const auto Is128Bit = SrcSize == OpSize::i128Bit;
@@ -3760,6 +3760,84 @@ void OpDispatchBuilder::VPMADDUBSWOp(OpcodeArgs) {
StoreResultFPR(Op, Result);
}
Ref OpDispatchBuilder::VPDPBUSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating) {
// Does four 8-bit unsigned * signed byte multiplies per 32-bit element, sums them and accumulates in to the destination
if (CTX->HostFeatures.SupportsI8MM) {
// The I8MM extension maps onto VPDP* almost directly.
if (!Saturating) {
return _VUSDot(Size, Acc, Src1, Src2);
}
auto DotProduct = _VUSDot(Size, LoadZeroVector(Size), Src1, Src2);
return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct);
}
if (CTX->HostFeatures.SupportsDotProd) {
// VSDOT assumes signed input, so we need to convert Src1 to signed, and then
// perform correction afterwards using 0x40 to account for the unsigned input.
auto Src1Signed = _VXor(Size, Src1, _VectorImm(Size, OpSize::i8Bit, 0x80));
auto SixtyFour = _VectorImm(Size, OpSize::i8Bit, 0x40);
auto DotProduct = _VSDot(Size, Saturating ? LoadZeroVector(Size) : Acc, Src2, SixtyFour);
DotProduct = _VSDot(Size, DotProduct, Src2, SixtyFour);
DotProduct = _VSDot(Size, DotProduct, Src1Signed, Src2);
if (Saturating) {
return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct);
}
return DotProduct;
}
// Naive software implementation.
auto Even = _VUnZip(Size, OpSize::i16Bit, Src1, Src2);
auto Even1_16b = _VUXTL(Size, OpSize::i8Bit, Even);
auto Even2_16b = _VSXTL2(Size, OpSize::i8Bit, Even);
auto ResMul_Even = _VMul(Size, OpSize::i16Bit, Even1_16b, Even2_16b);
auto Odd = _VUnZip2(Size, OpSize::i16Bit, Src1, Src2);
auto Odd1_16b = _VUXTL(Size, OpSize::i8Bit, Odd);
auto Odd2_16b = _VSXTL2(Size, OpSize::i8Bit, Odd);
auto ResMul_Odd = _VMul(Size, OpSize::i16Bit, Odd1_16b, Odd2_16b);
auto DotProduct = _VSAdALP(Size, OpSize::i16Bit, _VSAddLP(Size, OpSize::i16Bit, ResMul_Even), ResMul_Odd);
if (Saturating) {
return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct);
}
return _VAdd(Size, OpSize::i32Bit, Acc, DotProduct);
}
Ref OpDispatchBuilder::VPDPWSSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating) {
auto DotProduct = PMADDWDOpImpl(Size, Src1, Src2);
if (!Saturating) {
return _VAdd(Size, OpSize::i32Bit, Acc, DotProduct);
}
auto NegDotProduct = _VNeg(Size, OpSize::i32Bit, DotProduct);
return _VSQSub(Size, OpSize::i32Bit, Acc, NegDotProduct);
}
void OpDispatchBuilder::VPDPBUSDOp(OpcodeArgs, bool Saturating) {
const auto Size = OpSizeFromSrc(Op);
Ref Acc = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref Result = VPDPBUSDOpImpl(Size, Acc, Src1, Src2, Saturating);
StoreResultFPR(Op, Result);
}
void OpDispatchBuilder::VPDPWSSDOp(OpcodeArgs, bool Saturating) {
const auto Size = OpSizeFromSrc(Op);
Ref Acc = LoadSourceFPR(Op, Op->Dest, Op->Flags);
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
Ref Result = VPDPWSSDOpImpl(Size, Acc, Src1, Src2, Saturating);
StoreResultFPR(Op, Result);
}
Ref OpDispatchBuilder::PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2) {
const auto Size = OpSizeFromSrc(Op);
if (Signed) {
@@ -312,6 +312,11 @@ namespace AVX128 {
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VSSHR>}, // VPSRAVD
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VUSHL>}, // VPSLLV
{OPD(2, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPBUSD, false>},
{OPD(2, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPBUSD, true>},
{OPD(2, 0b01, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPWSSD, false>},
{OPD(2, 0b01, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPWSSD, true>},
{OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i32Bit>},
{OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i64Bit>},
{OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i128Bit>},
@@ -757,6 +762,11 @@ namespace AVX256 {
{OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::VPSRAVDOp},
{OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::VPSLLVOp},
{OPD(2, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPBUSDOp, false>},
{OPD(2, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPBUSDOp, true>},
{OPD(2, 0b01, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPWSSDOp, false>},
{OPD(2, 0b01, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPWSSDOp, true>},
{OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i32Bit>},
{OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i64Bit>},
{OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i128Bit>},
@@ -1207,6 +1217,11 @@ auto BaseTableLambda = [](const auto RuntimeTable) consteval {
{OPD(2, 0b01, 0x46), 1, X86InstInfo{"VPSRAVD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x47), 1, X86InstInfo{"VPSLLV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x50), 1, X86InstInfo{"VPDPBUSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x51), 1, X86InstInfo{"VPDPBUSDS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x52), 1, X86InstInfo{"VPDPWSSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x53), 1, X86InstInfo{"VPDPWSSDS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x58), 1, X86InstInfo{"VPBROADCASTD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x59), 1, X86InstInfo{"VPBROADCASTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
{OPD(2, 0b01, 0x5A), 1, X86InstInfo{"VBROADCASTI128", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_L_1 | FLAGS_SF_MOD_MEM_ONLY | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}},
+37
View File
@@ -2363,6 +2363,43 @@
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
"FPR = VUSDot OpSize:#RegisterSize, FPR:$Acc, FPR:$Vector1, FPR:$Vector2": {
"Desc": ["Unsigned by signed 8-bit dot product, accumulating four products in to each 32-bit element of Acc.",
"Requires FEAT_I8MM."
],
"DestSize": "RegisterSize",
"ElementSize": "OpSize::i32Bit",
"TiedSource": 0,
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
]
},
"FPR = VSDot OpSize:#RegisterSize, FPR:$Acc, FPR:$Vector1, FPR:$Vector2": {
"Desc": ["Signed 8-bit dot product, accumulating four products in to each 32-bit element of Acc.",
"Requires FEAT_DotProd."
],
"DestSize": "RegisterSize",
"ElementSize": "OpSize::i32Bit",
"TiedSource": 0,
"EmitValidation": [
"RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit"
]
},
"FPR = VSAddLP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": {
"Desc": ["Signed add long pairwise. Adds adjacent pairs of elements in to elements of twice the size.",
"ElementSize is the source size"
],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1"
},
"FPR = VSAdALP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Acc, FPR:$Vector": {
"Desc": ["Signed add and accumulate long pairwise. Adds adjacent pairs of elements in to the elements of Acc, which are twice the size.",
"ElementSize is the source size"
],
"DestSize": "RegisterSize",
"ElementSize": "ElementSize << 1",
"TiedSource": 0
},
"FPR = VUMulH OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"Desc": ["Wide unsigned multiply returning the high results"],
"DestSize": "RegisterSize",
+3 -1
View File
@@ -71,13 +71,15 @@ struct HostFeatures {
uint32_t Supports3DNow : 1 {};
uint32_t SupportsSSE4a : 1 {};
uint32_t SupportsMOPS : 1 {};
uint32_t SupportsI8MM : 1 {};
uint32_t SupportsDotProd : 1 {};
uint32_t PreferZVAForVZero : 1 {};
uint32_t SupportsAFP : 1 {};
uint32_t SupportsFloatExceptions : 1 {};
// Flag if this is InstCountCI
uint32_t IsInstCountCI : 1 {};
HostTypeEnum HostType : 2 {};
uint32_t pad : 26 {};
uint32_t pad : 24 {};
// MIDR information
// Also used for determining number of CPU cores for CPUID
+4
View File
@@ -60,6 +60,8 @@ class HostFeatures(Flag) :
FEATURE_LRCPC2 = (1 << 15)
FEATURE_FRINTTS = (1 << 16)
FEATURE_MOPS = (1 << 17)
FEATURE_I8MM = (1 << 18)
FEATURE_DOTPROD = (1 << 19)
HostFeaturesLookup = {
"SVE128" : HostFeatures.FEATURE_SVE128,
@@ -80,6 +82,8 @@ HostFeaturesLookup = {
"LRCPC2" : HostFeatures.FEATURE_LRCPC2,
"FRINTTS" : HostFeatures.FEATURE_FRINTTS,
"MOPS" : HostFeatures.FEATURE_MOPS,
"I8MM" : HostFeatures.FEATURE_I8MM,
"DOTPROD" : HostFeatures.FEATURE_DOTPROD,
}
def GetHostFeatures(data):
+2
View File
@@ -87,6 +87,7 @@ class HostFeatures(Flag) :
FEATURE_CLFLOPT = (1 << 21)
FEATURE_FSGSBASE = (1 << 22)
FEATURE_EMMI = (1 << 23)
FEATURE_AVX_VNNI = (1 << 24)
RegStringLookup = {
"NONE": Regs.REG_NONE,
@@ -173,6 +174,7 @@ HostFeaturesLookup = {
"CLFLOPT" : HostFeatures.FEATURE_CLFLOPT,
"FSGSBASE" : HostFeatures.FEATURE_FSGSBASE,
"EMMI" : HostFeatures.FEATURE_EMMI,
"AVX_VNNI" : HostFeatures.FEATURE_AVX_VNNI,
}
def parse_hexstring(s):
+4
View File
@@ -374,6 +374,8 @@ static void OverrideFeatures(FEXCore::HostFeatures* Features, uint64_t ForceSVEW
ENABLE_DISABLE_OPTION(Supports3DNow, 3DNOW, 3DNOW);
ENABLE_DISABLE_OPTION(SupportsSSE4a, SSE4A, SSE4A);
ENABLE_DISABLE_OPTION(SupportsMOPS, MOPS, MOPS);
ENABLE_DISABLE_OPTION(SupportsI8MM, I8MM, I8MM);
ENABLE_DISABLE_OPTION(SupportsDotProd, DOTPROD, DOTPROD);
GET_SINGLE_OPTION(Crypto, CRYPTO);
#undef ENABLE_DISABLE_OPTION
@@ -504,6 +506,8 @@ void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFe
HostFeatures.SupportsSVEBitPerm = Features.Supports(CPUFeatures::Feature::SVE_BitPerm);
HostFeatures.SupportsECV = Features.Supports(CPUFeatures::Feature::ECV);
HostFeatures.SupportsWFXT = Features.Supports(CPUFeatures::Feature::WFxt);
HostFeatures.SupportsI8MM = Features.Supports(CPUFeatures::Feature::I8MM);
HostFeatures.SupportsDotProd = Features.Supports(CPUFeatures::Feature::DotProd);
#ifdef VIXL_SIMULATOR
// Hardcode enable SVE with 256-bit wide registers.
+6
View File
@@ -40,6 +40,11 @@ public:
Feat_vaes = data_7.ecx & (1U << 9);
Feat_pclmulqdq = Feat_pclmulqdq && (data_7.ecx & (1U << 10));
Feat_rdpid = data_7.ecx & (1U << 22);
if (data_7.eax >= 1) {
auto data_7_1 = cpuid(0x7, 0x1);
Feat_avx_vnni = Feat_avx && (data_7_1.eax & (1U << 4));
}
}
data = cpuid(0x8000'0000U);
@@ -79,6 +84,7 @@ public:
bool Feat_rdpid {};
bool Feat_clflopt {};
bool Feat_fsgsbase {};
bool Feat_avx_vnni {};
private:
struct cpuid_data {
+14
View File
@@ -557,6 +557,8 @@ int main(int argc, char** argv, char** const envp) {
FEATURE_LRCPC2 = (1U << 15),
FEATURE_FRINTTS = (1U << 16),
FEATURE_MOPS = (1U << 17),
FEATURE_I8MM = (1U << 18),
FEATURE_DOTPROD = (1U << 19),
};
uint64_t SVEWidth = 0;
@@ -610,6 +612,12 @@ int main(int argc, char** argv, char** const envp) {
if (TestHeaderData->EnabledHostFeatures & FEATURE_MOPS) {
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEMOPS);
}
if (TestHeaderData->EnabledHostFeatures & FEATURE_I8MM) {
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEI8MM);
}
if (TestHeaderData->EnabledHostFeatures & FEATURE_DOTPROD) {
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::ENABLEDOTPROD);
}
if (TestHeaderData->EnabledHostFeatures & FEATURE_TSO) {
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "1");
@@ -668,6 +676,12 @@ int main(int argc, char** argv, char** const envp) {
if (TestHeaderData->DisabledHostFeatures & FEATURE_MOPS) {
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEMOPS);
}
if (TestHeaderData->DisabledHostFeatures & FEATURE_I8MM) {
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEI8MM);
}
if (TestHeaderData->DisabledHostFeatures & FEATURE_DOTPROD) {
HostFeatureControl |= static_cast<uint64_t>(FEXCore::Config::HostFeatures::DISABLEDOTPROD);
}
if (TestHeaderData->DisabledHostFeatures & FEATURE_TSO) {
FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "0");
@@ -321,6 +321,7 @@ public:
FEATURE_CLFLOPT = (1 << 21),
FEATURE_FSGSBASE = (1 << 22),
FEATURE_EMMI = (1 << 23),
FEATURE_AVX_VNNI = (1 << 24),
};
bool Requires3DNow() const {
@@ -395,6 +396,9 @@ public:
bool RequiresEMMI() const {
return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_EMMI;
}
bool RequiresAVXVNNI() const {
return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_AVX_VNNI;
}
private:
FEX_CONFIG_OPT(ConfigDumpGPRs, DUMPGPRS);
@@ -602,6 +606,9 @@ public:
bool RequiresEMMI() const {
return Config.RequiresEMMI();
}
bool RequiresAVXVNNI() const {
return Config.RequiresAVXVNNI();
}
private:
constexpr static uint64_t STACK_OFFSET = 0xc000'0000;
@@ -287,6 +287,7 @@ int main(int argc, char** argv, char** const envp) {
const bool SupportsRDPID = Feature.Feat_rdpid;
const bool SupportsCLFLOPT = Feature.Feat_clflopt;
const bool SupportsFSGSBase = Feature.Feat_fsgsbase;
const bool SupportsAVXVNNI = Feature.Feat_avx_vnni;
TestUnsupported |=
(!Supports3DNow && Loader.Requires3DNow()) || (!SupportsSSE4A && Loader.RequiresSSE4A()) || (!SupportsBMI1 && Loader.RequiresBMI1()) ||
@@ -295,6 +296,7 @@ int main(int argc, char** argv, char** const envp) {
(!SupportsAES && Loader.RequiresAES()) || (!SupportsPCLMUL && Loader.RequiresPCLMUL()) || (!SupportsMOVBE && Loader.RequiresMOVBE()) ||
(!SupportsADX && Loader.RequiresADX()) || (!SupportsXSAVE && Loader.RequiresXSAVE()) || (!SupportsRDPID && Loader.RequiresRDPID()) ||
(!SupportsCLFLOPT && Loader.RequiresCLFLOPT()) || (!SupportsFSGSBase && Loader.RequiresFSGSBase()) || Loader.RequiresEMMI();
TestUnsupported |= !SupportsAVXVNNI && Loader.RequiresAVXVNNI();
#endif
#ifdef _WIN32
+59
View File
@@ -0,0 +1,59 @@
%ifdef CONFIG
{
"HostFeatures": ["AVX", "AVX_VNNI"],
"RegData": {
"XMM4": ["0xFFFFFF3100000006", "0x7FFF07008001F904", "0x0000000000000000", "0x0000000000000000"],
"XMM5": ["0xFFFFFFFF00000020", "0x7FFFC08080013B03", "0x0000000000000000", "0x0000000000000000"],
"XMM6": ["0x02FE0140FC03FDF7", "0x807F86807F817983", "0xFCFDFD0101FF8000", "0x807E82807F817983"],
"XMM7": ["0xFFFFFF3100000006", "0x7FFF07008001F904", "0x87654123123456F8", "0x7FFE02108001F9F4"]
}
}
%endif
vmovdqa ymm0, [rel src1]
vmovdqa ymm1, [rel src2]
vmovdqa ymm4, [rel accumulator]
vmovdqa ymm5, [rel accumulator]
vmovdqa ymm6, [rel src2]
vmovdqa ymm7, [rel accumulator]
lea rdx, [rel src2]
; Hand assembled instructions, since NASM will only output
; the EVEX encodings for these instructions.
db 0xC4, 0xE2, 0x79, 0x50, 0xE1 ; vpdpbusd xmm4, xmm0, xmm1
db 0xC4, 0xE2, 0x51, 0x50, 0x2A ; vpdpbusd xmm5, xmm5, [rdx]
db 0xC4, 0xE2, 0x7D, 0x50, 0xF6 ; vpdpbusd ymm6, ymm0, ymm6
db 0xC4, 0xE2, 0x7D, 0x50, 0x3A ; vpdpbusd ymm7, ymm0, [rdx]
hlt
align 32
accumulator:
dd 0x00000010
dd 0xFFFFFFF0
dd 0x7FFFFF00
dd 0x80000100
dd 0x12345678
dd 0x87654321
dd 0x7FFFFFF0
dd 0x80000010
src1:
db 1, 2, 3, 4
db 255, 128, 64, 32
db 255, 255, 255, 255
db 200, 150, 100, 50
db 0, 1, 254, 255
db 17, 34, 51, 68
db 255, 255, 255, 255
db 255, 255, 255, 255
src2:
db 1, -2, 3, -4
db -1, 1, -2, 2
db 127, 127, 127, 127
db -128, -128, -128, -128
db -128, 127, -1, 1
db -1, -2, -3, -4
db 127, 127, 127, 127
db -128, -128, -128, -128
+59
View File
@@ -0,0 +1,59 @@
%ifdef CONFIG
{
"HostFeatures": ["AVX", "AVX_VNNI"],
"RegData": {
"XMM4": ["0xFFFFFF3100000006", "0x800000007FFFFFFF", "0x0000000000000000", "0x0000000000000000"],
"XMM5": ["0xFFFFFFFF00000020", "0x800000007FFFFFFF", "0x0000000000000000", "0x0000000000000000"],
"XMM6": ["0x02FE0140FC03FDF7", "0x807F86807F817983", "0xFCFDFD0101FF8000", "0x807E82807F817983"],
"XMM7": ["0xFFFFFF3100000006", "0x800000007FFFFFFF", "0x87654123123456F8", "0x800000007FFFFFFF"]
}
}
%endif
vmovdqa ymm0, [rel src1]
vmovdqa ymm1, [rel src2]
vmovdqa ymm4, [rel accumulator]
vmovdqa ymm5, [rel accumulator]
vmovdqa ymm6, [rel src2]
vmovdqa ymm7, [rel accumulator]
lea rdx, [rel src2]
; Hand assembled instructions, since NASM will only output
; the EVEX encodings for these instructions.
db 0xC4, 0xE2, 0x79, 0x51, 0xE1 ; vpdpbusds xmm4, xmm0, xmm1
db 0xC4, 0xE2, 0x51, 0x51, 0x2A ; vpdpbusds xmm5, xmm5, [rdx]
db 0xC4, 0xE2, 0x7D, 0x51, 0xF6 ; vpdpbusds ymm6, ymm0, ymm6
db 0xC4, 0xE2, 0x7D, 0x51, 0x3A ; vpdpbusds ymm7, ymm0, [rdx]
hlt
align 32
accumulator:
dd 0x00000010
dd 0xFFFFFFF0
dd 0x7FFFFF00
dd 0x80000100
dd 0x12345678
dd 0x87654321
dd 0x7FFFFFF0
dd 0x80000010
src1:
db 1, 2, 3, 4
db 255, 128, 64, 32
db 255, 255, 255, 255
db 200, 150, 100, 50
db 0, 1, 254, 255
db 17, 34, 51, 68
db 255, 255, 255, 255
db 255, 255, 255, 255
src2:
db 1, -2, 3, -4
db -1, 1, -2, 2
db 127, 127, 127, 127
db -128, -128, -128, -128
db -128, 127, -1, 1
db -1, -2, -3, -4
db 127, 127, 127, 127
db -128, -128, -128, -128
+59
View File
@@ -0,0 +1,59 @@
%ifdef CONFIG
{
"HostFeatures": ["AVX", "AVX_VNNI"],
"RegData": {
"XMM4": ["0x800000000000000B", "0xFFFF000180000002", "0x0000000000000000", "0x0000000000000000"],
"XMM5": ["0x0000000000000040", "0x400080000002FFFE", "0x0000000000000000", "0x0000000000000000"],
"XMM6": ["0x00008000FFFBFFFE", "0xFFFE8001FFFD8001", "0x800100001C173F20", "0x0001800000008000"],
"XMM7": ["0x800000000000000B", "0xFFFF000180000002", "0x8765C322E02B0AC8", "0x000100107FFFFFFE"]
}
}
%endif
vmovdqa ymm0, [rel src1]
vmovdqa ymm1, [rel src2]
vmovdqa ymm4, [rel accumulator]
vmovdqa ymm5, [rel accumulator]
vmovdqa ymm6, [rel src2]
vmovdqa ymm7, [rel accumulator]
lea rdx, [rel src2]
; Hand assembled instructions, since NASM will only output
; the EVEX encodings for these instructions
db 0xC4, 0xE2, 0x79, 0x52, 0xE1 ; vpdpwssd xmm4, xmm0, xmm1
db 0xC4, 0xE2, 0x51, 0x52, 0x2A ; vpdpwssd xmm5, xmm5, [rdx]
db 0xC4, 0xE2, 0x7D, 0x52, 0xF6 ; vpdpwssd ymm6, ymm0, ymm6
db 0xC4, 0xE2, 0x7D, 0x52, 0x3A ; vpdpwssd ymm7, ymm0, [rdx]
hlt
align 32
accumulator:
dd 0x00000010
dd 0x00000000
dd 0x00020000
dd 0x80000000
dd 0x12345678
dd 0x87654321
dd 0xFFFFFFFE
dd 0x80000010
src1:
dw 1, 2
dw -32768, -32768
dw 32767, 32767
dw -32768, 32767
dw 12345, -23456
dw -1, -2
dw -32768, -32768
dw 32767, 32767
src2:
dw 3, -4
dw -32768, -32768
dw 32767, 32767
dw -32768, 32767
dw -30000, 20000
dw 32767, -32768
dw -32768, -32768
dw -32768, -32768
+59
View File
@@ -0,0 +1,59 @@
%ifdef CONFIG
{
"HostFeatures": ["AVX", "AVX_VNNI"],
"RegData": {
"XMM4": ["0x7FFFFFFF0000000B", "0xFFFF00017FFFFFFF", "0x0000000000000000", "0x0000000000000000"],
"XMM5": ["0x0000000000000040", "0x800000000002FFFE", "0x0000000000000000", "0x0000000000000000"],
"XMM6": ["0x00008000FFFBFFFE", "0x7FFFFFFF7FFFFFFF", "0x800100001C173F20", "0x8000000000008000"],
"XMM7": ["0x7FFFFFFF0000000B", "0xFFFF00017FFFFFFF", "0x8765C322E02B0AC8", "0x800000007FFFFFFE"]
}
}
%endif
vmovdqa ymm0, [rel src1]
vmovdqa ymm1, [rel src2]
vmovdqa ymm4, [rel accumulator]
vmovdqa ymm5, [rel accumulator]
vmovdqa ymm6, [rel src2]
vmovdqa ymm7, [rel accumulator]
lea rdx, [rel src2]
; Hand assembled instructions, since NASM will only output
; the EVEX encodings for these instructions.
db 0xC4, 0xE2, 0x79, 0x53, 0xE1 ; vpdpwssds xmm4, xmm0, xmm1
db 0xC4, 0xE2, 0x51, 0x53, 0x2A ; vpdpwssds xmm5, xmm5, [rdx]
db 0xC4, 0xE2, 0x7D, 0x53, 0xF6 ; vpdpwssds ymm6, ymm0, ymm6
db 0xC4, 0xE2, 0x7D, 0x53, 0x3A ; vpdpwssds ymm7, ymm0, [rdx]
hlt
align 32
accumulator:
dd 0x00000010
dd 0x00000000
dd 0x00020000
dd 0x80000000
dd 0x12345678
dd 0x87654321
dd 0xFFFFFFFE
dd 0x80000010
src1:
dw 1, 2
dw -32768, -32768
dw 32767, 32767
dw -32768, 32767
dw 12345, -23456
dw -1, -2
dw -32768, -32768
dw 32767, 32767
src2:
dw 3, -4
dw -32768, -32768
dw 32767, 32767
dw -32768, 32767
dw -30000, 20000
dw 32767, -32768
dw -32768, -32768
dw -32768, -32768
@@ -7,7 +7,9 @@
"FLAGM",
"FLAGM2",
"SVE128",
"SVE256"
"SVE256",
"I8MM",
"DOTPROD"
],
"BinaryCacheVersion": 20
},
@@ -2075,6 +2077,231 @@
"str q2, [x28, #192]"
]
},
"vpdpbusd xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 12,
"Comment": [
"Map 2 0b01 0x50 128-bit",
"vpdpbusd xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x50, 0xc2"
],
"ExpectedArm64ASM": [
"uzp1 v2.8h, v17.8h, v18.8h",
"uxtl v3.8h, v2.8b",
"sxtl2 v2.8h, v2.16b",
"mul v2.8h, v3.8h, v2.8h",
"uzp2 v3.8h, v17.8h, v18.8h",
"uxtl v4.8h, v3.8b",
"sxtl2 v3.8h, v3.16b",
"mul v3.8h, v4.8h, v3.8h",
"saddlp v2.4s, v2.8h",
"sadalp v2.4s, v3.8h",
"add v16.4s, v16.4s, v2.4s",
"stp xzr, xzr, [x28, #192]"
]
},
"vpdpbusd ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 26,
"Comment": [
"Map 2 0b01 0x50 256-bit",
"vpdpbusd ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x50, 0xc2"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #192]",
"ldr q3, [x28, #208]",
"ldr q4, [x28, #224]",
"uzp1 v5.8h, v17.8h, v18.8h",
"uxtl v6.8h, v5.8b",
"sxtl2 v5.8h, v5.16b",
"mul v5.8h, v6.8h, v5.8h",
"uzp2 v6.8h, v17.8h, v18.8h",
"uxtl v7.8h, v6.8b",
"sxtl2 v6.8h, v6.16b",
"mul v6.8h, v7.8h, v6.8h",
"saddlp v5.4s, v5.8h",
"sadalp v5.4s, v6.8h",
"add v16.4s, v16.4s, v5.4s",
"uzp1 v5.8h, v3.8h, v4.8h",
"uxtl v6.8h, v5.8b",
"sxtl2 v5.8h, v5.16b",
"mul v5.8h, v6.8h, v5.8h",
"uzp2 v3.8h, v3.8h, v4.8h",
"uxtl v4.8h, v3.8b",
"sxtl2 v3.8h, v3.16b",
"mul v3.8h, v4.8h, v3.8h",
"saddlp v4.4s, v5.8h",
"sadalp v4.4s, v3.8h",
"add v2.4s, v2.4s, v4.4s",
"str q2, [x28, #192]"
]
},
"vpdpbusds xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 12,
"Comment": [
"Map 2 0b01 0x51 128-bit",
"vpdpbusds xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x51, 0xc2"
],
"ExpectedArm64ASM": [
"uzp1 v2.8h, v17.8h, v18.8h",
"uxtl v3.8h, v2.8b",
"sxtl2 v2.8h, v2.16b",
"mul v2.8h, v3.8h, v2.8h",
"uzp2 v3.8h, v17.8h, v18.8h",
"uxtl v4.8h, v3.8b",
"sxtl2 v3.8h, v3.16b",
"mul v3.8h, v4.8h, v3.8h",
"saddlp v2.4s, v2.8h",
"sadalp v2.4s, v3.8h",
"sqadd v16.4s, v16.4s, v2.4s",
"stp xzr, xzr, [x28, #192]"
]
},
"vpdpbusds ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 26,
"Comment": [
"Map 2 0b01 0x51 256-bit",
"vpdpbusds ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x51, 0xc2"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #192]",
"ldr q3, [x28, #208]",
"ldr q4, [x28, #224]",
"uzp1 v5.8h, v17.8h, v18.8h",
"uxtl v6.8h, v5.8b",
"sxtl2 v5.8h, v5.16b",
"mul v5.8h, v6.8h, v5.8h",
"uzp2 v6.8h, v17.8h, v18.8h",
"uxtl v7.8h, v6.8b",
"sxtl2 v6.8h, v6.16b",
"mul v6.8h, v7.8h, v6.8h",
"saddlp v5.4s, v5.8h",
"sadalp v5.4s, v6.8h",
"sqadd v16.4s, v16.4s, v5.4s",
"uzp1 v5.8h, v3.8h, v4.8h",
"uxtl v6.8h, v5.8b",
"sxtl2 v5.8h, v5.16b",
"mul v5.8h, v6.8h, v5.8h",
"uzp2 v3.8h, v3.8h, v4.8h",
"uxtl v4.8h, v3.8b",
"sxtl2 v3.8h, v3.16b",
"mul v3.8h, v4.8h, v3.8h",
"saddlp v4.4s, v5.8h",
"sadalp v4.4s, v3.8h",
"sqadd v2.4s, v2.4s, v4.4s",
"str q2, [x28, #192]"
]
},
"vpdpwssd xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 5,
"Comment": [
"Map 2 0b01 0x52 128-bit",
"vpdpwssd xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x52, 0xc2"
],
"ExpectedArm64ASM": [
"smull v2.4s, v17.4h, v18.4h",
"smull2 v3.4s, v17.8h, v18.8h",
"addp v2.4s, v2.4s, v3.4s",
"add v16.4s, v16.4s, v2.4s",
"stp xzr, xzr, [x28, #192]"
]
},
"vpdpwssd ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 12,
"Comment": [
"Map 2 0b01 0x52 256-bit",
"vpdpwssd ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x52, 0xc2"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #192]",
"ldr q3, [x28, #208]",
"ldr q4, [x28, #224]",
"smull v5.4s, v17.4h, v18.4h",
"smull2 v6.4s, v17.8h, v18.8h",
"addp v5.4s, v5.4s, v6.4s",
"add v16.4s, v16.4s, v5.4s",
"smull v5.4s, v3.4h, v4.4h",
"smull2 v3.4s, v3.8h, v4.8h",
"addp v3.4s, v5.4s, v3.4s",
"add v2.4s, v2.4s, v3.4s",
"str q2, [x28, #192]"
]
},
"vpdpwssds xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 6,
"Comment": [
"Map 2 0b01 0x53 128-bit",
"vpdpwssds xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x53, 0xc2"
],
"ExpectedArm64ASM": [
"smull v2.4s, v17.4h, v18.4h",
"smull2 v3.4s, v17.8h, v18.8h",
"addp v2.4s, v2.4s, v3.4s",
"neg v2.4s, v2.4s",
"sqsub v16.4s, v16.4s, v2.4s",
"stp xzr, xzr, [x28, #192]"
]
},
"vpdpwssds ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 14,
"Comment": [
"Map 2 0b01 0x53 256-bit",
"vpdpwssds ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x53, 0xc2"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #192]",
"ldr q3, [x28, #208]",
"ldr q4, [x28, #224]",
"smull v5.4s, v17.4h, v18.4h",
"smull2 v6.4s, v17.8h, v18.8h",
"addp v5.4s, v5.4s, v6.4s",
"neg v5.4s, v5.4s",
"sqsub v16.4s, v16.4s, v5.4s",
"smull v5.4s, v3.4h, v4.4h",
"smull2 v3.4s, v3.8h, v4.8h",
"addp v3.4s, v5.4s, v3.4s",
"neg v3.4s, v3.4s",
"sqsub v2.4s, v2.4s, v3.4s",
"str q2, [x28, #192]"
]
},
"vpbroadcastd xmm0, xmm1": {
"ExpectedInstructionCount": 2,
"Comment": [
@@ -0,0 +1,131 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"DOTPROD"
],
"DisabledHostFeatures": [
"AFP",
"FLAGM",
"FLAGM2",
"SVE128",
"SVE256",
"I8MM"
],
"BinaryCacheVersion": 20
},
"Instructions": {
"vpdpbusd xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 9,
"Comment": [
"Map 2 0b01 0x50 128-bit",
"vpdpbusd xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x50, 0xc2"
],
"ExpectedArm64ASM": [
"movi v2.16b, #0x80",
"eor v2.16b, v17.16b, v2.16b",
"movi v3.16b, #0x40",
"mov v4.16b, v16.16b",
"sdot v4.4s, v18.16b, v3.16b",
"sdot v4.4s, v18.16b, v3.16b",
"mov v16.16b, v4.16b",
"sdot v16.4s, v2.16b, v18.16b",
"stp xzr, xzr, [x28, #192]"
]
},
"vpdpbusd ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 18,
"Comment": [
"Map 2 0b01 0x50 256-bit",
"vpdpbusd ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x50, 0xc2"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #192]",
"ldr q3, [x28, #208]",
"ldr q4, [x28, #224]",
"movi v5.16b, #0x80",
"eor v5.16b, v17.16b, v5.16b",
"movi v6.16b, #0x40",
"mov v7.16b, v16.16b",
"sdot v7.4s, v18.16b, v6.16b",
"sdot v7.4s, v18.16b, v6.16b",
"mov v16.16b, v7.16b",
"sdot v16.4s, v5.16b, v18.16b",
"movi v5.16b, #0x80",
"eor v3.16b, v3.16b, v5.16b",
"movi v5.16b, #0x40",
"sdot v2.4s, v4.16b, v5.16b",
"sdot v2.4s, v4.16b, v5.16b",
"sdot v2.4s, v3.16b, v4.16b",
"str q2, [x28, #192]"
]
},
"vpdpbusds xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 9,
"Comment": [
"Map 2 0b01 0x51 128-bit",
"vpdpbusds xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x51, 0xc2"
],
"ExpectedArm64ASM": [
"movi v2.16b, #0x80",
"eor v2.16b, v17.16b, v2.16b",
"movi v3.16b, #0x40",
"movi v4.2d, #0x0",
"sdot v4.4s, v18.16b, v3.16b",
"sdot v4.4s, v18.16b, v3.16b",
"sdot v4.4s, v2.16b, v18.16b",
"sqadd v16.4s, v16.4s, v4.4s",
"stp xzr, xzr, [x28, #192]"
]
},
"vpdpbusds ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 20,
"Comment": [
"Map 2 0b01 0x51 256-bit",
"vpdpbusds ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x51, 0xc2"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #192]",
"ldr q3, [x28, #208]",
"ldr q4, [x28, #224]",
"movi v5.16b, #0x80",
"eor v5.16b, v17.16b, v5.16b",
"movi v6.16b, #0x40",
"movi v7.2d, #0x0",
"mov v8.16b, v7.16b",
"sdot v8.4s, v18.16b, v6.16b",
"sdot v8.4s, v18.16b, v6.16b",
"sdot v8.4s, v5.16b, v18.16b",
"sqadd v16.4s, v16.4s, v8.4s",
"movi v5.16b, #0x80",
"eor v3.16b, v3.16b, v5.16b",
"movi v5.16b, #0x40",
"sdot v7.4s, v4.16b, v5.16b",
"sdot v7.4s, v4.16b, v5.16b",
"sdot v7.4s, v3.16b, v4.16b",
"sqadd v2.4s, v2.4s, v7.4s",
"str q2, [x28, #192]"
]
}
}
}
@@ -0,0 +1,189 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"I8MM"
],
"DisabledHostFeatures": [
"AFP",
"FLAGM",
"FLAGM2",
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 20
},
"Instructions": {
"vpdpbusd xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 2,
"Comment": [
"Map 2 0b01 0x50 128-bit",
"vpdpbusd xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x50, 0xc2"
],
"ExpectedArm64ASM": [
"usdot v16.4s, v17.16b, v18.16b",
"stp xzr, xzr, [x28, #192]"
]
},
"vpdpbusd ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 6,
"Comment": [
"Map 2 0b01 0x50 256-bit",
"vpdpbusd ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x50, 0xc2"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #192]",
"ldr q3, [x28, #208]",
"ldr q4, [x28, #224]",
"usdot v16.4s, v17.16b, v18.16b",
"usdot v2.4s, v3.16b, v4.16b",
"str q2, [x28, #192]"
]
},
"vpdpbusds xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 4,
"Comment": [
"Map 2 0b01 0x51 128-bit",
"vpdpbusds xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x51, 0xc2"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"usdot v2.4s, v17.16b, v18.16b",
"sqadd v16.4s, v16.4s, v2.4s",
"stp xzr, xzr, [x28, #192]"
]
},
"vpdpbusds ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 10,
"Comment": [
"Map 2 0b01 0x51 256-bit",
"vpdpbusds ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x51, 0xc2"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #192]",
"ldr q3, [x28, #208]",
"ldr q4, [x28, #224]",
"movi v5.2d, #0x0",
"mov v6.16b, v5.16b",
"usdot v6.4s, v17.16b, v18.16b",
"sqadd v16.4s, v16.4s, v6.4s",
"usdot v5.4s, v3.16b, v4.16b",
"sqadd v2.4s, v2.4s, v5.4s",
"str q2, [x28, #192]"
]
},
"vpdpwssd xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 5,
"Comment": [
"Map 2 0b01 0x52 128-bit",
"vpdpwssd xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x52, 0xc2"
],
"ExpectedArm64ASM": [
"smull v2.4s, v17.4h, v18.4h",
"smull2 v3.4s, v17.8h, v18.8h",
"addp v2.4s, v2.4s, v3.4s",
"add v16.4s, v16.4s, v2.4s",
"stp xzr, xzr, [x28, #192]"
]
},
"vpdpwssd ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 12,
"Comment": [
"Map 2 0b01 0x52 256-bit",
"vpdpwssd ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x52, 0xc2"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #192]",
"ldr q3, [x28, #208]",
"ldr q4, [x28, #224]",
"smull v5.4s, v17.4h, v18.4h",
"smull2 v6.4s, v17.8h, v18.8h",
"addp v5.4s, v5.4s, v6.4s",
"add v16.4s, v16.4s, v5.4s",
"smull v5.4s, v3.4h, v4.4h",
"smull2 v3.4s, v3.8h, v4.8h",
"addp v3.4s, v5.4s, v3.4s",
"add v2.4s, v2.4s, v3.4s",
"str q2, [x28, #192]"
]
},
"vpdpwssds xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 6,
"Comment": [
"Map 2 0b01 0x53 128-bit",
"vpdpwssds xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x53, 0xc2"
],
"ExpectedArm64ASM": [
"smull v2.4s, v17.4h, v18.4h",
"smull2 v3.4s, v17.8h, v18.8h",
"addp v2.4s, v2.4s, v3.4s",
"neg v2.4s, v2.4s",
"sqsub v16.4s, v16.4s, v2.4s",
"stp xzr, xzr, [x28, #192]"
]
},
"vpdpwssds ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 14,
"Comment": [
"Map 2 0b01 0x53 256-bit",
"vpdpwssds ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x53, 0xc2"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #192]",
"ldr q3, [x28, #208]",
"ldr q4, [x28, #224]",
"smull v5.4s, v17.4h, v18.4h",
"smull2 v6.4s, v17.8h, v18.8h",
"addp v5.4s, v5.4s, v6.4s",
"neg v5.4s, v5.4s",
"sqsub v16.4s, v16.4s, v5.4s",
"smull v5.4s, v3.4h, v4.4h",
"smull2 v3.4s, v3.8h, v4.8h",
"addp v3.4s, v5.4s, v3.4s",
"neg v3.4s, v3.4s",
"sqsub v2.4s, v2.4s, v3.4s",
"str q2, [x28, #192]"
]
}
}
}
@@ -95,8 +95,8 @@
"msr nzcv, x0",
"and z2.d, z3.d, z2.d",
"addp z2.b, p7/m, z2.b, z2.b",
"uzp1 z2.b, z2.b, z2.b",
"uzp2 z1.b, z2.b, z2.b",
"uzp1 z2.b, z2.b, z2.b",
"splice z2.d, p6, z2.d, z1.d",
"addp v2.16b, v2.16b, v2.16b",
"addp v2.8b, v2.8b, v2.8b",
+2 -2
View File
@@ -4540,8 +4540,8 @@
"msr nzcv, x0",
"and z2.d, z3.d, z2.d",
"addp z2.b, p7/m, z2.b, z2.b",
"uzp1 z2.b, z2.b, z2.b",
"uzp2 z1.b, z2.b, z2.b",
"uzp1 z2.b, z2.b, z2.b",
"splice z2.d, p6, z2.d, z1.d",
"addp v2.16b, v2.16b, v2.16b",
"addp v2.8b, v2.8b, v2.8b",
@@ -5357,8 +5357,8 @@
"zip2 z3.s, z0.s, z1.s",
"movprfx z0, z2",
"addp z0.s, p7/m, z0.s, z3.s",
"uzp1 z16.s, z0.s, z0.s",
"uzp2 z1.s, z0.s, z0.s",
"uzp1 z16.s, z0.s, z0.s",
"splice z16.d, p6, z16.d, z1.d"
]
},
+197 -3
View File
@@ -9,7 +9,9 @@
"AFP",
"FLAGM",
"FLAGM2",
"SVEBITPERM"
"SVEBITPERM",
"I8MM",
"DOTPROD"
],
"BinaryCacheVersion": 20
},
@@ -59,8 +61,8 @@
"ExpectedArm64ASM": [
"movprfx z0, z17",
"addp z0.h, p7/m, z0.h, z18.h",
"uzp1 z2.h, z0.h, z0.h",
"uzp2 z1.h, z0.h, z0.h",
"uzp1 z2.h, z0.h, z0.h",
"splice z2.d, p6, z2.d, z1.d",
"ldr x0, [x28, #2552]",
"ld1b {z3.b}, p7/z, [x0]",
@@ -84,8 +86,8 @@
"ExpectedArm64ASM": [
"movprfx z0, z17",
"addp z0.s, p7/m, z0.s, z18.s",
"uzp1 z2.s, z0.s, z0.s",
"uzp2 z1.s, z0.s, z0.s",
"uzp1 z2.s, z0.s, z0.s",
"splice z2.d, p6, z2.d, z1.d",
"ldr x0, [x28, #2552]",
"ld1b {z3.b}, p7/z, [x0]",
@@ -1639,6 +1641,198 @@
"lsl z16.d, p7/m, z16.d, z1.d"
]
},
"vpdpbusd xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 11,
"Comment": [
"Map 2 0b01 0x50 128-bit",
"vpdpbusd xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x50, 0xc2"
],
"ExpectedArm64ASM": [
"uzp1 v2.8h, v17.8h, v18.8h",
"uxtl v3.8h, v2.8b",
"sxtl2 v2.8h, v2.16b",
"mul v2.8h, v3.8h, v2.8h",
"uzp2 v3.8h, v17.8h, v18.8h",
"uxtl v4.8h, v3.8b",
"sxtl2 v3.8h, v3.16b",
"mul v3.8h, v4.8h, v3.8h",
"saddlp v2.4s, v2.8h",
"sadalp v2.4s, v3.8h",
"add v16.4s, v16.4s, v2.4s"
]
},
"vpdpbusd ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 13,
"Comment": [
"Map 2 0b01 0x50 256-bit",
"vpdpbusd ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x50, 0xc2"
],
"ExpectedArm64ASM": [
"uzp1 z2.h, z17.h, z18.h",
"uunpklo z3.h, z2.b",
"sunpkhi z2.h, z2.b",
"mul z2.h, z3.h, z2.h",
"uzp2 z3.h, z17.h, z18.h",
"uunpklo z4.h, z3.b",
"sunpkhi z3.h, z3.b",
"mul z3.h, z4.h, z3.h",
"mov z0.s, #0",
"sadalp z0.s, p7/m, z2.h",
"mov z2.d, z0.d",
"sadalp z2.s, p7/m, z3.h",
"add z16.s, z16.s, z2.s"
]
},
"vpdpbusds xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 11,
"Comment": [
"Map 2 0b01 0x51 128-bit",
"vpdpbusds xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x51, 0xc2"
],
"ExpectedArm64ASM": [
"uzp1 v2.8h, v17.8h, v18.8h",
"uxtl v3.8h, v2.8b",
"sxtl2 v2.8h, v2.16b",
"mul v2.8h, v3.8h, v2.8h",
"uzp2 v3.8h, v17.8h, v18.8h",
"uxtl v4.8h, v3.8b",
"sxtl2 v3.8h, v3.16b",
"mul v3.8h, v4.8h, v3.8h",
"saddlp v2.4s, v2.8h",
"sadalp v2.4s, v3.8h",
"sqadd v16.4s, v16.4s, v2.4s"
]
},
"vpdpbusds ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 13,
"Comment": [
"Map 2 0b01 0x51 256-bit",
"vpdpbusds ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x51, 0xc2"
],
"ExpectedArm64ASM": [
"uzp1 z2.h, z17.h, z18.h",
"uunpklo z3.h, z2.b",
"sunpkhi z2.h, z2.b",
"mul z2.h, z3.h, z2.h",
"uzp2 z3.h, z17.h, z18.h",
"uunpklo z4.h, z3.b",
"sunpkhi z3.h, z3.b",
"mul z3.h, z4.h, z3.h",
"mov z0.s, #0",
"sadalp z0.s, p7/m, z2.h",
"mov z2.d, z0.d",
"sadalp z2.s, p7/m, z3.h",
"sqadd z16.s, z16.s, z2.s"
]
},
"vpdpwssd xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 4,
"Comment": [
"Map 2 0b01 0x52 128-bit",
"vpdpwssd xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x52, 0xc2"
],
"ExpectedArm64ASM": [
"smull v2.4s, v17.4h, v18.4h",
"smull2 v3.4s, v17.8h, v18.8h",
"addp v2.4s, v2.4s, v3.4s",
"add v16.4s, v16.4s, v2.4s"
]
},
"vpdpwssd ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 11,
"Comment": [
"Map 2 0b01 0x52 256-bit",
"vpdpwssd ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x52, 0xc2"
],
"ExpectedArm64ASM": [
"smullb z0.s, z17.h, z18.h",
"smullt z1.s, z17.h, z18.h",
"zip1 z2.s, z0.s, z1.s",
"smullb z0.s, z17.h, z18.h",
"smullt z1.s, z17.h, z18.h",
"zip2 z3.s, z0.s, z1.s",
"addp z2.s, p7/m, z2.s, z3.s",
"uzp2 z1.s, z2.s, z2.s",
"uzp1 z2.s, z2.s, z2.s",
"splice z2.d, p6, z2.d, z1.d",
"add z16.s, z16.s, z2.s"
]
},
"vpdpwssds xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 5,
"Comment": [
"Map 2 0b01 0x53 128-bit",
"vpdpwssds xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x53, 0xc2"
],
"ExpectedArm64ASM": [
"smull v2.4s, v17.4h, v18.4h",
"smull2 v3.4s, v17.8h, v18.8h",
"addp v2.4s, v2.4s, v3.4s",
"neg v2.4s, v2.4s",
"sqsub v16.4s, v16.4s, v2.4s"
]
},
"vpdpwssds ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 12,
"Comment": [
"Map 2 0b01 0x53 256-bit",
"vpdpwssds ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x53, 0xc2"
],
"ExpectedArm64ASM": [
"smullb z0.s, z17.h, z18.h",
"smullt z1.s, z17.h, z18.h",
"zip1 z2.s, z0.s, z1.s",
"smullb z0.s, z17.h, z18.h",
"smullt z1.s, z17.h, z18.h",
"zip2 z3.s, z0.s, z1.s",
"addp z2.s, p7/m, z2.s, z3.s",
"uzp2 z1.s, z2.s, z2.s",
"uzp1 z2.s, z2.s, z2.s",
"splice z2.d, p6, z2.d, z1.d",
"neg z2.s, p7/m, z2.s",
"sqsub z16.s, z16.s, z2.s"
]
},
"vpbroadcastd xmm0, xmm1": {
"ExpectedInstructionCount": 1,
"Comment": [
@@ -0,0 +1,108 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"SVE256",
"SVE128",
"DOTPROD"
],
"DisabledHostFeatures": [
"AFP",
"FLAGM",
"FLAGM2",
"SVEBITPERM",
"I8MM"
],
"BinaryCacheVersion": 20
},
"Instructions": {
"vpdpbusd xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 8,
"Comment": [
"Map 2 0b01 0x50 128-bit",
"vpdpbusd xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x50, 0xc2"
],
"ExpectedArm64ASM": [
"movi v2.16b, #0x80",
"eor v2.16b, v17.16b, v2.16b",
"movi v3.16b, #0x40",
"mov v4.16b, v16.16b",
"sdot v4.4s, v18.16b, v3.16b",
"sdot v4.4s, v18.16b, v3.16b",
"mov v16.16b, v4.16b",
"sdot v16.4s, v2.16b, v18.16b"
]
},
"vpdpbusd ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 8,
"Comment": [
"Map 2 0b01 0x50 256-bit",
"vpdpbusd ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x50, 0xc2"
],
"ExpectedArm64ASM": [
"mov z2.b, #-128",
"eor z2.d, z17.d, z2.d",
"mov z3.b, #64",
"mov z4.d, z16.d",
"sdot z4.s, z18.b, z3.b",
"sdot z4.s, z18.b, z3.b",
"mov z16.d, z4.d",
"sdot z16.s, z2.b, z18.b"
]
},
"vpdpbusds xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 8,
"Comment": [
"Map 2 0b01 0x51 128-bit",
"vpdpbusds xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x51, 0xc2"
],
"ExpectedArm64ASM": [
"movi v2.16b, #0x80",
"eor v2.16b, v17.16b, v2.16b",
"movi v3.16b, #0x40",
"movi v4.2d, #0x0",
"sdot v4.4s, v18.16b, v3.16b",
"sdot v4.4s, v18.16b, v3.16b",
"sdot v4.4s, v2.16b, v18.16b",
"sqadd v16.4s, v16.4s, v4.4s"
]
},
"vpdpbusds ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 8,
"Comment": [
"Map 2 0b01 0x51 256-bit",
"vpdpbusds ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x51, 0xc2"
],
"ExpectedArm64ASM": [
"mov z2.b, #-128",
"eor z2.d, z17.d, z2.d",
"mov z3.b, #64",
"movi v4.2d, #0x0",
"sdot z4.s, z18.b, z3.b",
"sdot z4.s, z18.b, z3.b",
"sdot z4.s, z2.b, z18.b",
"sqadd z16.s, z16.s, z4.s"
]
}
}
}
@@ -0,0 +1,171 @@
{
"Features": {
"Bitness": 64,
"EnabledHostFeatures": [
"SVE256",
"SVE128",
"I8MM"
],
"DisabledHostFeatures": [
"AFP",
"FLAGM",
"FLAGM2",
"SVEBITPERM"
],
"BinaryCacheVersion": 20
},
"Instructions": {
"vpdpbusd xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 1,
"Comment": [
"Map 2 0b01 0x50 128-bit",
"vpdpbusd xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x50, 0xc2"
],
"ExpectedArm64ASM": [
"usdot v16.4s, v17.16b, v18.16b"
]
},
"vpdpbusd ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 1,
"Comment": [
"Map 2 0b01 0x50 256-bit",
"vpdpbusd ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x50, 0xc2"
],
"ExpectedArm64ASM": [
"usdot z16.s, z17.b, z18.b"
]
},
"vpdpbusds xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 3,
"Comment": [
"Map 2 0b01 0x51 128-bit",
"vpdpbusds xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x51, 0xc2"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"usdot v2.4s, v17.16b, v18.16b",
"sqadd v16.4s, v16.4s, v2.4s"
]
},
"vpdpbusds ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 3,
"Comment": [
"Map 2 0b01 0x51 256-bit",
"vpdpbusds ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x51, 0xc2"
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"usdot z2.s, z17.b, z18.b",
"sqadd z16.s, z16.s, z2.s"
]
},
"vpdpwssd xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 4,
"Comment": [
"Map 2 0b01 0x52 128-bit",
"vpdpwssd xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x52, 0xc2"
],
"ExpectedArm64ASM": [
"smull v2.4s, v17.4h, v18.4h",
"smull2 v3.4s, v17.8h, v18.8h",
"addp v2.4s, v2.4s, v3.4s",
"add v16.4s, v16.4s, v2.4s"
]
},
"vpdpwssd ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 11,
"Comment": [
"Map 2 0b01 0x52 256-bit",
"vpdpwssd ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x52, 0xc2"
],
"ExpectedArm64ASM": [
"smullb z0.s, z17.h, z18.h",
"smullt z1.s, z17.h, z18.h",
"zip1 z2.s, z0.s, z1.s",
"smullb z0.s, z17.h, z18.h",
"smullt z1.s, z17.h, z18.h",
"zip2 z3.s, z0.s, z1.s",
"addp z2.s, p7/m, z2.s, z3.s",
"uzp2 z1.s, z2.s, z2.s",
"uzp1 z2.s, z2.s, z2.s",
"splice z2.d, p6, z2.d, z1.d",
"add z16.s, z16.s, z2.s"
]
},
"vpdpwssds xmm0, xmm1, xmm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 5,
"Comment": [
"Map 2 0b01 0x53 128-bit",
"vpdpwssds xmm0, xmm1, xmm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x71, 0x53, 0xc2"
],
"ExpectedArm64ASM": [
"smull v2.4s, v17.4h, v18.4h",
"smull2 v3.4s, v17.8h, v18.8h",
"addp v2.4s, v2.4s, v3.4s",
"neg v2.4s, v2.4s",
"sqsub v16.4s, v16.4s, v2.4s"
]
},
"vpdpwssds ymm0, ymm1, ymm2": {
"x86InstructionCount": 1,
"ExpectedInstructionCount": 12,
"Comment": [
"Map 2 0b01 0x53 256-bit",
"vpdpwssds ymm0, ymm1, ymm2",
"NASM only emits the EVEX encoding, hand encode the VEX form"
],
"x86Insts": [
"db 0xc4, 0xe2, 0x75, 0x53, 0xc2"
],
"ExpectedArm64ASM": [
"smullb z0.s, z17.h, z18.h",
"smullt z1.s, z17.h, z18.h",
"zip1 z2.s, z0.s, z1.s",
"smullb z0.s, z17.h, z18.h",
"smullt z1.s, z17.h, z18.h",
"zip2 z3.s, z0.s, z1.s",
"addp z2.s, p7/m, z2.s, z3.s",
"uzp2 z1.s, z2.s, z2.s",
"uzp1 z2.s, z2.s, z2.s",
"splice z2.d, p6, z2.d, z1.d",
"neg z2.s, p7/m, z2.s",
"sqsub z16.s, z16.s, z2.s"
]
}
}
}