diff --git a/FEXCore/Source/Interface/Config/Config.json.in b/FEXCore/Source/Interface/Config/Config.json.in index 136aa1cf6..030c1e6d7 100644 --- a/FEXCore/Source/Interface/Config/Config.json.in +++ b/FEXCore/Source/Interface/Config/Config.json.in @@ -91,7 +91,11 @@ "ENABLESSE4A": "enablesse4a", "DISABLESSE4A": "disablesse4a", "ENABLEMOPS": "enablemops", - "DISABLEMOPS": "disablemops" + "DISABLEMOPS": "disablemops", + "ENABLEI8MM": "enablei8mm", + "DISABLEI8MM": "disablei8mm", + "ENABLEDOTPROD": "enabledotprod", + "DISABLEDOTPROD": "disabledotprod" }, "Desc": [ "Allows controlling of the CPU features in the JIT.", @@ -116,7 +120,9 @@ "\t{enable,disable}wfxt: Will force enable or disable wfxt even if the host doesn't support it", "\t{enable,disable}3dnow: Will force enable or disable 3DNow! even if the host doesn't support it", "\t{enable,disable}sse4a: Will force enable or disable SSE4a even if the host doesn't support it", - "\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it" + "\t{enable,disable}mops: Will force enable or disable FEAT_MOPS even if the host doesn't support it", + "\t{enable,disable}i8mm: Will force enable or disable i8mm even if the host doesn't support it", + "\t{enable,disable}dotprod: Will force enable or disable dotprod even if the host doesn't support it" ] }, "SmallTSCScale": { diff --git a/FEXCore/Source/Interface/Core/CPUID.cpp b/FEXCore/Source/Interface/Core/CPUID.cpp index 3c42d3043..fa8cd8ed1 100644 --- a/FEXCore/Source/Interface/Core/CPUID.cpp +++ b/FEXCore/Source/Interface/Core/CPUID.cpp @@ -648,6 +648,12 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_06h(uint32_t Leaf) const { FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const { FEXCore::CPUID::FunctionResults Res {}; + + // AVX-VNNI is only advertised when the CPU supports I8MM or Dot Product. + // Without these features the implementation is so slow that it is likely + // to harm performance. + const uint32_t SupportsAVXVNNI = SupportsAVX() && (CTX->HostFeatures.SupportsI8MM || CTX->HostFeatures.SupportsDotProd); + if (Leaf == 0) { #ifndef _WIN32 constexpr uint32_t SUPPORTS_RDPID = 1; @@ -665,7 +671,9 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const { const uint32_t SupportsWFXT = CTX->HostFeatures.SupportsWFXT; // Number of subfunctions - Res.eax = 0x0; + // TODO: For now, subfunction 1 only exposes AVX-VNNI so we make it conditional + // on AVX-VNNI support. We should revisit this if/when we add more to this leaf. + Res.eax = SupportsAVXVNNI; Res.ebx = (1 << 0) | // FS/GS support (0 << 1) | // TSC adjust MSR (0 << 2) | // SGX @@ -765,38 +773,38 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const { (0 << 30) | // Arch capabilities - MSR module specific (0 << 31); // SSBD - Speculative Store Bypass Disable } else if (Leaf == 1) { - Res.eax = (0U << 0) | // SHA512 - (0U << 1) | // SM3 - (0U << 2) | // SM4 - (0U << 3) | // RAO_INT - (0U << 4) | // AVX_VNNI - (0U << 5) | // AVX512_BF16 - (0U << 6) | // LASS (Linear Address Space Separation) - (0U << 7) | // CMPCCXADD - (0U << 8) | // ARCH_PERFMON_EXT - (0U << 9) | // Reserved - (0U << 10) | // FAST_REP_MOVSB - (0U << 11) | // FAST_REP_STOSB - (0U << 12) | // FAST_REP_CMPSB_SCASB - (0U << 13) | // Reserved - (0U << 14) | // Reserved - (0U << 15) | // Reserved - (0U << 16) | // Reserved - (0U << 17) | // FRED (Flexible Return and Event Delivery) - (0U << 18) | // LKGS (Load into Kernel GS Base) - (0U << 19) | // WRMSRNS - (0U << 20) | // NMI_SRC - (0U << 21) | // AMX_FP16 - (0U << 22) | // HRESET - (0U << 23) | // AVX_IFMA - (0U << 24) | // Reserved - (0U << 25) | // Reserved - (0U << 26) | // LAM (Linear Address Masking) - (0U << 27) | // MSRLIST - (0U << 28) | // Reserved - (0U << 29) | // Reserved - (0U << 30) | // INVD_DISABLE_POST_BIOS_DONE - (0U << 31); // MOVRS + Res.eax = (0U << 0) | // SHA512 + (0U << 1) | // SM3 + (0U << 2) | // SM4 + (0U << 3) | // RAO_INT + (SupportsAVXVNNI << 4) | // AVX_VNNI + (0U << 5) | // AVX512_BF16 + (0U << 6) | // LASS (Linear Address Space Separation) + (0U << 7) | // CMPCCXADD + (0U << 8) | // ARCH_PERFMON_EXT + (0U << 9) | // Reserved + (0U << 10) | // FAST_REP_MOVSB + (0U << 11) | // FAST_REP_STOSB + (0U << 12) | // FAST_REP_CMPSB_SCASB + (0U << 13) | // Reserved + (0U << 14) | // Reserved + (0U << 15) | // Reserved + (0U << 16) | // Reserved + (0U << 17) | // FRED (Flexible Return and Event Delivery) + (0U << 18) | // LKGS (Load into Kernel GS Base) + (0U << 19) | // WRMSRNS + (0U << 20) | // NMI_SRC + (0U << 21) | // AMX_FP16 + (0U << 22) | // HRESET + (0U << 23) | // AVX_IFMA + (0U << 24) | // Reserved + (0U << 25) | // Reserved + (0U << 26) | // LAM (Linear Address Masking) + (0U << 27) | // MSRLIST + (0U << 28) | // Reserved + (0U << 29) | // Reserved + (0U << 30) | // INVD_DISABLE_POST_BIOS_DONE + (0U << 31); // MOVRS // Bits 4-31 currently reserved. Res.ebx = (0U << 0) | // PPIN diff --git a/FEXCore/Source/Interface/Core/JIT/VectorOps.cpp b/FEXCore/Source/Interface/Core/JIT/VectorOps.cpp index 58160e952..b07de9ac5 100644 --- a/FEXCore/Source/Interface/Core/JIT/VectorOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/VectorOps.cpp @@ -1112,8 +1112,10 @@ DEF_OP(VAddP) { // pairwise addition, the SVE version actually interleaves the // results of the pairwise addition (gross!), so we need to undo that. addp(SubRegSize, LHS.Z(), Pred, LHS.Z(), VectorUpper.Z()); - uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z()); + + // Extract the upper half first, since Dst may alias the LHS. uzp2(SubRegSize, VTMP2.Z(), LHS.Z(), LHS.Z()); + uzp1(SubRegSize, Dst.Z(), LHS.Z(), LHS.Z()); // Merge upper half with lower half. splice(ARMEmitter::SubRegSize::i64Bit, Dst.Z(), PRED_TMP_16B, Dst.Z(), VTMP2.Z()); @@ -3833,6 +3835,171 @@ DEF_OP(VMul) { } } +DEF_OP(VUSDot) { + ///< Dest = Acc + dot(Vector1 (unsigned 8-bit), Vector2 (signed 8-bit)) + // Matches: + // - SVE - USDOT + // - ASIMD - USDOT + const auto Op = IROp->C(); + const auto OpSize = IROp->Size; + + const auto Is256Bit = OpSize == IR::OpSize::i256Bit; + LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); + + const auto Dst = GetVReg(Node); + const auto Acc = GetVReg(Op->Acc); + const auto Vector1 = GetVReg(Op->Vector1); + const auto Vector2 = GetVReg(Op->Vector2); + + // USDOT accumulates in to its destination register, + // so we need to emit a move if Acc != Dst + ARMEmitter::VRegister DestTmp = Dst; + if (Dst != Acc) { + if (Dst != Vector1 && Dst != Vector2) { + DestTmp = Dst; + } else { + DestTmp = VTMP1; + } + } + + if (HostSupportsSVE256 && Is256Bit) { + if (Dst != Acc) { + mov(DestTmp.Z(), Acc.Z()); + } + + usdot(DestTmp.Z(), Vector1.Z(), Vector2.Z()); + if (Dst != DestTmp) { + mov(Dst.Z(), DestTmp.Z()); + } + } else { + if (Dst != Acc) { + mov(DestTmp.Q(), Acc.Q()); + } + + usdot(DestTmp.Q(), Vector1.Q(), Vector2.Q()); + if (Dst != DestTmp) { + mov(Dst.Q(), DestTmp.Q()); + } + } +} + +DEF_OP(VSDot) { + ///< Dest = Acc + dot(Vector1 (signed 8-bit), Vector2 (signed 8-bit)) + // Matches: + // - SVE - SDOT + // - ASIMD - SDOT + const auto Op = IROp->C(); + const auto OpSize = IROp->Size; + + const auto Is256Bit = OpSize == IR::OpSize::i256Bit; + LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); + + const auto Dst = GetVReg(Node); + const auto Acc = GetVReg(Op->Acc); + const auto Vector1 = GetVReg(Op->Vector1); + const auto Vector2 = GetVReg(Op->Vector2); + + // SDOT accumulates in to its destination register, + // so we need to emit a move if Acc != Dst + ARMEmitter::VRegister DestTmp = Dst; + if (Dst != Acc) { + if (Dst != Vector1 && Dst != Vector2) { + DestTmp = Dst; + } else { + DestTmp = VTMP1; + } + } + + if (HostSupportsSVE256 && Is256Bit) { + if (Dst != Acc) { + mov(DestTmp.Z(), Acc.Z()); + } + + sdot(ARMEmitter::SubRegSize::i32Bit, DestTmp.Z(), Vector1.Z(), Vector2.Z()); + if (Dst != DestTmp) { + mov(Dst.Z(), DestTmp.Z()); + } + } else { + if (Dst != Acc) { + mov(DestTmp.Q(), Acc.Q()); + } + + sdot(ARMEmitter::SubRegSize::i32Bit, DestTmp.Q(), Vector1.Q(), Vector2.Q()); + if (Dst != DestTmp) { + mov(Dst.Q(), DestTmp.Q()); + } + } +} + +DEF_OP(VSAddLP) { + const auto Op = IROp->C(); + const auto OpSize = IROp->Size; + + const auto SubRegSize = ConvertSubRegSize248(IROp); + const auto Is256Bit = OpSize == IR::OpSize::i256Bit; + LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); + + const auto Dst = GetVReg(Node); + const auto Vector = GetVReg(Op->Vector); + + if (HostSupportsSVE256 && Is256Bit) { + // SVE only has the accumulating form, so accumulate in to a zeroed register. + // Zero a temporary instead if Dst aliases the source. + const auto DestTmp = Dst == Vector ? VTMP1 : Dst; + dup_imm(SubRegSize, DestTmp.Z(), 0); + sadalp(SubRegSize, DestTmp.Z(), PRED_TMP_32B.Merging(), Vector.Z()); + if (Dst != DestTmp) { + mov(Dst.Z(), DestTmp.Z()); + } + } else { + saddlp(SubRegSize, Dst.Q(), Vector.Q()); + } +} + +DEF_OP(VSAdALP) { + const auto Op = IROp->C(); + const auto OpSize = IROp->Size; + + const auto SubRegSize = ConvertSubRegSize248(IROp); + const auto Is256Bit = OpSize == IR::OpSize::i256Bit; + LOGMAN_THROW_A_FMT(!Is256Bit || HostSupportsSVE256, "Need SVE256 support in order to use {} with 256-bit operation", __func__); + + const auto Dst = GetVReg(Node); + const auto Acc = GetVReg(Op->Acc); + const auto Vector = GetVReg(Op->Vector); + + // SADALP accumulates in to its destination register, + // so we need to emit a move if Acc != Dst + ARMEmitter::VRegister DestTmp = Dst; + if (Dst != Acc) { + DestTmp = Dst != Vector ? Dst : VTMP1; + } + + if (HostSupportsSVE256 && Is256Bit) { + if (Dst != Acc) { + mov(DestTmp.Z(), Acc.Z()); + } + + sadalp(SubRegSize, DestTmp.Z(), PRED_TMP_32B.Merging(), Vector.Z()); + if (Dst != DestTmp) { + mov(Dst.Z(), DestTmp.Z()); + } + } else { + if (Dst == Vector) { + // ASIMD has the non-accumulating form, which is cheaper than shuffling through a temporary. + saddlp(SubRegSize, VTMP1.Q(), Vector.Q()); + add(SubRegSize, Dst.Q(), Acc.Q(), VTMP1.Q()); + return; + } + + if (Dst != Acc) { + mov(Dst.Q(), Acc.Q()); + } + + sadalp(SubRegSize, Dst.Q(), Vector.Q()); + } +} + DEF_OP(VUMull) { const auto Op = IROp->C(); const auto OpSize = IROp->Size; diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h index ff5c38951..6d4ef0262 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h @@ -665,6 +665,8 @@ public: void VPMADDUBSWOp(OpcodeArgs); void VPMADDWDOp(OpcodeArgs); + void VPDPBUSDOp(OpcodeArgs, bool Saturating); + void VPDPWSSDOp(OpcodeArgs, bool Saturating); void VPMASKMOVOp(OpcodeArgs, bool IsStore); @@ -1036,6 +1038,9 @@ public: void AVX128_VPMADDUBSW(OpcodeArgs); void AVX128_VPMADDWD(OpcodeArgs); + void AVX128_VPDPImpl(OpcodeArgs, std::function Helper); + void AVX128_VPDPBUSD(OpcodeArgs, bool Saturating); + void AVX128_VPDPWSSD(OpcodeArgs, bool Saturating); void AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize); @@ -1400,6 +1405,10 @@ private: Ref PMADDUBSWOpImpl(IR::OpSize Size, Ref Src1, Ref Src2); + Ref VPDPBUSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating); + + Ref VPDPWSSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating); + Ref PMULHRSWOpImpl(OpSize Size, Ref Src1, Ref Src2); Ref PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp index cfce2569d..59466b976 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp @@ -1470,6 +1470,35 @@ void OpDispatchBuilder::AVX128_VPMADDWD(OpcodeArgs) { [this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return PMADDWDOpImpl(OpSize::i128Bit, Src1, Src2); }); } +void OpDispatchBuilder::AVX128_VPDPImpl(OpcodeArgs, std::function Helper) { + const auto Size = OpSizeFromDst(Op); + const auto Is128Bit = Size == OpSize::i128Bit; + + auto Acc = AVX128_LoadSource_WithOpSize(Op, Op->Dest, Op->Flags, !Is128Bit); + auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit); + auto Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, !Is128Bit); + + RefPair Result {}; + Result.Low = Helper(Acc.Low, Src1.Low, Src2.Low); + if (Is128Bit) { + Result.High = LoadZeroVector(OpSize::i128Bit); + } else { + Result.High = Helper(Acc.High, Src1.High, Src2.High); + } + + AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result); +} + +void OpDispatchBuilder::AVX128_VPDPBUSD(OpcodeArgs, bool Saturating) { + AVX128_VPDPImpl(Op, + [this, Saturating](Ref Acc, Ref Src1, Ref Src2) { return VPDPBUSDOpImpl(OpSize::i128Bit, Acc, Src1, Src2, Saturating); }); +} + +void OpDispatchBuilder::AVX128_VPDPWSSD(OpcodeArgs, bool Saturating) { + AVX128_VPDPImpl(Op, + [this, Saturating](Ref Acc, Ref Src1, Ref Src2) { return VPDPWSSDOpImpl(OpSize::i128Bit, Acc, Src1, Src2, Saturating); }); +} + void OpDispatchBuilder::AVX128_VBLEND(OpcodeArgs, IR::OpSize ElementSize) { const auto SrcSize = OpSizeFromSrc(Op); const auto Is128Bit = SrcSize == OpSize::i128Bit; diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp index 97d12922b..56e083031 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp @@ -3760,6 +3760,84 @@ void OpDispatchBuilder::VPMADDUBSWOp(OpcodeArgs) { StoreResultFPR(Op, Result); } +Ref OpDispatchBuilder::VPDPBUSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating) { + // Does four 8-bit unsigned * signed byte multiplies per 32-bit element, sums them and accumulates in to the destination + + if (CTX->HostFeatures.SupportsI8MM) { + // The I8MM extension maps onto VPDP* almost directly. + if (!Saturating) { + return _VUSDot(Size, Acc, Src1, Src2); + } + + auto DotProduct = _VUSDot(Size, LoadZeroVector(Size), Src1, Src2); + return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct); + } + + if (CTX->HostFeatures.SupportsDotProd) { + // VSDOT assumes signed input, so we need to convert Src1 to signed, and then + // perform correction afterwards using 0x40 to account for the unsigned input. + auto Src1Signed = _VXor(Size, Src1, _VectorImm(Size, OpSize::i8Bit, 0x80)); + auto SixtyFour = _VectorImm(Size, OpSize::i8Bit, 0x40); + + auto DotProduct = _VSDot(Size, Saturating ? LoadZeroVector(Size) : Acc, Src2, SixtyFour); + DotProduct = _VSDot(Size, DotProduct, Src2, SixtyFour); + DotProduct = _VSDot(Size, DotProduct, Src1Signed, Src2); + if (Saturating) { + return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct); + } + return DotProduct; + } + + // Naive software implementation. + auto Even = _VUnZip(Size, OpSize::i16Bit, Src1, Src2); + auto Even1_16b = _VUXTL(Size, OpSize::i8Bit, Even); + auto Even2_16b = _VSXTL2(Size, OpSize::i8Bit, Even); + auto ResMul_Even = _VMul(Size, OpSize::i16Bit, Even1_16b, Even2_16b); + + auto Odd = _VUnZip2(Size, OpSize::i16Bit, Src1, Src2); + auto Odd1_16b = _VUXTL(Size, OpSize::i8Bit, Odd); + auto Odd2_16b = _VSXTL2(Size, OpSize::i8Bit, Odd); + auto ResMul_Odd = _VMul(Size, OpSize::i16Bit, Odd1_16b, Odd2_16b); + + auto DotProduct = _VSAdALP(Size, OpSize::i16Bit, _VSAddLP(Size, OpSize::i16Bit, ResMul_Even), ResMul_Odd); + if (Saturating) { + return _VSQAdd(Size, OpSize::i32Bit, Acc, DotProduct); + } + return _VAdd(Size, OpSize::i32Bit, Acc, DotProduct); +} + +Ref OpDispatchBuilder::VPDPWSSDOpImpl(IR::OpSize Size, Ref Acc, Ref Src1, Ref Src2, bool Saturating) { + auto DotProduct = PMADDWDOpImpl(Size, Src1, Src2); + if (!Saturating) { + return _VAdd(Size, OpSize::i32Bit, Acc, DotProduct); + } + + auto NegDotProduct = _VNeg(Size, OpSize::i32Bit, DotProduct); + return _VSQSub(Size, OpSize::i32Bit, Acc, NegDotProduct); +} + +void OpDispatchBuilder::VPDPBUSDOp(OpcodeArgs, bool Saturating) { + const auto Size = OpSizeFromSrc(Op); + + Ref Acc = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); + + Ref Result = VPDPBUSDOpImpl(Size, Acc, Src1, Src2, Saturating); + StoreResultFPR(Op, Result); +} + +void OpDispatchBuilder::VPDPWSSDOp(OpcodeArgs, bool Saturating) { + const auto Size = OpSizeFromSrc(Op); + + Ref Acc = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); + + Ref Result = VPDPWSSDOpImpl(Size, Acc, Src1, Src2, Saturating); + StoreResultFPR(Op, Result); +} + Ref OpDispatchBuilder::PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2) { const auto Size = OpSizeFromSrc(Op); if (Signed) { diff --git a/FEXCore/Source/Interface/Core/X86Tables/VEXTables.cpp b/FEXCore/Source/Interface/Core/X86Tables/VEXTables.cpp index 4ecfe5658..4aa60b440 100644 --- a/FEXCore/Source/Interface/Core/X86Tables/VEXTables.cpp +++ b/FEXCore/Source/Interface/Core/X86Tables/VEXTables.cpp @@ -312,6 +312,11 @@ namespace AVX128 { {OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VSSHR>}, // VPSRAVD {OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VariableShiftImpl, IROps::OP_VUSHL>}, // VPSLLV + {OPD(2, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPBUSD, false>}, + {OPD(2, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPBUSD, true>}, + {OPD(2, 0b01, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPWSSD, false>}, + {OPD(2, 0b01, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VPDPWSSD, true>}, + {OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i32Bit>}, {OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i64Bit>}, {OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::AVX128_VBROADCAST, OpSize::i128Bit>}, @@ -757,6 +762,11 @@ namespace AVX256 { {OPD(2, 0b01, 0x46), 1, &OpDispatchBuilder::VPSRAVDOp}, {OPD(2, 0b01, 0x47), 1, &OpDispatchBuilder::VPSLLVOp}, + {OPD(2, 0b01, 0x50), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPBUSDOp, false>}, + {OPD(2, 0b01, 0x51), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPBUSDOp, true>}, + {OPD(2, 0b01, 0x52), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPWSSDOp, false>}, + {OPD(2, 0b01, 0x53), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VPDPWSSDOp, true>}, + {OPD(2, 0b01, 0x58), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i32Bit>}, {OPD(2, 0b01, 0x59), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i64Bit>}, {OPD(2, 0b01, 0x5A), 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VBROADCASTOp, OpSize::i128Bit>}, @@ -1207,6 +1217,11 @@ auto BaseTableLambda = [](const auto RuntimeTable) consteval { {OPD(2, 0b01, 0x46), 1, X86InstInfo{"VPSRAVD", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}}, {OPD(2, 0b01, 0x47), 1, X86InstInfo{"VPSLLV", TYPE_INST, GenFlagsSizes(SIZE_128BIT, SIZE_32BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_XMM_FLAGS, 0}}, + {OPD(2, 0b01, 0x50), 1, X86InstInfo{"VPDPBUSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}}, + {OPD(2, 0b01, 0x51), 1, X86InstInfo{"VPDPBUSDS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}}, + {OPD(2, 0b01, 0x52), 1, X86InstInfo{"VPDPWSSD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}}, + {OPD(2, 0b01, 0x53), 1, X86InstInfo{"VPDPWSSDS", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_1ST_SRC | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}}, + {OPD(2, 0b01, 0x58), 1, X86InstInfo{"VPBROADCASTD", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}}, {OPD(2, 0b01, 0x59), 1, X86InstInfo{"VPBROADCASTQ", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}}, {OPD(2, 0b01, 0x5A), 1, X86InstInfo{"VBROADCASTI128", TYPE_INST, GenFlagsSameSize(SIZE_128BIT) | FLAGS_MODRM | FLAGS_VEX_L_1 | FLAGS_SF_MOD_MEM_ONLY | FLAGS_REX_W_0 | FLAGS_XMM_FLAGS, 0}}, diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index c1a80970f..26aa45371 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -2363,6 +2363,43 @@ "DestSize": "RegisterSize", "ElementSize": "ElementSize << 1" }, + "FPR = VUSDot OpSize:#RegisterSize, FPR:$Acc, FPR:$Vector1, FPR:$Vector2": { + "Desc": ["Unsigned by signed 8-bit dot product, accumulating four products in to each 32-bit element of Acc.", + "Requires FEAT_I8MM." + ], + "DestSize": "RegisterSize", + "ElementSize": "OpSize::i32Bit", + "TiedSource": 0, + "EmitValidation": [ + "RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit" + ] + }, + "FPR = VSDot OpSize:#RegisterSize, FPR:$Acc, FPR:$Vector1, FPR:$Vector2": { + "Desc": ["Signed 8-bit dot product, accumulating four products in to each 32-bit element of Acc.", + "Requires FEAT_DotProd." + ], + "DestSize": "RegisterSize", + "ElementSize": "OpSize::i32Bit", + "TiedSource": 0, + "EmitValidation": [ + "RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i256Bit" + ] + }, + "FPR = VSAddLP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": { + "Desc": ["Signed add long pairwise. Adds adjacent pairs of elements in to elements of twice the size.", + "ElementSize is the source size" + ], + "DestSize": "RegisterSize", + "ElementSize": "ElementSize << 1" + }, + "FPR = VSAdALP OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Acc, FPR:$Vector": { + "Desc": ["Signed add and accumulate long pairwise. Adds adjacent pairs of elements in to the elements of Acc, which are twice the size.", + "ElementSize is the source size" + ], + "DestSize": "RegisterSize", + "ElementSize": "ElementSize << 1", + "TiedSource": 0 + }, "FPR = VUMulH OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": { "Desc": ["Wide unsigned multiply returning the high results"], "DestSize": "RegisterSize", diff --git a/FEXCore/include/FEXCore/Core/HostFeatures.h b/FEXCore/include/FEXCore/Core/HostFeatures.h index 9aa1452a7..d7c5225b3 100644 --- a/FEXCore/include/FEXCore/Core/HostFeatures.h +++ b/FEXCore/include/FEXCore/Core/HostFeatures.h @@ -71,13 +71,15 @@ struct HostFeatures { uint32_t Supports3DNow : 1 {}; uint32_t SupportsSSE4a : 1 {}; uint32_t SupportsMOPS : 1 {}; + uint32_t SupportsI8MM : 1 {}; + uint32_t SupportsDotProd : 1 {}; uint32_t PreferZVAForVZero : 1 {}; uint32_t SupportsAFP : 1 {}; uint32_t SupportsFloatExceptions : 1 {}; // Flag if this is InstCountCI uint32_t IsInstCountCI : 1 {}; HostTypeEnum HostType : 2 {}; - uint32_t pad : 26 {}; + uint32_t pad : 24 {}; // MIDR information // Also used for determining number of CPU cores for CPUID diff --git a/Scripts/InstructionCountParser.py b/Scripts/InstructionCountParser.py index c4b10ec1f..cc0a2b7f4 100755 --- a/Scripts/InstructionCountParser.py +++ b/Scripts/InstructionCountParser.py @@ -60,6 +60,8 @@ class HostFeatures(Flag) : FEATURE_LRCPC2 = (1 << 15) FEATURE_FRINTTS = (1 << 16) FEATURE_MOPS = (1 << 17) + FEATURE_I8MM = (1 << 18) + FEATURE_DOTPROD = (1 << 19) HostFeaturesLookup = { "SVE128" : HostFeatures.FEATURE_SVE128, @@ -80,6 +82,8 @@ HostFeaturesLookup = { "LRCPC2" : HostFeatures.FEATURE_LRCPC2, "FRINTTS" : HostFeatures.FEATURE_FRINTTS, "MOPS" : HostFeatures.FEATURE_MOPS, + "I8MM" : HostFeatures.FEATURE_I8MM, + "DOTPROD" : HostFeatures.FEATURE_DOTPROD, } def GetHostFeatures(data): diff --git a/Scripts/json_config_parse.py b/Scripts/json_config_parse.py index 187aa7b84..c707ef891 100644 --- a/Scripts/json_config_parse.py +++ b/Scripts/json_config_parse.py @@ -87,6 +87,7 @@ class HostFeatures(Flag) : FEATURE_CLFLOPT = (1 << 21) FEATURE_FSGSBASE = (1 << 22) FEATURE_EMMI = (1 << 23) + FEATURE_AVX_VNNI = (1 << 24) RegStringLookup = { "NONE": Regs.REG_NONE, @@ -173,6 +174,7 @@ HostFeaturesLookup = { "CLFLOPT" : HostFeatures.FEATURE_CLFLOPT, "FSGSBASE" : HostFeatures.FEATURE_FSGSBASE, "EMMI" : HostFeatures.FEATURE_EMMI, + "AVX_VNNI" : HostFeatures.FEATURE_AVX_VNNI, } def parse_hexstring(s): diff --git a/Source/Common/HostFeatures.cpp b/Source/Common/HostFeatures.cpp index a7b16fc65..e08db9478 100644 --- a/Source/Common/HostFeatures.cpp +++ b/Source/Common/HostFeatures.cpp @@ -374,6 +374,8 @@ static void OverrideFeatures(FEXCore::HostFeatures* Features, uint64_t ForceSVEW ENABLE_DISABLE_OPTION(Supports3DNow, 3DNOW, 3DNOW); ENABLE_DISABLE_OPTION(SupportsSSE4a, SSE4A, SSE4A); ENABLE_DISABLE_OPTION(SupportsMOPS, MOPS, MOPS); + ENABLE_DISABLE_OPTION(SupportsI8MM, I8MM, I8MM); + ENABLE_DISABLE_OPTION(SupportsDotProd, DOTPROD, DOTPROD); GET_SINGLE_OPTION(Crypto, CRYPTO); #undef ENABLE_DISABLE_OPTION @@ -504,6 +506,8 @@ void FetchHostFeatures(FEX::CPUFeatures& Features, FEXCore::HostFeatures& HostFe HostFeatures.SupportsSVEBitPerm = Features.Supports(CPUFeatures::Feature::SVE_BitPerm); HostFeatures.SupportsECV = Features.Supports(CPUFeatures::Feature::ECV); HostFeatures.SupportsWFXT = Features.Supports(CPUFeatures::Feature::WFxt); + HostFeatures.SupportsI8MM = Features.Supports(CPUFeatures::Feature::I8MM); + HostFeatures.SupportsDotProd = Features.Supports(CPUFeatures::Feature::DotProd); #ifdef VIXL_SIMULATOR // Hardcode enable SVE with 256-bit wide registers. diff --git a/Source/Common/X86Features.h b/Source/Common/X86Features.h index 54d6d3a0b..684c75754 100644 --- a/Source/Common/X86Features.h +++ b/Source/Common/X86Features.h @@ -40,6 +40,11 @@ public: Feat_vaes = data_7.ecx & (1U << 9); Feat_pclmulqdq = Feat_pclmulqdq && (data_7.ecx & (1U << 10)); Feat_rdpid = data_7.ecx & (1U << 22); + + if (data_7.eax >= 1) { + auto data_7_1 = cpuid(0x7, 0x1); + Feat_avx_vnni = Feat_avx && (data_7_1.eax & (1U << 4)); + } } data = cpuid(0x8000'0000U); @@ -79,6 +84,7 @@ public: bool Feat_rdpid {}; bool Feat_clflopt {}; bool Feat_fsgsbase {}; + bool Feat_avx_vnni {}; private: struct cpuid_data { diff --git a/Source/Tools/CodeSizeValidation/Main.cpp b/Source/Tools/CodeSizeValidation/Main.cpp index 07436620a..f8d4e9b16 100644 --- a/Source/Tools/CodeSizeValidation/Main.cpp +++ b/Source/Tools/CodeSizeValidation/Main.cpp @@ -557,6 +557,8 @@ int main(int argc, char** argv, char** const envp) { FEATURE_LRCPC2 = (1U << 15), FEATURE_FRINTTS = (1U << 16), FEATURE_MOPS = (1U << 17), + FEATURE_I8MM = (1U << 18), + FEATURE_DOTPROD = (1U << 19), }; uint64_t SVEWidth = 0; @@ -610,6 +612,12 @@ int main(int argc, char** argv, char** const envp) { if (TestHeaderData->EnabledHostFeatures & FEATURE_MOPS) { HostFeatureControl |= static_cast(FEXCore::Config::HostFeatures::ENABLEMOPS); } + if (TestHeaderData->EnabledHostFeatures & FEATURE_I8MM) { + HostFeatureControl |= static_cast(FEXCore::Config::HostFeatures::ENABLEI8MM); + } + if (TestHeaderData->EnabledHostFeatures & FEATURE_DOTPROD) { + HostFeatureControl |= static_cast(FEXCore::Config::HostFeatures::ENABLEDOTPROD); + } if (TestHeaderData->EnabledHostFeatures & FEATURE_TSO) { FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "1"); @@ -668,6 +676,12 @@ int main(int argc, char** argv, char** const envp) { if (TestHeaderData->DisabledHostFeatures & FEATURE_MOPS) { HostFeatureControl |= static_cast(FEXCore::Config::HostFeatures::DISABLEMOPS); } + if (TestHeaderData->DisabledHostFeatures & FEATURE_I8MM) { + HostFeatureControl |= static_cast(FEXCore::Config::HostFeatures::DISABLEI8MM); + } + if (TestHeaderData->DisabledHostFeatures & FEATURE_DOTPROD) { + HostFeatureControl |= static_cast(FEXCore::Config::HostFeatures::DISABLEDOTPROD); + } if (TestHeaderData->DisabledHostFeatures & FEATURE_TSO) { FEXCore::Config::Set(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, "0"); diff --git a/Source/Tools/CommonTools/HarnessHelpers.h b/Source/Tools/CommonTools/HarnessHelpers.h index 161684658..47a45d305 100644 --- a/Source/Tools/CommonTools/HarnessHelpers.h +++ b/Source/Tools/CommonTools/HarnessHelpers.h @@ -321,6 +321,7 @@ public: FEATURE_CLFLOPT = (1 << 21), FEATURE_FSGSBASE = (1 << 22), FEATURE_EMMI = (1 << 23), + FEATURE_AVX_VNNI = (1 << 24), }; bool Requires3DNow() const { @@ -395,6 +396,9 @@ public: bool RequiresEMMI() const { return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_EMMI; } + bool RequiresAVXVNNI() const { + return BaseConfig.OptionHostFeatures & HostFeatures::FEATURE_AVX_VNNI; + } private: FEX_CONFIG_OPT(ConfigDumpGPRs, DUMPGPRS); @@ -602,6 +606,9 @@ public: bool RequiresEMMI() const { return Config.RequiresEMMI(); } + bool RequiresAVXVNNI() const { + return Config.RequiresAVXVNNI(); + } private: constexpr static uint64_t STACK_OFFSET = 0xc000'0000; diff --git a/Source/Tools/TestHarnessRunner/TestHarnessRunner.cpp b/Source/Tools/TestHarnessRunner/TestHarnessRunner.cpp index 4cda1815a..94742a1b1 100644 --- a/Source/Tools/TestHarnessRunner/TestHarnessRunner.cpp +++ b/Source/Tools/TestHarnessRunner/TestHarnessRunner.cpp @@ -287,6 +287,7 @@ int main(int argc, char** argv, char** const envp) { const bool SupportsRDPID = Feature.Feat_rdpid; const bool SupportsCLFLOPT = Feature.Feat_clflopt; const bool SupportsFSGSBase = Feature.Feat_fsgsbase; + const bool SupportsAVXVNNI = Feature.Feat_avx_vnni; TestUnsupported |= (!Supports3DNow && Loader.Requires3DNow()) || (!SupportsSSE4A && Loader.RequiresSSE4A()) || (!SupportsBMI1 && Loader.RequiresBMI1()) || @@ -295,6 +296,7 @@ int main(int argc, char** argv, char** const envp) { (!SupportsAES && Loader.RequiresAES()) || (!SupportsPCLMUL && Loader.RequiresPCLMUL()) || (!SupportsMOVBE && Loader.RequiresMOVBE()) || (!SupportsADX && Loader.RequiresADX()) || (!SupportsXSAVE && Loader.RequiresXSAVE()) || (!SupportsRDPID && Loader.RequiresRDPID()) || (!SupportsCLFLOPT && Loader.RequiresCLFLOPT()) || (!SupportsFSGSBase && Loader.RequiresFSGSBase()) || Loader.RequiresEMMI(); + TestUnsupported |= !SupportsAVXVNNI && Loader.RequiresAVXVNNI(); #endif #ifdef _WIN32 diff --git a/unittests/ASM/VEX/vpdpbusd.asm b/unittests/ASM/VEX/vpdpbusd.asm new file mode 100644 index 000000000..df3367c34 --- /dev/null +++ b/unittests/ASM/VEX/vpdpbusd.asm @@ -0,0 +1,59 @@ +%ifdef CONFIG +{ + "HostFeatures": ["AVX", "AVX_VNNI"], + "RegData": { + "XMM4": ["0xFFFFFF3100000006", "0x7FFF07008001F904", "0x0000000000000000", "0x0000000000000000"], + "XMM5": ["0xFFFFFFFF00000020", "0x7FFFC08080013B03", "0x0000000000000000", "0x0000000000000000"], + "XMM6": ["0x02FE0140FC03FDF7", "0x807F86807F817983", "0xFCFDFD0101FF8000", "0x807E82807F817983"], + "XMM7": ["0xFFFFFF3100000006", "0x7FFF07008001F904", "0x87654123123456F8", "0x7FFE02108001F9F4"] + } +} +%endif + +vmovdqa ymm0, [rel src1] +vmovdqa ymm1, [rel src2] +vmovdqa ymm4, [rel accumulator] +vmovdqa ymm5, [rel accumulator] +vmovdqa ymm6, [rel src2] +vmovdqa ymm7, [rel accumulator] +lea rdx, [rel src2] + +; Hand assembled instructions, since NASM will only output +; the EVEX encodings for these instructions. +db 0xC4, 0xE2, 0x79, 0x50, 0xE1 ; vpdpbusd xmm4, xmm0, xmm1 +db 0xC4, 0xE2, 0x51, 0x50, 0x2A ; vpdpbusd xmm5, xmm5, [rdx] +db 0xC4, 0xE2, 0x7D, 0x50, 0xF6 ; vpdpbusd ymm6, ymm0, ymm6 +db 0xC4, 0xE2, 0x7D, 0x50, 0x3A ; vpdpbusd ymm7, ymm0, [rdx] + +hlt + +align 32 +accumulator: +dd 0x00000010 +dd 0xFFFFFFF0 +dd 0x7FFFFF00 +dd 0x80000100 +dd 0x12345678 +dd 0x87654321 +dd 0x7FFFFFF0 +dd 0x80000010 + +src1: +db 1, 2, 3, 4 +db 255, 128, 64, 32 +db 255, 255, 255, 255 +db 200, 150, 100, 50 +db 0, 1, 254, 255 +db 17, 34, 51, 68 +db 255, 255, 255, 255 +db 255, 255, 255, 255 + +src2: +db 1, -2, 3, -4 +db -1, 1, -2, 2 +db 127, 127, 127, 127 +db -128, -128, -128, -128 +db -128, 127, -1, 1 +db -1, -2, -3, -4 +db 127, 127, 127, 127 +db -128, -128, -128, -128 diff --git a/unittests/ASM/VEX/vpdpbusds.asm b/unittests/ASM/VEX/vpdpbusds.asm new file mode 100644 index 000000000..df8b9dae4 --- /dev/null +++ b/unittests/ASM/VEX/vpdpbusds.asm @@ -0,0 +1,59 @@ +%ifdef CONFIG +{ + "HostFeatures": ["AVX", "AVX_VNNI"], + "RegData": { + "XMM4": ["0xFFFFFF3100000006", "0x800000007FFFFFFF", "0x0000000000000000", "0x0000000000000000"], + "XMM5": ["0xFFFFFFFF00000020", "0x800000007FFFFFFF", "0x0000000000000000", "0x0000000000000000"], + "XMM6": ["0x02FE0140FC03FDF7", "0x807F86807F817983", "0xFCFDFD0101FF8000", "0x807E82807F817983"], + "XMM7": ["0xFFFFFF3100000006", "0x800000007FFFFFFF", "0x87654123123456F8", "0x800000007FFFFFFF"] + } +} +%endif + +vmovdqa ymm0, [rel src1] +vmovdqa ymm1, [rel src2] +vmovdqa ymm4, [rel accumulator] +vmovdqa ymm5, [rel accumulator] +vmovdqa ymm6, [rel src2] +vmovdqa ymm7, [rel accumulator] +lea rdx, [rel src2] + +; Hand assembled instructions, since NASM will only output +; the EVEX encodings for these instructions. +db 0xC4, 0xE2, 0x79, 0x51, 0xE1 ; vpdpbusds xmm4, xmm0, xmm1 +db 0xC4, 0xE2, 0x51, 0x51, 0x2A ; vpdpbusds xmm5, xmm5, [rdx] +db 0xC4, 0xE2, 0x7D, 0x51, 0xF6 ; vpdpbusds ymm6, ymm0, ymm6 +db 0xC4, 0xE2, 0x7D, 0x51, 0x3A ; vpdpbusds ymm7, ymm0, [rdx] + +hlt + +align 32 +accumulator: +dd 0x00000010 +dd 0xFFFFFFF0 +dd 0x7FFFFF00 +dd 0x80000100 +dd 0x12345678 +dd 0x87654321 +dd 0x7FFFFFF0 +dd 0x80000010 + +src1: +db 1, 2, 3, 4 +db 255, 128, 64, 32 +db 255, 255, 255, 255 +db 200, 150, 100, 50 +db 0, 1, 254, 255 +db 17, 34, 51, 68 +db 255, 255, 255, 255 +db 255, 255, 255, 255 + +src2: +db 1, -2, 3, -4 +db -1, 1, -2, 2 +db 127, 127, 127, 127 +db -128, -128, -128, -128 +db -128, 127, -1, 1 +db -1, -2, -3, -4 +db 127, 127, 127, 127 +db -128, -128, -128, -128 diff --git a/unittests/ASM/VEX/vpdpwssd.asm b/unittests/ASM/VEX/vpdpwssd.asm new file mode 100644 index 000000000..2c36cf3fa --- /dev/null +++ b/unittests/ASM/VEX/vpdpwssd.asm @@ -0,0 +1,59 @@ +%ifdef CONFIG +{ + "HostFeatures": ["AVX", "AVX_VNNI"], + "RegData": { + "XMM4": ["0x800000000000000B", "0xFFFF000180000002", "0x0000000000000000", "0x0000000000000000"], + "XMM5": ["0x0000000000000040", "0x400080000002FFFE", "0x0000000000000000", "0x0000000000000000"], + "XMM6": ["0x00008000FFFBFFFE", "0xFFFE8001FFFD8001", "0x800100001C173F20", "0x0001800000008000"], + "XMM7": ["0x800000000000000B", "0xFFFF000180000002", "0x8765C322E02B0AC8", "0x000100107FFFFFFE"] + } +} +%endif + +vmovdqa ymm0, [rel src1] +vmovdqa ymm1, [rel src2] +vmovdqa ymm4, [rel accumulator] +vmovdqa ymm5, [rel accumulator] +vmovdqa ymm6, [rel src2] +vmovdqa ymm7, [rel accumulator] +lea rdx, [rel src2] + +; Hand assembled instructions, since NASM will only output +; the EVEX encodings for these instructions +db 0xC4, 0xE2, 0x79, 0x52, 0xE1 ; vpdpwssd xmm4, xmm0, xmm1 +db 0xC4, 0xE2, 0x51, 0x52, 0x2A ; vpdpwssd xmm5, xmm5, [rdx] +db 0xC4, 0xE2, 0x7D, 0x52, 0xF6 ; vpdpwssd ymm6, ymm0, ymm6 +db 0xC4, 0xE2, 0x7D, 0x52, 0x3A ; vpdpwssd ymm7, ymm0, [rdx] + +hlt + +align 32 +accumulator: +dd 0x00000010 +dd 0x00000000 +dd 0x00020000 +dd 0x80000000 +dd 0x12345678 +dd 0x87654321 +dd 0xFFFFFFFE +dd 0x80000010 + +src1: +dw 1, 2 +dw -32768, -32768 +dw 32767, 32767 +dw -32768, 32767 +dw 12345, -23456 +dw -1, -2 +dw -32768, -32768 +dw 32767, 32767 + +src2: +dw 3, -4 +dw -32768, -32768 +dw 32767, 32767 +dw -32768, 32767 +dw -30000, 20000 +dw 32767, -32768 +dw -32768, -32768 +dw -32768, -32768 diff --git a/unittests/ASM/VEX/vpdpwssds.asm b/unittests/ASM/VEX/vpdpwssds.asm new file mode 100644 index 000000000..0935f5449 --- /dev/null +++ b/unittests/ASM/VEX/vpdpwssds.asm @@ -0,0 +1,59 @@ +%ifdef CONFIG +{ + "HostFeatures": ["AVX", "AVX_VNNI"], + "RegData": { + "XMM4": ["0x7FFFFFFF0000000B", "0xFFFF00017FFFFFFF", "0x0000000000000000", "0x0000000000000000"], + "XMM5": ["0x0000000000000040", "0x800000000002FFFE", "0x0000000000000000", "0x0000000000000000"], + "XMM6": ["0x00008000FFFBFFFE", "0x7FFFFFFF7FFFFFFF", "0x800100001C173F20", "0x8000000000008000"], + "XMM7": ["0x7FFFFFFF0000000B", "0xFFFF00017FFFFFFF", "0x8765C322E02B0AC8", "0x800000007FFFFFFE"] + } +} +%endif + +vmovdqa ymm0, [rel src1] +vmovdqa ymm1, [rel src2] +vmovdqa ymm4, [rel accumulator] +vmovdqa ymm5, [rel accumulator] +vmovdqa ymm6, [rel src2] +vmovdqa ymm7, [rel accumulator] +lea rdx, [rel src2] + +; Hand assembled instructions, since NASM will only output +; the EVEX encodings for these instructions. +db 0xC4, 0xE2, 0x79, 0x53, 0xE1 ; vpdpwssds xmm4, xmm0, xmm1 +db 0xC4, 0xE2, 0x51, 0x53, 0x2A ; vpdpwssds xmm5, xmm5, [rdx] +db 0xC4, 0xE2, 0x7D, 0x53, 0xF6 ; vpdpwssds ymm6, ymm0, ymm6 +db 0xC4, 0xE2, 0x7D, 0x53, 0x3A ; vpdpwssds ymm7, ymm0, [rdx] + +hlt + +align 32 +accumulator: +dd 0x00000010 +dd 0x00000000 +dd 0x00020000 +dd 0x80000000 +dd 0x12345678 +dd 0x87654321 +dd 0xFFFFFFFE +dd 0x80000010 + +src1: +dw 1, 2 +dw -32768, -32768 +dw 32767, 32767 +dw -32768, 32767 +dw 12345, -23456 +dw -1, -2 +dw -32768, -32768 +dw 32767, 32767 + +src2: +dw 3, -4 +dw -32768, -32768 +dw 32767, 32767 +dw -32768, 32767 +dw -30000, 20000 +dw 32767, -32768 +dw -32768, -32768 +dw -32768, -32768 diff --git a/unittests/InstructionCountCI/AVX128/VEX_map2.json b/unittests/InstructionCountCI/AVX128/VEX_map2.json index 2cd8cfc04..160d0eb91 100644 --- a/unittests/InstructionCountCI/AVX128/VEX_map2.json +++ b/unittests/InstructionCountCI/AVX128/VEX_map2.json @@ -7,7 +7,9 @@ "FLAGM", "FLAGM2", "SVE128", - "SVE256" + "SVE256", + "I8MM", + "DOTPROD" ], "BinaryCacheVersion": 20 }, @@ -2075,6 +2077,231 @@ "str q2, [x28, #192]" ] }, + "vpdpbusd xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 12, + "Comment": [ + "Map 2 0b01 0x50 128-bit", + "vpdpbusd xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x50, 0xc2" + ], + "ExpectedArm64ASM": [ + "uzp1 v2.8h, v17.8h, v18.8h", + "uxtl v3.8h, v2.8b", + "sxtl2 v2.8h, v2.16b", + "mul v2.8h, v3.8h, v2.8h", + "uzp2 v3.8h, v17.8h, v18.8h", + "uxtl v4.8h, v3.8b", + "sxtl2 v3.8h, v3.16b", + "mul v3.8h, v4.8h, v3.8h", + "saddlp v2.4s, v2.8h", + "sadalp v2.4s, v3.8h", + "add v16.4s, v16.4s, v2.4s", + "stp xzr, xzr, [x28, #192]" + ] + }, + "vpdpbusd ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 26, + "Comment": [ + "Map 2 0b01 0x50 256-bit", + "vpdpbusd ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x50, 0xc2" + ], + "ExpectedArm64ASM": [ + "ldr q2, [x28, #192]", + "ldr q3, [x28, #208]", + "ldr q4, [x28, #224]", + "uzp1 v5.8h, v17.8h, v18.8h", + "uxtl v6.8h, v5.8b", + "sxtl2 v5.8h, v5.16b", + "mul v5.8h, v6.8h, v5.8h", + "uzp2 v6.8h, v17.8h, v18.8h", + "uxtl v7.8h, v6.8b", + "sxtl2 v6.8h, v6.16b", + "mul v6.8h, v7.8h, v6.8h", + "saddlp v5.4s, v5.8h", + "sadalp v5.4s, v6.8h", + "add v16.4s, v16.4s, v5.4s", + "uzp1 v5.8h, v3.8h, v4.8h", + "uxtl v6.8h, v5.8b", + "sxtl2 v5.8h, v5.16b", + "mul v5.8h, v6.8h, v5.8h", + "uzp2 v3.8h, v3.8h, v4.8h", + "uxtl v4.8h, v3.8b", + "sxtl2 v3.8h, v3.16b", + "mul v3.8h, v4.8h, v3.8h", + "saddlp v4.4s, v5.8h", + "sadalp v4.4s, v3.8h", + "add v2.4s, v2.4s, v4.4s", + "str q2, [x28, #192]" + ] + }, + "vpdpbusds xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 12, + "Comment": [ + "Map 2 0b01 0x51 128-bit", + "vpdpbusds xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x51, 0xc2" + ], + "ExpectedArm64ASM": [ + "uzp1 v2.8h, v17.8h, v18.8h", + "uxtl v3.8h, v2.8b", + "sxtl2 v2.8h, v2.16b", + "mul v2.8h, v3.8h, v2.8h", + "uzp2 v3.8h, v17.8h, v18.8h", + "uxtl v4.8h, v3.8b", + "sxtl2 v3.8h, v3.16b", + "mul v3.8h, v4.8h, v3.8h", + "saddlp v2.4s, v2.8h", + "sadalp v2.4s, v3.8h", + "sqadd v16.4s, v16.4s, v2.4s", + "stp xzr, xzr, [x28, #192]" + ] + }, + "vpdpbusds ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 26, + "Comment": [ + "Map 2 0b01 0x51 256-bit", + "vpdpbusds ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x51, 0xc2" + ], + "ExpectedArm64ASM": [ + "ldr q2, [x28, #192]", + "ldr q3, [x28, #208]", + "ldr q4, [x28, #224]", + "uzp1 v5.8h, v17.8h, v18.8h", + "uxtl v6.8h, v5.8b", + "sxtl2 v5.8h, v5.16b", + "mul v5.8h, v6.8h, v5.8h", + "uzp2 v6.8h, v17.8h, v18.8h", + "uxtl v7.8h, v6.8b", + "sxtl2 v6.8h, v6.16b", + "mul v6.8h, v7.8h, v6.8h", + "saddlp v5.4s, v5.8h", + "sadalp v5.4s, v6.8h", + "sqadd v16.4s, v16.4s, v5.4s", + "uzp1 v5.8h, v3.8h, v4.8h", + "uxtl v6.8h, v5.8b", + "sxtl2 v5.8h, v5.16b", + "mul v5.8h, v6.8h, v5.8h", + "uzp2 v3.8h, v3.8h, v4.8h", + "uxtl v4.8h, v3.8b", + "sxtl2 v3.8h, v3.16b", + "mul v3.8h, v4.8h, v3.8h", + "saddlp v4.4s, v5.8h", + "sadalp v4.4s, v3.8h", + "sqadd v2.4s, v2.4s, v4.4s", + "str q2, [x28, #192]" + ] + }, + "vpdpwssd xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 5, + "Comment": [ + "Map 2 0b01 0x52 128-bit", + "vpdpwssd xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x52, 0xc2" + ], + "ExpectedArm64ASM": [ + "smull v2.4s, v17.4h, v18.4h", + "smull2 v3.4s, v17.8h, v18.8h", + "addp v2.4s, v2.4s, v3.4s", + "add v16.4s, v16.4s, v2.4s", + "stp xzr, xzr, [x28, #192]" + ] + }, + "vpdpwssd ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 12, + "Comment": [ + "Map 2 0b01 0x52 256-bit", + "vpdpwssd ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x52, 0xc2" + ], + "ExpectedArm64ASM": [ + "ldr q2, [x28, #192]", + "ldr q3, [x28, #208]", + "ldr q4, [x28, #224]", + "smull v5.4s, v17.4h, v18.4h", + "smull2 v6.4s, v17.8h, v18.8h", + "addp v5.4s, v5.4s, v6.4s", + "add v16.4s, v16.4s, v5.4s", + "smull v5.4s, v3.4h, v4.4h", + "smull2 v3.4s, v3.8h, v4.8h", + "addp v3.4s, v5.4s, v3.4s", + "add v2.4s, v2.4s, v3.4s", + "str q2, [x28, #192]" + ] + }, + "vpdpwssds xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 6, + "Comment": [ + "Map 2 0b01 0x53 128-bit", + "vpdpwssds xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x53, 0xc2" + ], + "ExpectedArm64ASM": [ + "smull v2.4s, v17.4h, v18.4h", + "smull2 v3.4s, v17.8h, v18.8h", + "addp v2.4s, v2.4s, v3.4s", + "neg v2.4s, v2.4s", + "sqsub v16.4s, v16.4s, v2.4s", + "stp xzr, xzr, [x28, #192]" + ] + }, + "vpdpwssds ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 14, + "Comment": [ + "Map 2 0b01 0x53 256-bit", + "vpdpwssds ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x53, 0xc2" + ], + "ExpectedArm64ASM": [ + "ldr q2, [x28, #192]", + "ldr q3, [x28, #208]", + "ldr q4, [x28, #224]", + "smull v5.4s, v17.4h, v18.4h", + "smull2 v6.4s, v17.8h, v18.8h", + "addp v5.4s, v5.4s, v6.4s", + "neg v5.4s, v5.4s", + "sqsub v16.4s, v16.4s, v5.4s", + "smull v5.4s, v3.4h, v4.4h", + "smull2 v3.4s, v3.8h, v4.8h", + "addp v3.4s, v5.4s, v3.4s", + "neg v3.4s, v3.4s", + "sqsub v2.4s, v2.4s, v3.4s", + "str q2, [x28, #192]" + ] + }, "vpbroadcastd xmm0, xmm1": { "ExpectedInstructionCount": 2, "Comment": [ diff --git a/unittests/InstructionCountCI/AVX128/VEX_map2_DOTPROD.json b/unittests/InstructionCountCI/AVX128/VEX_map2_DOTPROD.json new file mode 100644 index 000000000..3910b52f1 --- /dev/null +++ b/unittests/InstructionCountCI/AVX128/VEX_map2_DOTPROD.json @@ -0,0 +1,131 @@ +{ + "Features": { + "Bitness": 64, + "EnabledHostFeatures": [ + "DOTPROD" + ], + "DisabledHostFeatures": [ + "AFP", + "FLAGM", + "FLAGM2", + "SVE128", + "SVE256", + "I8MM" + ], + "BinaryCacheVersion": 20 + }, + "Instructions": { + "vpdpbusd xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 9, + "Comment": [ + "Map 2 0b01 0x50 128-bit", + "vpdpbusd xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x50, 0xc2" + ], + "ExpectedArm64ASM": [ + "movi v2.16b, #0x80", + "eor v2.16b, v17.16b, v2.16b", + "movi v3.16b, #0x40", + "mov v4.16b, v16.16b", + "sdot v4.4s, v18.16b, v3.16b", + "sdot v4.4s, v18.16b, v3.16b", + "mov v16.16b, v4.16b", + "sdot v16.4s, v2.16b, v18.16b", + "stp xzr, xzr, [x28, #192]" + ] + }, + "vpdpbusd ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 18, + "Comment": [ + "Map 2 0b01 0x50 256-bit", + "vpdpbusd ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x50, 0xc2" + ], + "ExpectedArm64ASM": [ + "ldr q2, [x28, #192]", + "ldr q3, [x28, #208]", + "ldr q4, [x28, #224]", + "movi v5.16b, #0x80", + "eor v5.16b, v17.16b, v5.16b", + "movi v6.16b, #0x40", + "mov v7.16b, v16.16b", + "sdot v7.4s, v18.16b, v6.16b", + "sdot v7.4s, v18.16b, v6.16b", + "mov v16.16b, v7.16b", + "sdot v16.4s, v5.16b, v18.16b", + "movi v5.16b, #0x80", + "eor v3.16b, v3.16b, v5.16b", + "movi v5.16b, #0x40", + "sdot v2.4s, v4.16b, v5.16b", + "sdot v2.4s, v4.16b, v5.16b", + "sdot v2.4s, v3.16b, v4.16b", + "str q2, [x28, #192]" + ] + }, + "vpdpbusds xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 9, + "Comment": [ + "Map 2 0b01 0x51 128-bit", + "vpdpbusds xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x51, 0xc2" + ], + "ExpectedArm64ASM": [ + "movi v2.16b, #0x80", + "eor v2.16b, v17.16b, v2.16b", + "movi v3.16b, #0x40", + "movi v4.2d, #0x0", + "sdot v4.4s, v18.16b, v3.16b", + "sdot v4.4s, v18.16b, v3.16b", + "sdot v4.4s, v2.16b, v18.16b", + "sqadd v16.4s, v16.4s, v4.4s", + "stp xzr, xzr, [x28, #192]" + ] + }, + "vpdpbusds ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 20, + "Comment": [ + "Map 2 0b01 0x51 256-bit", + "vpdpbusds ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x51, 0xc2" + ], + "ExpectedArm64ASM": [ + "ldr q2, [x28, #192]", + "ldr q3, [x28, #208]", + "ldr q4, [x28, #224]", + "movi v5.16b, #0x80", + "eor v5.16b, v17.16b, v5.16b", + "movi v6.16b, #0x40", + "movi v7.2d, #0x0", + "mov v8.16b, v7.16b", + "sdot v8.4s, v18.16b, v6.16b", + "sdot v8.4s, v18.16b, v6.16b", + "sdot v8.4s, v5.16b, v18.16b", + "sqadd v16.4s, v16.4s, v8.4s", + "movi v5.16b, #0x80", + "eor v3.16b, v3.16b, v5.16b", + "movi v5.16b, #0x40", + "sdot v7.4s, v4.16b, v5.16b", + "sdot v7.4s, v4.16b, v5.16b", + "sdot v7.4s, v3.16b, v4.16b", + "sqadd v2.4s, v2.4s, v7.4s", + "str q2, [x28, #192]" + ] + } + } +} diff --git a/unittests/InstructionCountCI/AVX128/VEX_map2_I8MM.json b/unittests/InstructionCountCI/AVX128/VEX_map2_I8MM.json new file mode 100644 index 000000000..7adde3763 --- /dev/null +++ b/unittests/InstructionCountCI/AVX128/VEX_map2_I8MM.json @@ -0,0 +1,189 @@ +{ + "Features": { + "Bitness": 64, + "EnabledHostFeatures": [ + "I8MM" + ], + "DisabledHostFeatures": [ + "AFP", + "FLAGM", + "FLAGM2", + "SVE128", + "SVE256" + ], + "BinaryCacheVersion": 20 + }, + "Instructions": { + "vpdpbusd xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 2, + "Comment": [ + "Map 2 0b01 0x50 128-bit", + "vpdpbusd xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x50, 0xc2" + ], + "ExpectedArm64ASM": [ + "usdot v16.4s, v17.16b, v18.16b", + "stp xzr, xzr, [x28, #192]" + ] + }, + "vpdpbusd ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 6, + "Comment": [ + "Map 2 0b01 0x50 256-bit", + "vpdpbusd ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x50, 0xc2" + ], + "ExpectedArm64ASM": [ + "ldr q2, [x28, #192]", + "ldr q3, [x28, #208]", + "ldr q4, [x28, #224]", + "usdot v16.4s, v17.16b, v18.16b", + "usdot v2.4s, v3.16b, v4.16b", + "str q2, [x28, #192]" + ] + }, + "vpdpbusds xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 4, + "Comment": [ + "Map 2 0b01 0x51 128-bit", + "vpdpbusds xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x51, 0xc2" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "usdot v2.4s, v17.16b, v18.16b", + "sqadd v16.4s, v16.4s, v2.4s", + "stp xzr, xzr, [x28, #192]" + ] + }, + "vpdpbusds ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 10, + "Comment": [ + "Map 2 0b01 0x51 256-bit", + "vpdpbusds ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x51, 0xc2" + ], + "ExpectedArm64ASM": [ + "ldr q2, [x28, #192]", + "ldr q3, [x28, #208]", + "ldr q4, [x28, #224]", + "movi v5.2d, #0x0", + "mov v6.16b, v5.16b", + "usdot v6.4s, v17.16b, v18.16b", + "sqadd v16.4s, v16.4s, v6.4s", + "usdot v5.4s, v3.16b, v4.16b", + "sqadd v2.4s, v2.4s, v5.4s", + "str q2, [x28, #192]" + ] + }, + "vpdpwssd xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 5, + "Comment": [ + "Map 2 0b01 0x52 128-bit", + "vpdpwssd xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x52, 0xc2" + ], + "ExpectedArm64ASM": [ + "smull v2.4s, v17.4h, v18.4h", + "smull2 v3.4s, v17.8h, v18.8h", + "addp v2.4s, v2.4s, v3.4s", + "add v16.4s, v16.4s, v2.4s", + "stp xzr, xzr, [x28, #192]" + ] + }, + "vpdpwssd ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 12, + "Comment": [ + "Map 2 0b01 0x52 256-bit", + "vpdpwssd ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x52, 0xc2" + ], + "ExpectedArm64ASM": [ + "ldr q2, [x28, #192]", + "ldr q3, [x28, #208]", + "ldr q4, [x28, #224]", + "smull v5.4s, v17.4h, v18.4h", + "smull2 v6.4s, v17.8h, v18.8h", + "addp v5.4s, v5.4s, v6.4s", + "add v16.4s, v16.4s, v5.4s", + "smull v5.4s, v3.4h, v4.4h", + "smull2 v3.4s, v3.8h, v4.8h", + "addp v3.4s, v5.4s, v3.4s", + "add v2.4s, v2.4s, v3.4s", + "str q2, [x28, #192]" + ] + }, + "vpdpwssds xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 6, + "Comment": [ + "Map 2 0b01 0x53 128-bit", + "vpdpwssds xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x53, 0xc2" + ], + "ExpectedArm64ASM": [ + "smull v2.4s, v17.4h, v18.4h", + "smull2 v3.4s, v17.8h, v18.8h", + "addp v2.4s, v2.4s, v3.4s", + "neg v2.4s, v2.4s", + "sqsub v16.4s, v16.4s, v2.4s", + "stp xzr, xzr, [x28, #192]" + ] + }, + "vpdpwssds ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 14, + "Comment": [ + "Map 2 0b01 0x53 256-bit", + "vpdpwssds ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x53, 0xc2" + ], + "ExpectedArm64ASM": [ + "ldr q2, [x28, #192]", + "ldr q3, [x28, #208]", + "ldr q4, [x28, #224]", + "smull v5.4s, v17.4h, v18.4h", + "smull2 v6.4s, v17.8h, v18.8h", + "addp v5.4s, v5.4s, v6.4s", + "neg v5.4s, v5.4s", + "sqsub v16.4s, v16.4s, v5.4s", + "smull v5.4s, v3.4h, v4.4h", + "smull2 v3.4s, v3.8h, v4.8h", + "addp v3.4s, v5.4s, v3.4s", + "neg v3.4s, v3.4s", + "sqsub v2.4s, v2.4s, v3.4s", + "str q2, [x28, #192]" + ] + } + } +} diff --git a/unittests/InstructionCountCI/FlagM/VEX_map1.json b/unittests/InstructionCountCI/FlagM/VEX_map1.json index eb9221608..b8712f6f3 100644 --- a/unittests/InstructionCountCI/FlagM/VEX_map1.json +++ b/unittests/InstructionCountCI/FlagM/VEX_map1.json @@ -95,8 +95,8 @@ "msr nzcv, x0", "and z2.d, z3.d, z2.d", "addp z2.b, p7/m, z2.b, z2.b", - "uzp1 z2.b, z2.b, z2.b", "uzp2 z1.b, z2.b, z2.b", + "uzp1 z2.b, z2.b, z2.b", "splice z2.d, p6, z2.d, z1.d", "addp v2.16b, v2.16b, v2.16b", "addp v2.8b, v2.8b, v2.8b", diff --git a/unittests/InstructionCountCI/VEX_map1.json b/unittests/InstructionCountCI/VEX_map1.json index 992f9694b..23f6b3cb4 100644 --- a/unittests/InstructionCountCI/VEX_map1.json +++ b/unittests/InstructionCountCI/VEX_map1.json @@ -4540,8 +4540,8 @@ "msr nzcv, x0", "and z2.d, z3.d, z2.d", "addp z2.b, p7/m, z2.b, z2.b", - "uzp1 z2.b, z2.b, z2.b", "uzp2 z1.b, z2.b, z2.b", + "uzp1 z2.b, z2.b, z2.b", "splice z2.d, p6, z2.d, z1.d", "addp v2.16b, v2.16b, v2.16b", "addp v2.8b, v2.8b, v2.8b", @@ -5357,8 +5357,8 @@ "zip2 z3.s, z0.s, z1.s", "movprfx z0, z2", "addp z0.s, p7/m, z0.s, z3.s", - "uzp1 z16.s, z0.s, z0.s", "uzp2 z1.s, z0.s, z0.s", + "uzp1 z16.s, z0.s, z0.s", "splice z16.d, p6, z16.d, z1.d" ] }, diff --git a/unittests/InstructionCountCI/VEX_map2.json b/unittests/InstructionCountCI/VEX_map2.json index 7bf4ebae3..59d975e7f 100644 --- a/unittests/InstructionCountCI/VEX_map2.json +++ b/unittests/InstructionCountCI/VEX_map2.json @@ -9,7 +9,9 @@ "AFP", "FLAGM", "FLAGM2", - "SVEBITPERM" + "SVEBITPERM", + "I8MM", + "DOTPROD" ], "BinaryCacheVersion": 20 }, @@ -59,8 +61,8 @@ "ExpectedArm64ASM": [ "movprfx z0, z17", "addp z0.h, p7/m, z0.h, z18.h", - "uzp1 z2.h, z0.h, z0.h", "uzp2 z1.h, z0.h, z0.h", + "uzp1 z2.h, z0.h, z0.h", "splice z2.d, p6, z2.d, z1.d", "ldr x0, [x28, #2552]", "ld1b {z3.b}, p7/z, [x0]", @@ -84,8 +86,8 @@ "ExpectedArm64ASM": [ "movprfx z0, z17", "addp z0.s, p7/m, z0.s, z18.s", - "uzp1 z2.s, z0.s, z0.s", "uzp2 z1.s, z0.s, z0.s", + "uzp1 z2.s, z0.s, z0.s", "splice z2.d, p6, z2.d, z1.d", "ldr x0, [x28, #2552]", "ld1b {z3.b}, p7/z, [x0]", @@ -1639,6 +1641,198 @@ "lsl z16.d, p7/m, z16.d, z1.d" ] }, + "vpdpbusd xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 11, + "Comment": [ + "Map 2 0b01 0x50 128-bit", + "vpdpbusd xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x50, 0xc2" + ], + "ExpectedArm64ASM": [ + "uzp1 v2.8h, v17.8h, v18.8h", + "uxtl v3.8h, v2.8b", + "sxtl2 v2.8h, v2.16b", + "mul v2.8h, v3.8h, v2.8h", + "uzp2 v3.8h, v17.8h, v18.8h", + "uxtl v4.8h, v3.8b", + "sxtl2 v3.8h, v3.16b", + "mul v3.8h, v4.8h, v3.8h", + "saddlp v2.4s, v2.8h", + "sadalp v2.4s, v3.8h", + "add v16.4s, v16.4s, v2.4s" + ] + }, + "vpdpbusd ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 13, + "Comment": [ + "Map 2 0b01 0x50 256-bit", + "vpdpbusd ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x50, 0xc2" + ], + "ExpectedArm64ASM": [ + "uzp1 z2.h, z17.h, z18.h", + "uunpklo z3.h, z2.b", + "sunpkhi z2.h, z2.b", + "mul z2.h, z3.h, z2.h", + "uzp2 z3.h, z17.h, z18.h", + "uunpklo z4.h, z3.b", + "sunpkhi z3.h, z3.b", + "mul z3.h, z4.h, z3.h", + "mov z0.s, #0", + "sadalp z0.s, p7/m, z2.h", + "mov z2.d, z0.d", + "sadalp z2.s, p7/m, z3.h", + "add z16.s, z16.s, z2.s" + ] + }, + "vpdpbusds xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 11, + "Comment": [ + "Map 2 0b01 0x51 128-bit", + "vpdpbusds xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x51, 0xc2" + ], + "ExpectedArm64ASM": [ + "uzp1 v2.8h, v17.8h, v18.8h", + "uxtl v3.8h, v2.8b", + "sxtl2 v2.8h, v2.16b", + "mul v2.8h, v3.8h, v2.8h", + "uzp2 v3.8h, v17.8h, v18.8h", + "uxtl v4.8h, v3.8b", + "sxtl2 v3.8h, v3.16b", + "mul v3.8h, v4.8h, v3.8h", + "saddlp v2.4s, v2.8h", + "sadalp v2.4s, v3.8h", + "sqadd v16.4s, v16.4s, v2.4s" + ] + }, + "vpdpbusds ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 13, + "Comment": [ + "Map 2 0b01 0x51 256-bit", + "vpdpbusds ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x51, 0xc2" + ], + "ExpectedArm64ASM": [ + "uzp1 z2.h, z17.h, z18.h", + "uunpklo z3.h, z2.b", + "sunpkhi z2.h, z2.b", + "mul z2.h, z3.h, z2.h", + "uzp2 z3.h, z17.h, z18.h", + "uunpklo z4.h, z3.b", + "sunpkhi z3.h, z3.b", + "mul z3.h, z4.h, z3.h", + "mov z0.s, #0", + "sadalp z0.s, p7/m, z2.h", + "mov z2.d, z0.d", + "sadalp z2.s, p7/m, z3.h", + "sqadd z16.s, z16.s, z2.s" + ] + }, + "vpdpwssd xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 4, + "Comment": [ + "Map 2 0b01 0x52 128-bit", + "vpdpwssd xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x52, 0xc2" + ], + "ExpectedArm64ASM": [ + "smull v2.4s, v17.4h, v18.4h", + "smull2 v3.4s, v17.8h, v18.8h", + "addp v2.4s, v2.4s, v3.4s", + "add v16.4s, v16.4s, v2.4s" + ] + }, + "vpdpwssd ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 11, + "Comment": [ + "Map 2 0b01 0x52 256-bit", + "vpdpwssd ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x52, 0xc2" + ], + "ExpectedArm64ASM": [ + "smullb z0.s, z17.h, z18.h", + "smullt z1.s, z17.h, z18.h", + "zip1 z2.s, z0.s, z1.s", + "smullb z0.s, z17.h, z18.h", + "smullt z1.s, z17.h, z18.h", + "zip2 z3.s, z0.s, z1.s", + "addp z2.s, p7/m, z2.s, z3.s", + "uzp2 z1.s, z2.s, z2.s", + "uzp1 z2.s, z2.s, z2.s", + "splice z2.d, p6, z2.d, z1.d", + "add z16.s, z16.s, z2.s" + ] + }, + "vpdpwssds xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 5, + "Comment": [ + "Map 2 0b01 0x53 128-bit", + "vpdpwssds xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x53, 0xc2" + ], + "ExpectedArm64ASM": [ + "smull v2.4s, v17.4h, v18.4h", + "smull2 v3.4s, v17.8h, v18.8h", + "addp v2.4s, v2.4s, v3.4s", + "neg v2.4s, v2.4s", + "sqsub v16.4s, v16.4s, v2.4s" + ] + }, + "vpdpwssds ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 12, + "Comment": [ + "Map 2 0b01 0x53 256-bit", + "vpdpwssds ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x53, 0xc2" + ], + "ExpectedArm64ASM": [ + "smullb z0.s, z17.h, z18.h", + "smullt z1.s, z17.h, z18.h", + "zip1 z2.s, z0.s, z1.s", + "smullb z0.s, z17.h, z18.h", + "smullt z1.s, z17.h, z18.h", + "zip2 z3.s, z0.s, z1.s", + "addp z2.s, p7/m, z2.s, z3.s", + "uzp2 z1.s, z2.s, z2.s", + "uzp1 z2.s, z2.s, z2.s", + "splice z2.d, p6, z2.d, z1.d", + "neg z2.s, p7/m, z2.s", + "sqsub z16.s, z16.s, z2.s" + ] + }, "vpbroadcastd xmm0, xmm1": { "ExpectedInstructionCount": 1, "Comment": [ diff --git a/unittests/InstructionCountCI/VEX_map2_DOTPROD.json b/unittests/InstructionCountCI/VEX_map2_DOTPROD.json new file mode 100644 index 000000000..a5a279e59 --- /dev/null +++ b/unittests/InstructionCountCI/VEX_map2_DOTPROD.json @@ -0,0 +1,108 @@ +{ + "Features": { + "Bitness": 64, + "EnabledHostFeatures": [ + "SVE256", + "SVE128", + "DOTPROD" + ], + "DisabledHostFeatures": [ + "AFP", + "FLAGM", + "FLAGM2", + "SVEBITPERM", + "I8MM" + ], + "BinaryCacheVersion": 20 + }, + "Instructions": { + "vpdpbusd xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 8, + "Comment": [ + "Map 2 0b01 0x50 128-bit", + "vpdpbusd xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x50, 0xc2" + ], + "ExpectedArm64ASM": [ + "movi v2.16b, #0x80", + "eor v2.16b, v17.16b, v2.16b", + "movi v3.16b, #0x40", + "mov v4.16b, v16.16b", + "sdot v4.4s, v18.16b, v3.16b", + "sdot v4.4s, v18.16b, v3.16b", + "mov v16.16b, v4.16b", + "sdot v16.4s, v2.16b, v18.16b" + ] + }, + "vpdpbusd ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 8, + "Comment": [ + "Map 2 0b01 0x50 256-bit", + "vpdpbusd ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x50, 0xc2" + ], + "ExpectedArm64ASM": [ + "mov z2.b, #-128", + "eor z2.d, z17.d, z2.d", + "mov z3.b, #64", + "mov z4.d, z16.d", + "sdot z4.s, z18.b, z3.b", + "sdot z4.s, z18.b, z3.b", + "mov z16.d, z4.d", + "sdot z16.s, z2.b, z18.b" + ] + }, + "vpdpbusds xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 8, + "Comment": [ + "Map 2 0b01 0x51 128-bit", + "vpdpbusds xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x51, 0xc2" + ], + "ExpectedArm64ASM": [ + "movi v2.16b, #0x80", + "eor v2.16b, v17.16b, v2.16b", + "movi v3.16b, #0x40", + "movi v4.2d, #0x0", + "sdot v4.4s, v18.16b, v3.16b", + "sdot v4.4s, v18.16b, v3.16b", + "sdot v4.4s, v2.16b, v18.16b", + "sqadd v16.4s, v16.4s, v4.4s" + ] + }, + "vpdpbusds ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 8, + "Comment": [ + "Map 2 0b01 0x51 256-bit", + "vpdpbusds ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x51, 0xc2" + ], + "ExpectedArm64ASM": [ + "mov z2.b, #-128", + "eor z2.d, z17.d, z2.d", + "mov z3.b, #64", + "movi v4.2d, #0x0", + "sdot z4.s, z18.b, z3.b", + "sdot z4.s, z18.b, z3.b", + "sdot z4.s, z2.b, z18.b", + "sqadd z16.s, z16.s, z4.s" + ] + } + } +} diff --git a/unittests/InstructionCountCI/VEX_map2_I8MM.json b/unittests/InstructionCountCI/VEX_map2_I8MM.json new file mode 100644 index 000000000..0b585a3f1 --- /dev/null +++ b/unittests/InstructionCountCI/VEX_map2_I8MM.json @@ -0,0 +1,171 @@ +{ + "Features": { + "Bitness": 64, + "EnabledHostFeatures": [ + "SVE256", + "SVE128", + "I8MM" + ], + "DisabledHostFeatures": [ + "AFP", + "FLAGM", + "FLAGM2", + "SVEBITPERM" + ], + "BinaryCacheVersion": 20 + }, + "Instructions": { + "vpdpbusd xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 1, + "Comment": [ + "Map 2 0b01 0x50 128-bit", + "vpdpbusd xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x50, 0xc2" + ], + "ExpectedArm64ASM": [ + "usdot v16.4s, v17.16b, v18.16b" + ] + }, + "vpdpbusd ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 1, + "Comment": [ + "Map 2 0b01 0x50 256-bit", + "vpdpbusd ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x50, 0xc2" + ], + "ExpectedArm64ASM": [ + "usdot z16.s, z17.b, z18.b" + ] + }, + "vpdpbusds xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 3, + "Comment": [ + "Map 2 0b01 0x51 128-bit", + "vpdpbusds xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x51, 0xc2" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "usdot v2.4s, v17.16b, v18.16b", + "sqadd v16.4s, v16.4s, v2.4s" + ] + }, + "vpdpbusds ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 3, + "Comment": [ + "Map 2 0b01 0x51 256-bit", + "vpdpbusds ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x51, 0xc2" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "usdot z2.s, z17.b, z18.b", + "sqadd z16.s, z16.s, z2.s" + ] + }, + "vpdpwssd xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 4, + "Comment": [ + "Map 2 0b01 0x52 128-bit", + "vpdpwssd xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x52, 0xc2" + ], + "ExpectedArm64ASM": [ + "smull v2.4s, v17.4h, v18.4h", + "smull2 v3.4s, v17.8h, v18.8h", + "addp v2.4s, v2.4s, v3.4s", + "add v16.4s, v16.4s, v2.4s" + ] + }, + "vpdpwssd ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 11, + "Comment": [ + "Map 2 0b01 0x52 256-bit", + "vpdpwssd ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x52, 0xc2" + ], + "ExpectedArm64ASM": [ + "smullb z0.s, z17.h, z18.h", + "smullt z1.s, z17.h, z18.h", + "zip1 z2.s, z0.s, z1.s", + "smullb z0.s, z17.h, z18.h", + "smullt z1.s, z17.h, z18.h", + "zip2 z3.s, z0.s, z1.s", + "addp z2.s, p7/m, z2.s, z3.s", + "uzp2 z1.s, z2.s, z2.s", + "uzp1 z2.s, z2.s, z2.s", + "splice z2.d, p6, z2.d, z1.d", + "add z16.s, z16.s, z2.s" + ] + }, + "vpdpwssds xmm0, xmm1, xmm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 5, + "Comment": [ + "Map 2 0b01 0x53 128-bit", + "vpdpwssds xmm0, xmm1, xmm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x71, 0x53, 0xc2" + ], + "ExpectedArm64ASM": [ + "smull v2.4s, v17.4h, v18.4h", + "smull2 v3.4s, v17.8h, v18.8h", + "addp v2.4s, v2.4s, v3.4s", + "neg v2.4s, v2.4s", + "sqsub v16.4s, v16.4s, v2.4s" + ] + }, + "vpdpwssds ymm0, ymm1, ymm2": { + "x86InstructionCount": 1, + "ExpectedInstructionCount": 12, + "Comment": [ + "Map 2 0b01 0x53 256-bit", + "vpdpwssds ymm0, ymm1, ymm2", + "NASM only emits the EVEX encoding, hand encode the VEX form" + ], + "x86Insts": [ + "db 0xc4, 0xe2, 0x75, 0x53, 0xc2" + ], + "ExpectedArm64ASM": [ + "smullb z0.s, z17.h, z18.h", + "smullt z1.s, z17.h, z18.h", + "zip1 z2.s, z0.s, z1.s", + "smullb z0.s, z17.h, z18.h", + "smullt z1.s, z17.h, z18.h", + "zip2 z3.s, z0.s, z1.s", + "addp z2.s, p7/m, z2.s, z3.s", + "uzp2 z1.s, z2.s, z2.s", + "uzp1 z2.s, z2.s, z2.s", + "splice z2.d, p6, z2.d, z1.d", + "neg z2.s, p7/m, z2.s", + "sqsub z16.s, z16.s, z2.s" + ] + } + } +}