From b3d76bd2f1981390c7136c2407e10131eae55ae2 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Thu, 19 Oct 2023 16:36:19 +0200 Subject: [PATCH 1/6] IR: Adds DPPS and DPPD source masks This will get used for these instructions soon --- FEXCore/Source/Interface/Core/CPUBackend.cpp | 50 ++++++++++++++++++++ FEXCore/include/FEXCore/IR/IR.h | 2 + 2 files changed, 52 insertions(+) diff --git a/FEXCore/Source/Interface/Core/CPUBackend.cpp b/FEXCore/Source/Interface/Core/CPUBackend.cpp index dec22b54f..cfc5eed47 100644 --- a/FEXCore/Source/Interface/Core/CPUBackend.cpp +++ b/FEXCore/Source/Interface/Core/CPUBackend.cpp @@ -176,6 +176,54 @@ constexpr static auto SHUFPS_LUT { }() }; +constexpr static auto DPPS_MASK { +[]() consteval { + struct LUTType { + uint32_t Val[4]; + }; + + std::array TotalLUT{}; + for (size_t i = 0; i < TotalLUT.size(); ++i) { + auto &LUT = TotalLUT[i]; + constexpr auto GetLUT = [](size_t i, size_t Index) { + if (i & (1U << Index)) { + return -1U; + } + return 0U; + }; + + LUT.Val[0] = GetLUT(i, 0); + LUT.Val[1] = GetLUT(i, 1); + LUT.Val[2] = GetLUT(i, 2); + LUT.Val[3] = GetLUT(i, 3); + } + return TotalLUT; +}() +}; + +constexpr static auto DPPD_MASK { +[]() consteval { + struct LUTType { + uint64_t Val[2]; + }; + + std::array TotalLUT{}; + for (size_t i = 0; i < TotalLUT.size(); ++i) { + auto &LUT = TotalLUT[i]; + constexpr auto GetLUT = [](size_t i, size_t Index) { + if (i & (1U << Index)) { + return -1ULL; + } + return 0ULL; + }; + + LUT.Val[0] = GetLUT(i, 0); + LUT.Val[1] = GetLUT(i, 1); + } + return TotalLUT; +}() +}; + CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize) : ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) { @@ -194,6 +242,8 @@ CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t I Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW] = reinterpret_cast(PSHUFHW_LUT.data()); Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFD] = reinterpret_cast(PSHUFD_LUT.data()); Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_SHUFPS] = reinterpret_cast(SHUFPS_LUT.data()); + Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK] = reinterpret_cast(DPPS_MASK.data()); + Common.IndexedNamedVectorConstantPointers[FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK] = reinterpret_cast(DPPD_MASK.data()); #ifndef FEX_DISABLE_TELEMETRY // Fill in telemetry values diff --git a/FEXCore/include/FEXCore/IR/IR.h b/FEXCore/include/FEXCore/IR/IR.h index 5c130e31c..9596abf01 100644 --- a/FEXCore/include/FEXCore/IR/IR.h +++ b/FEXCore/include/FEXCore/IR/IR.h @@ -541,6 +541,8 @@ enum IndexNamedVectorConstant : uint8_t { INDEXED_NAMED_VECTOR_PSHUFHW, INDEXED_NAMED_VECTOR_PSHUFD, INDEXED_NAMED_VECTOR_SHUFPS, + INDEXED_NAMED_VECTOR_DPPS_MASK, + INDEXED_NAMED_VECTOR_DPPD_MASK, INDEXED_NAMED_VECTOR_MAX, }; From 2c0bc0654d92850362285c3e3b87ea802e22e07f Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Thu, 19 Oct 2023 16:38:11 +0200 Subject: [PATCH 2/6] IR: Adds new VFAddV operation SVE added this instruction natively, we can take advantage of it on SVE-128bit systems which is quite nice. Will be used soon. --- .../Interface/Core/JIT/Arm64/VectorOps.cpp | 35 +++++++++++++++++++ FEXCore/Source/Interface/IR/IR.json | 7 ++++ 2 files changed, 42 insertions(+) diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp b/FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp index 7e52eae52..0456b6be4 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp @@ -1268,6 +1268,41 @@ DEF_OP(VAddP) { } } +DEF_OP(VFAddV) { + const auto Op = IROp->C(); + const auto OpSize = IROp->Size; + const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; + + const auto ElementSize = Op->Header.ElementSize; + + const auto Dst = GetVReg(Node); + const auto Vector = GetVReg(Op->Vector.ID()); + + LOGMAN_THROW_AA_FMT(OpSize == Core::CPUState::XMM_SSE_REG_SIZE || OpSize == Core::CPUState::XMM_AVX_REG_SIZE, "Only AVX and SSE size supported"); + LOGMAN_THROW_AA_FMT(ElementSize == 2 || ElementSize == 4 || ElementSize == 8, "Invalid size"); + const auto SubRegSize = ARMEmitter::ToVectorSizePair( + ElementSize == 2 ? ARMEmitter::SubRegSize::i16Bit : + ElementSize == 4 ? ARMEmitter::SubRegSize::i32Bit : ARMEmitter::SubRegSize::i64Bit); + + if (HostSupportsSVE256 && Is256Bit) { + const auto Pred = PRED_TMP_32B.Merging(); + faddv(SubRegSize.Vector, Dst, Pred, Vector.Z()); + } + if (HostSupportsSVE128) { + const auto Pred = PRED_TMP_16B.Merging(); + faddv(SubRegSize.Vector, Dst, Pred, Vector.Z()); + } else { + // ASIMD doesn't support faddv, need to use multiple faddp to match behaviour. + if (ElementSize == 4) { + faddp(SubRegSize.Vector, Dst.Q(), Vector.Q(), Vector.Q()); + faddp(SubRegSize.Scalar, Dst, Vector); + } + else { + faddp(SubRegSize.Scalar, Dst, Vector); + } + } +} + DEF_OP(VAddV) { const auto Op = IROp->C(); const auto OpSize = IROp->Size; diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index 0e2ff7a44..a72e336ba 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -1763,6 +1763,13 @@ "DestSize": "RegisterSize", "NumElements": "RegisterSize / ElementSize" }, + "FPR = VFAddV u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": { + "Desc": ["Does a horizontal float vector add of elements across the source vector", + "Result is a zero extended scalar" + ], + "DestSize": "RegisterSize", + "NumElements": "RegisterSize / ElementSize" + }, "FPR = VFSub u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": { "DestSize": "RegisterSize", "NumElements": "RegisterSize / ElementSize" From 887200e5716e409d31fc9cca6c19b3d66c608218 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Thu, 19 Oct 2023 16:39:14 +0200 Subject: [PATCH 3/6] OpcodeDispatcher: Optimize 128-bit DPPS and DPPD These instructions aren't super amazing due to the fact that they have both a source mask and a destination duplication mask. Setup a case where we can generate more optimal code in /most/ cases. There are a few that still fall down a "bad" path for the result broadcast but in most cases they are optimal. Still to be seen what games typically use the broadcast mask as. AVX in its infinite wisdom expanded DPPS to 256-bit, while leaving DPPD to only support 128-bit still. This leaves the original implementation alone for 256-bit DPPS since I don't want to break it. This is another instruction that gets a free optimization when SVE-128bit is supported! --- .../Source/Interface/Core/OpcodeDispatcher.h | 4 + .../Core/OpcodeDispatcher/Vector.cpp | 221 +++++++++++++++++- 2 files changed, 213 insertions(+), 12 deletions(-) diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h index 6b6c1e4e5..c32ece0aa 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h @@ -975,6 +975,10 @@ private: const X86Tables::DecodedOperand& Src2, const X86Tables::DecodedOperand& Imm, size_t ElementSize); + OrderedNode* VDPPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, + const X86Tables::DecodedOperand& Src2, + const X86Tables::DecodedOperand& Imm); + OrderedNode* ExtendVectorElementsImpl(OpcodeArgs, size_t ElementSize, size_t DstElementSize, bool Signed); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp index c24ebaf0e..fc623dd9b 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp @@ -4668,6 +4668,205 @@ OrderedNode* OpDispatchBuilder::DPPOpImpl(OpcodeArgs, const X86Tables::DecodedOp const X86Tables::DecodedOperand& Imm, size_t ElementSize) { LOGMAN_THROW_A_FMT(Imm.IsLiteral(), "Imm needs to be literal here"); const uint8_t Mask = Imm.Data.Literal.Value; + const auto SizeMask = [ElementSize]() { + if (ElementSize == 4) { + return 0b1111; + } + return 0b11; + }(); + + const uint8_t SrcMask = (Mask >> 4) & SizeMask; + const uint8_t DstMask = Mask & SizeMask; + + const auto NamedIndexMask = [ElementSize]() { + if (ElementSize == 4) { + return FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPS_MASK; + } + + return FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_DPPD_MASK; + }(); + const auto DstSize = GetDstSize(Op); + + OrderedNode *ZeroVec = LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO); + if (SrcMask == 0 || DstMask == 0) { + // What are you even doing here? Go away. + return ZeroVec; + } + + OrderedNode *Src1V = LoadSource(FPRClass, Op, Src1, Op->Flags); + OrderedNode *Src2V = LoadSource(FPRClass, Op, Src2, Op->Flags); + + // First step is to do an FMUL + OrderedNode *Temp = _VFMul(DstSize, ElementSize, Src1V, Src2V); + + // Now mask results based on IndexMask. + if (SrcMask != SizeMask) { + auto InputMask = LoadAndCacheIndexedNamedVectorConstant(DstSize, NamedIndexMask, SrcMask * 16); + Temp = _VAnd(DstSize, ElementSize, Temp, InputMask); + } + + // Now due a float reduction + Temp = _VFAddV(DstSize, ElementSize, Temp); + + // Now using the destination mask we choose where the result ends up + // It can duplicate and zero results + if (ElementSize == 8) { + switch (DstMask) { + case 0b01: + // Dest[63:0] = Result + // Dest[127:64] = Zero + return _VZip(DstSize, ElementSize, Temp, ZeroVec); + case 0b10: + // Dest[63:0] = Zero + // Dest[127:64] = Result + return _VZip(DstSize, ElementSize, ZeroVec, Temp); + case 0b11: + // Broadcast + // Dest[63:0] = Result + // Dest[127:64] = Result + return _VDupElement(DstSize, ElementSize, Temp, 0); + case 0: + default: + LOGMAN_MSG_A_FMT("Unsupported"); + } + } + else { + auto BadPath = [&]() { + OrderedNode *Result = ZeroVec; + + for (size_t i = 0; i < (DstSize / ElementSize); ++i) { + const auto Bit = 1U << (i % 4); + + if ((DstMask & Bit) != 0) { + Result = _VInsElement(DstSize, ElementSize, i, 0, Result, Temp); + } + } + + return Result; + }; + switch (DstMask) { + case 0b0001: + // Dest[31:0] = Result + // Dest[63:32] = Zero + // Dest[95:64] = Zero + // Dest[127:96] = Zero + return _VZip(DstSize, ElementSize, Temp, ZeroVec); + case 0b0010: + // Dest[31:0] = Zero + // Dest[63:32] = Result + // Dest[95:64] = Zero + // Dest[127:96] = Zero + return _VZip(DstSize / 2, ElementSize, ZeroVec, Temp); + case 0b0011: + // Dest[31:0] = Result + // Dest[63:32] = Result + // Dest[95:64] = Zero + // Dest[127:96] = Zero + return _VDupElement(DstSize / 2, ElementSize, Temp, 0); + case 0b0100: + // Dest[31:0] = Zero + // Dest[63:32] = Zero + // Dest[95:64] = Result + // Dest[127:96] = Zero + return _VZip(DstSize, 8, ZeroVec, Temp); + case 0b0101: + // Dest[31:0] = Result + // Dest[63:32] = Zero + // Dest[95:64] = Result + // Dest[127:96] = Zero + return _VZip(DstSize, 8, Temp, Temp); + case 0b0110: + // Dest[31:0] = Zero + // Dest[63:32] = Result + // Dest[95:64] = Result + // Dest[127:96] = Zero + return BadPath(); + case 0b0111: + // Dest[31:0] = Result + // Dest[63:32] = Result + // Dest[95:64] = Result + // Dest[127:96] = Zero + Temp = _VDupElement(DstSize, ElementSize, Temp, 0); + return _VInsElement(DstSize, ElementSize, 3, 0, Temp, ZeroVec); + case 0b1000: + // Dest[31:0] = Zero + // Dest[63:32] = Zero + // Dest[95:64] = Zero + // Dest[127:96] = Result + return _VExtr(DstSize, 1, Temp, ZeroVec, 4); + case 0b1001: + // Dest[31:0] = Result + // Dest[63:32] = Zero + // Dest[95:64] = Zero + // Dest[127:96] = Result + return BadPath(); + case 0b1010: + // Dest[31:0] = Zero + // Dest[63:32] = Result + // Dest[95:64] = Zero + // Dest[127:96] = Result + Temp = _VDupElement(DstSize, ElementSize, Temp, 0); + return _VZip(DstSize, 4, ZeroVec, Temp); + case 0b1011: + // Dest[31:0] = Result + // Dest[63:32] = Result + // Dest[95:64] = Zero + // Dest[127:96] = Result + Temp = _VDupElement(DstSize, ElementSize, Temp, 0); + return _VInsElement(DstSize, ElementSize, 2, 0, Temp, ZeroVec); + case 0b1100: + // Dest[31:0] = Zero + // Dest[63:32] = Zero + // Dest[95:64] = Result + // Dest[127:96] = Result + Temp = _VDupElement(DstSize, ElementSize, Temp, 0); + return _VZip(DstSize, 8, ZeroVec, Temp); + case 0b1101: + // Dest[31:0] = Result + // Dest[63:32] = Zero + // Dest[95:64] = Result + // Dest[127:96] = Result + Temp = _VDupElement(DstSize, ElementSize, Temp, 0); + return _VInsElement(DstSize, ElementSize, 1, 0, Temp, ZeroVec); + case 0b1110: + // Dest[31:0] = Zero + // Dest[63:32] = Result + // Dest[95:64] = Result + // Dest[127:96] = Result + Temp = _VDupElement(DstSize, ElementSize, Temp, 0); + return _VInsElement(DstSize, ElementSize, 0, 0, Temp, ZeroVec); + case 0b1111: + // Broadcast + // Dest[31:0] = Result + // Dest[63:32] = Zero + // Dest[95:64] = Zero + // Dest[127:96] = Zero + return _VDupElement(DstSize, ElementSize, Temp, 0); + case 0: + default: + LOGMAN_MSG_A_FMT("Unsupported"); + } + } + FEX_UNREACHABLE; +} + +template +void OpDispatchBuilder::DPPOp(OpcodeArgs) { + OrderedNode *Result = DPPOpImpl(Op, Op->Dest, Op->Src[0], Op->Src[1], ElementSize); + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::DPPOp<4>(OpcodeArgs); +template +void OpDispatchBuilder::DPPOp<8>(OpcodeArgs); + +OrderedNode* OpDispatchBuilder::VDPPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, + const X86Tables::DecodedOperand& Src2, + const X86Tables::DecodedOperand& Imm) { + LOGMAN_THROW_A_FMT(Imm.IsLiteral(), "Imm needs to be literal here"); + constexpr size_t ElementSize = 4; + const uint8_t Mask = Imm.Data.Literal.Value; const uint8_t SrcMask = Mask >> 4; const uint8_t DstMask = Mask & 0xF; @@ -4714,20 +4913,18 @@ OrderedNode* OpDispatchBuilder::DPPOpImpl(OpcodeArgs, const X86Tables::DecodedOp return Result; } -template -void OpDispatchBuilder::DPPOp(OpcodeArgs) { - OrderedNode *Result = DPPOpImpl(Op, Op->Dest, Op->Src[0], Op->Src[1], ElementSize); - StoreResult(FPRClass, Op, Result, -1); -} - -template -void OpDispatchBuilder::DPPOp<4>(OpcodeArgs); -template -void OpDispatchBuilder::DPPOp<8>(OpcodeArgs); - template void OpDispatchBuilder::VDPPOp(OpcodeArgs) { - OrderedNode *Result = DPPOpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2], ElementSize); + const auto DstSize = GetDstSize(Op); + + OrderedNode *Result{}; + if (ElementSize == 4 && DstSize == Core::CPUState::XMM_AVX_REG_SIZE) { + // 256-bit DPPS isn't handled by the 128-bit solution. + Result = VDPPSOpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2]); + } + else { + Result = DPPOpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2], ElementSize); + } // We don't need to emit a _VMov to clear the upper lane, since DPPOpImpl uses a zero vector // to construct the results, so the upper lane will always be cleared for the 128-bit version. From 165d3d3d4df2b40e860d8e9ee65058c299e17850 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Fri, 20 Oct 2023 18:13:01 +0200 Subject: [PATCH 4/6] Arm64JIT: Fixes VDupElement so it respects 64-bit vector duping In some cases when we want the upper bits to be zero, this is the desired behaviour --- FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp | 8 +++++++- 1 file changed, 7 insertions(+), 1 deletion(-) diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp b/FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp index 0456b6be4..1adaba5cb 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp @@ -3777,6 +3777,7 @@ DEF_OP(VDupElement) { const auto Index = Op->Index; const auto ElementSize = Op->Header.ElementSize; const auto Is256Bit = OpSize == Core::CPUState::XMM_AVX_REG_SIZE; + const auto Is128Bit = OpSize == Core::CPUState::XMM_SSE_REG_SIZE; const auto Dst = GetVReg(Node); const auto Vector = GetVReg(Op->Vector.ID()); @@ -3791,7 +3792,12 @@ DEF_OP(VDupElement) { if (HostSupportsSVE256 && Is256Bit) { dup(SubRegSize, Dst.Z(), Vector.Z(), Index); } else { - dup(SubRegSize, Dst.Q(), Vector.Q(), Index); + if (Is128Bit) { + dup(SubRegSize, Dst.Q(), Vector.Q(), Index); + } + else { + dup(SubRegSize, Dst.D(), Vector.D(), Index); + } } } From 14e80ce2285c830732374ed032271b041fb69e90 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Thu, 19 Oct 2023 16:41:55 +0200 Subject: [PATCH 5/6] InstCountCI: Update for DPPS/DPPD Adds some new destination broadcast masks to ensure we handle most of them. --- unittests/InstructionCountCI/DDD.json | 6 +- unittests/InstructionCountCI/H0F38.json | 2 +- unittests/InstructionCountCI/H0F3A.json | 263 +++++++++++++--- .../InstructionCountCI/H0F3A_SVE128.json | 292 ++++++++++++++++++ .../InstructionCountCI/PrimaryGroup.json | 8 +- unittests/InstructionCountCI/RPRES/DDD.json | 4 +- unittests/InstructionCountCI/Secondary.json | 12 +- .../InstructionCountCI/Secondary_OpSize.json | 2 +- .../InstructionCountCI/Secondary_REPNE.json | 2 +- unittests/InstructionCountCI/VEX_map1.json | 4 +- unittests/InstructionCountCI/VEX_map2.json | 2 +- unittests/InstructionCountCI/VEX_map3.json | 57 +--- 12 files changed, 550 insertions(+), 104 deletions(-) create mode 100644 unittests/InstructionCountCI/H0F3A_SVE128.json diff --git a/unittests/InstructionCountCI/DDD.json b/unittests/InstructionCountCI/DDD.json index acbded228..0b9b0a351 100644 --- a/unittests/InstructionCountCI/DDD.json +++ b/unittests/InstructionCountCI/DDD.json @@ -123,7 +123,7 @@ "ExpectedArm64ASM": [ "ldr d2, [x28, #752]", "ldr d3, [x28, #768]", - "dup v4.4s, v2.s[1]", + "dup v4.2s, v2.s[1]", "fsub s2, s2, s4", "faddp v3.4s, v3.4s, v3.4s", "mov v2.s[1], v3.s[0]", @@ -163,7 +163,7 @@ "ldr d2, [x28, #768]", "fmov s0, #0x70 (1.0000)", "fdiv s2, s0, s2", - "dup v2.4s, v2.s[0]", + "dup v2.2s, v2.s[0]", "str d2, [x28, #752]" ] }, @@ -178,7 +178,7 @@ "fmov s0, #0x70 (1.0000)", "fsqrt s1, s2", "fdiv s2, s0, s1", - "dup v2.4s, v2.s[0]", + "dup v2.2s, v2.s[0]", "str d2, [x28, #752]" ] }, diff --git a/unittests/InstructionCountCI/H0F38.json b/unittests/InstructionCountCI/H0F38.json index ebf8529d9..72eeb69cb 100644 --- a/unittests/InstructionCountCI/H0F38.json +++ b/unittests/InstructionCountCI/H0F38.json @@ -680,7 +680,7 @@ "0x66 0x0f 0x38 0x41" ], "ExpectedArm64ASM": [ - "ldr q2, [x28, #1888]", + "ldr q2, [x28, #1904]", "zip1 v3.8h, v2.8h, v17.8h", "zip2 v2.8h, v2.8h, v17.8h", "umin v2.4s, v3.4s, v2.4s", diff --git a/unittests/InstructionCountCI/H0F3A.json b/unittests/InstructionCountCI/H0F3A.json index 103ca28a6..fcba2ec6d 100644 --- a/unittests/InstructionCountCI/H0F3A.json +++ b/unittests/InstructionCountCI/H0F3A.json @@ -951,25 +951,13 @@ ] }, "dpps xmm0, xmm1, 00001111b": { - "ExpectedInstructionCount": 13, - "Optimal": "No", + "ExpectedInstructionCount": 1, + "Optimal": "Yes", "Comment": [ "0x66 0x0f 0x3a 0x40" ], "ExpectedArm64ASM": [ - "movi v2.2d, #0x0", - "fmul v3.4s, v16.4s, v17.4s", - "mov v3.s[0], v2.s[0]", - "mov v3.s[1], v2.s[0]", - "mov v3.s[2], v2.s[0]", - "mov v3.s[3], v2.s[0]", - "faddp v3.4s, v3.4s, v2.4s", - "faddp v3.4s, v3.4s, v2.4s", - "mov v2.s[0], v3.s[0]", - "mov v2.s[1], v3.s[0]", - "mov v2.s[2], v3.s[0]", - "mov v16.16b, v2.16b", - "mov v16.s[3], v3.s[0]" + "movi v16.2d, #0x0" ] }, "dpps xmm0, xmm1, 11110000b": { @@ -982,8 +970,76 @@ "movi v16.2d, #0x0" ] }, - "dpps xmm0, xmm1, 11111111b": { - "ExpectedInstructionCount": 9, + "dpps xmm0, xmm1, 11110001b": { + "ExpectedInstructionCount": 5, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddp v3.4s, v3.4s, v3.4s", + "faddp s3, v3.2s", + "zip1 v16.4s, v3.4s, v2.4s" + ] + }, + "dpps xmm0, xmm1, 11110010b": { + "ExpectedInstructionCount": 5, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddp v3.4s, v3.4s, v3.4s", + "faddp s3, v3.2s", + "zip1 v16.2s, v2.2s, v3.2s" + ] + }, + "dpps xmm0, xmm1, 11110011b": { + "ExpectedInstructionCount": 4, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "fmul v2.4s, v16.4s, v17.4s", + "faddp v2.4s, v2.4s, v2.4s", + "faddp s2, v2.2s", + "dup v16.2s, v2.s[0]" + ] + }, + "dpps xmm0, xmm1, 11110100b": { + "ExpectedInstructionCount": 5, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddp v3.4s, v3.4s, v3.4s", + "faddp s3, v3.2s", + "zip1 v16.2d, v2.2d, v3.2d" + ] + }, + "dpps xmm0, xmm1, 11110101b": { + "ExpectedInstructionCount": 4, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "fmul v2.4s, v16.4s, v17.4s", + "faddp v2.4s, v2.4s, v2.4s", + "faddp s2, v2.2s", + "zip1 v16.2d, v2.2d, v2.2d" + ] + }, + "dpps xmm0, xmm1, 11110110b": { + "ExpectedInstructionCount": 7, "Optimal": "No", "Comment": [ "0x66 0x0f 0x3a 0x40" @@ -991,15 +1047,150 @@ "ExpectedArm64ASM": [ "movi v2.2d, #0x0", "fmul v3.4s, v16.4s, v17.4s", - "faddp v3.4s, v3.4s, v2.4s", - "faddp v3.4s, v3.4s, v2.4s", - "mov v2.s[0], v3.s[0]", + "faddp v3.4s, v3.4s, v3.4s", + "faddp s3, v3.2s", "mov v2.s[1], v3.s[0]", - "mov v2.s[2], v3.s[0]", + "mov v16.16b, v2.16b", + "mov v16.s[2], v3.s[0]" + ] + }, + "dpps xmm0, xmm1, 11110111b": { + "ExpectedInstructionCount": 7, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddp v3.4s, v3.4s, v3.4s", + "faddp s3, v3.2s", + "dup v3.4s, v3.s[0]", + "mov v16.16b, v3.16b", + "mov v16.s[3], v2.s[0]" + ] + }, + "dpps xmm0, xmm1, 11111000b": { + "ExpectedInstructionCount": 5, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddp v3.4s, v3.4s, v3.4s", + "faddp s3, v3.2s", + "ext v16.16b, v2.16b, v3.16b, #4" + ] + }, + "dpps xmm0, xmm1, 11111001b": { + "ExpectedInstructionCount": 7, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddp v3.4s, v3.4s, v3.4s", + "faddp s3, v3.2s", + "mov v2.s[0], v3.s[0]", "mov v16.16b, v2.16b", "mov v16.s[3], v3.s[0]" ] }, + "dpps xmm0, xmm1, 11111010b": { + "ExpectedInstructionCount": 6, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddp v3.4s, v3.4s, v3.4s", + "faddp s3, v3.2s", + "dup v3.4s, v3.s[0]", + "zip1 v16.4s, v2.4s, v3.4s" + ] + }, + "dpps xmm0, xmm1, 11111011b": { + "ExpectedInstructionCount": 7, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddp v3.4s, v3.4s, v3.4s", + "faddp s3, v3.2s", + "dup v3.4s, v3.s[0]", + "mov v16.16b, v3.16b", + "mov v16.s[2], v2.s[0]" + ] + }, + "dpps xmm0, xmm1, 11111100b": { + "ExpectedInstructionCount": 6, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddp v3.4s, v3.4s, v3.4s", + "faddp s3, v3.2s", + "dup v3.4s, v3.s[0]", + "zip1 v16.2d, v2.2d, v3.2d" + ] + }, + "dpps xmm0, xmm1, 11111101b": { + "ExpectedInstructionCount": 7, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddp v3.4s, v3.4s, v3.4s", + "faddp s3, v3.2s", + "dup v3.4s, v3.s[0]", + "mov v16.16b, v3.16b", + "mov v16.s[1], v2.s[0]" + ] + }, + "dpps xmm0, xmm1, 11111110b": { + "ExpectedInstructionCount": 7, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddp v3.4s, v3.4s, v3.4s", + "faddp s3, v3.2s", + "dup v3.4s, v3.s[0]", + "mov v16.16b, v3.16b", + "mov v16.s[0], v2.s[0]" + ] + }, + "dpps xmm0, xmm1, 11111111b": { + "ExpectedInstructionCount": 4, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "fmul v2.4s, v16.4s, v17.4s", + "faddp v2.4s, v2.4s, v2.4s", + "faddp s2, v2.2s", + "dup v16.4s, v2.s[0]" + ] + }, "dppd xmm0, xmm1, 00000000b": { "ExpectedInstructionCount": 1, "Optimal": "Yes", @@ -1011,20 +1202,13 @@ ] }, "dppd xmm0, xmm1, 00001111b": { - "ExpectedInstructionCount": 8, - "Optimal": "No", + "ExpectedInstructionCount": 1, + "Optimal": "Yes", "Comment": [ "0x66 0x0f 0x3a 0x41" ], "ExpectedArm64ASM": [ - "movi v2.2d, #0x0", - "fmul v3.2d, v16.2d, v17.2d", - "mov v3.d[0], v2.d[0]", - "mov v3.d[1], v2.d[0]", - "faddp v3.2d, v3.2d, v2.2d", - "mov v2.d[0], v3.d[0]", - "mov v16.16b, v2.16b", - "mov v16.d[1], v3.d[0]" + "movi v16.2d, #0x0" ] }, "dppd xmm0, xmm1, 11110000b": { @@ -1038,18 +1222,15 @@ ] }, "dppd xmm0, xmm1, 11111111b": { - "ExpectedInstructionCount": 6, - "Optimal": "No", + "ExpectedInstructionCount": 3, + "Optimal": "Yes", "Comment": [ "0x66 0x0f 0x3a 0x41" ], "ExpectedArm64ASM": [ - "movi v2.2d, #0x0", - "fmul v3.2d, v16.2d, v17.2d", - "faddp v3.2d, v3.2d, v2.2d", - "mov v2.d[0], v3.d[0]", - "mov v16.16b, v2.16b", - "mov v16.d[1], v3.d[0]" + "fmul v2.2d, v16.2d, v17.2d", + "faddp d2, v2.2d", + "dup v16.2d, v2.d[0]" ] }, "mpsadbw xmm0, xmm1, 000b": { @@ -1557,7 +1738,7 @@ "0x66 0x0f 0x3a 0xdf" ], "ExpectedArm64ASM": [ - "ldr q2, [x28, #2000]", + "ldr q2, [x28, #2016]", "movi v3.2d, #0x0", "mov v16.16b, v17.16b", "unimplemented (Unimplemented)", @@ -1571,7 +1752,7 @@ "0x66 0x0f 0x3a 0xdf" ], "ExpectedArm64ASM": [ - "ldr q2, [x28, #2000]", + "ldr q2, [x28, #2016]", "movi v3.2d, #0x0", "mov v16.16b, v17.16b", "unimplemented (Unimplemented)", diff --git a/unittests/InstructionCountCI/H0F3A_SVE128.json b/unittests/InstructionCountCI/H0F3A_SVE128.json new file mode 100644 index 000000000..f96ed407c --- /dev/null +++ b/unittests/InstructionCountCI/H0F3A_SVE128.json @@ -0,0 +1,292 @@ +{ + "Features": { + "Bitness": 64, + "EnabledHostFeatures": [ + "SVE128" + ], + "DisabledHostFeatures": [ + "SVE256", + "AFP" + ] + }, + "Instructions": { + "dpps xmm0, xmm1, 00000000b": { + "ExpectedInstructionCount": 1, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v16.2d, #0x0" + ] + }, + "dpps xmm0, xmm1, 00001111b": { + "ExpectedInstructionCount": 1, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v16.2d, #0x0" + ] + }, + "dpps xmm0, xmm1, 11110000b": { + "ExpectedInstructionCount": 1, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v16.2d, #0x0" + ] + }, + "dpps xmm0, xmm1, 11110001b": { + "ExpectedInstructionCount": 4, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddv s3, p6, z3.s", + "zip1 v16.4s, v3.4s, v2.4s" + ] + }, + "dpps xmm0, xmm1, 11110010b": { + "ExpectedInstructionCount": 4, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddv s3, p6, z3.s", + "zip1 v16.2s, v2.2s, v3.2s" + ] + }, + "dpps xmm0, xmm1, 11110011b": { + "ExpectedInstructionCount": 3, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "fmul v2.4s, v16.4s, v17.4s", + "faddv s2, p6, z2.s", + "dup v16.2s, v2.s[0]" + ] + }, + "dpps xmm0, xmm1, 11110100b": { + "ExpectedInstructionCount": 4, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddv s3, p6, z3.s", + "zip1 v16.2d, v2.2d, v3.2d" + ] + }, + "dpps xmm0, xmm1, 11110101b": { + "ExpectedInstructionCount": 3, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "fmul v2.4s, v16.4s, v17.4s", + "faddv s2, p6, z2.s", + "zip1 v16.2d, v2.2d, v2.2d" + ] + }, + "dpps xmm0, xmm1, 11110110b": { + "ExpectedInstructionCount": 6, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddv s3, p6, z3.s", + "mov v2.s[1], v3.s[0]", + "mov v16.16b, v2.16b", + "mov v16.s[2], v3.s[0]" + ] + }, + "dpps xmm0, xmm1, 11110111b": { + "ExpectedInstructionCount": 6, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddv s3, p6, z3.s", + "dup v3.4s, v3.s[0]", + "mov v16.16b, v3.16b", + "mov v16.s[3], v2.s[0]" + ] + }, + "dpps xmm0, xmm1, 11111000b": { + "ExpectedInstructionCount": 4, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddv s3, p6, z3.s", + "ext v16.16b, v2.16b, v3.16b, #4" + ] + }, + "dpps xmm0, xmm1, 11111001b": { + "ExpectedInstructionCount": 6, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddv s3, p6, z3.s", + "mov v2.s[0], v3.s[0]", + "mov v16.16b, v2.16b", + "mov v16.s[3], v3.s[0]" + ] + }, + "dpps xmm0, xmm1, 11111010b": { + "ExpectedInstructionCount": 5, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddv s3, p6, z3.s", + "dup v3.4s, v3.s[0]", + "zip1 v16.4s, v2.4s, v3.4s" + ] + }, + "dpps xmm0, xmm1, 11111011b": { + "ExpectedInstructionCount": 6, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddv s3, p6, z3.s", + "dup v3.4s, v3.s[0]", + "mov v16.16b, v3.16b", + "mov v16.s[2], v2.s[0]" + ] + }, + "dpps xmm0, xmm1, 11111100b": { + "ExpectedInstructionCount": 5, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddv s3, p6, z3.s", + "dup v3.4s, v3.s[0]", + "zip1 v16.2d, v2.2d, v3.2d" + ] + }, + "dpps xmm0, xmm1, 11111101b": { + "ExpectedInstructionCount": 6, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddv s3, p6, z3.s", + "dup v3.4s, v3.s[0]", + "mov v16.16b, v3.16b", + "mov v16.s[1], v2.s[0]" + ] + }, + "dpps xmm0, xmm1, 11111110b": { + "ExpectedInstructionCount": 6, + "Optimal": "No", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "movi v2.2d, #0x0", + "fmul v3.4s, v16.4s, v17.4s", + "faddv s3, p6, z3.s", + "dup v3.4s, v3.s[0]", + "mov v16.16b, v3.16b", + "mov v16.s[0], v2.s[0]" + ] + }, + "dpps xmm0, xmm1, 11111111b": { + "ExpectedInstructionCount": 3, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x40" + ], + "ExpectedArm64ASM": [ + "fmul v2.4s, v16.4s, v17.4s", + "faddv s2, p6, z2.s", + "dup v16.4s, v2.s[0]" + ] + }, + "dppd xmm0, xmm1, 00000000b": { + "ExpectedInstructionCount": 1, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x41" + ], + "ExpectedArm64ASM": [ + "movi v16.2d, #0x0" + ] + }, + "dppd xmm0, xmm1, 00001111b": { + "ExpectedInstructionCount": 1, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x41" + ], + "ExpectedArm64ASM": [ + "movi v16.2d, #0x0" + ] + }, + "dppd xmm0, xmm1, 11110000b": { + "ExpectedInstructionCount": 1, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x41" + ], + "ExpectedArm64ASM": [ + "movi v16.2d, #0x0" + ] + }, + "dppd xmm0, xmm1, 11111111b": { + "ExpectedInstructionCount": 3, + "Optimal": "Yes", + "Comment": [ + "0x66 0x0f 0x3a 0x41" + ], + "ExpectedArm64ASM": [ + "fmul v2.2d, v16.2d, v17.2d", + "faddv d2, p6, z2.d", + "dup v16.2d, v2.d[0]" + ] + } + } +} diff --git a/unittests/InstructionCountCI/PrimaryGroup.json b/unittests/InstructionCountCI/PrimaryGroup.json index 1f564b4d9..87480b0ed 100644 --- a/unittests/InstructionCountCI/PrimaryGroup.json +++ b/unittests/InstructionCountCI/PrimaryGroup.json @@ -3795,7 +3795,7 @@ "mov x0, x6", "mov x1, x20", "mov x2, x7", - "ldr x3, [x28, #2048]", + "ldr x3, [x28, #2064]", "str x30, [sp, #-16]!", "blr x3", "ldr x30, [sp], #16", @@ -3806,7 +3806,7 @@ "mov x0, x6", "mov x1, x20", "mov x2, x7", - "ldr x3, [x28, #2064]", + "ldr x3, [x28, #2080]", "str x30, [sp, #-16]!", "blr x3", "ldr x30, [sp], #16", @@ -3870,7 +3870,7 @@ "mov x0, x6", "mov x1, x20", "mov x2, x7", - "ldr x3, [x28, #2056]", + "ldr x3, [x28, #2072]", "str x30, [sp, #-16]!", "blr x3", "ldr x30, [sp], #16", @@ -3883,7 +3883,7 @@ "mov x0, x6", "mov x1, x20", "mov x2, x7", - "ldr x3, [x28, #2072]", + "ldr x3, [x28, #2088]", "str x30, [sp, #-16]!", "blr x3", "ldr x30, [sp], #16", diff --git a/unittests/InstructionCountCI/RPRES/DDD.json b/unittests/InstructionCountCI/RPRES/DDD.json index 8b154f2fa..09ac5014e 100644 --- a/unittests/InstructionCountCI/RPRES/DDD.json +++ b/unittests/InstructionCountCI/RPRES/DDD.json @@ -43,7 +43,7 @@ "ExpectedArm64ASM": [ "ldr d2, [x28, #768]", "frecpe s2, s2", - "dup v2.4s, v2.s[0]", + "dup v2.2s, v2.s[0]", "str d2, [x28, #752]" ] }, @@ -56,7 +56,7 @@ "ExpectedArm64ASM": [ "ldr d2, [x28, #768]", "frsqrte s2, s2", - "dup v2.4s, v2.s[0]", + "dup v2.2s, v2.s[0]", "str d2, [x28, #752]" ] } diff --git a/unittests/InstructionCountCI/Secondary.json b/unittests/InstructionCountCI/Secondary.json index da8e2dc9c..f4508b91a 100644 --- a/unittests/InstructionCountCI/Secondary.json +++ b/unittests/InstructionCountCI/Secondary.json @@ -866,7 +866,7 @@ "Comment": "0x0f 0x50", "ExpectedArm64ASM": [ "ushr v2.4s, v16.4s, #31", - "ldr q3, [x28, #1984]", + "ldr q3, [x28, #2000]", "ushl v2.4s, v2.4s, v3.4s", "addv s2, v2.4s", "mov w4, v2.s[0]" @@ -878,7 +878,7 @@ "Comment": "0x0f 0x50", "ExpectedArm64ASM": [ "ushr v2.4s, v16.4s, #31", - "ldr q3, [x28, #1984]", + "ldr q3, [x28, #2000]", "ushl v2.4s, v2.4s, v3.4s", "addv s2, v2.4s", "mov w4, v2.s[0]" @@ -1285,7 +1285,7 @@ "Comment": "0x0f 0x70", "ExpectedArm64ASM": [ "ldr d2, [x28, #768]", - "dup v2.8h, v2.h[0]", + "dup v2.4h, v2.h[0]", "str d2, [x28, #752]" ] }, @@ -1295,7 +1295,7 @@ "Comment": "0x0f 0x70", "ExpectedArm64ASM": [ "ldr d2, [x4]", - "dup v2.8h, v2.h[0]", + "dup v2.4h, v2.h[0]", "str d2, [x28, #752]" ] }, @@ -1329,7 +1329,7 @@ "Comment": "0x0f 0x70", "ExpectedArm64ASM": [ "ldr d2, [x28, #768]", - "dup v2.8h, v2.h[3]", + "dup v2.4h, v2.h[3]", "str d2, [x28, #752]" ] }, @@ -1339,7 +1339,7 @@ "Comment": "0x0f 0x70", "ExpectedArm64ASM": [ "ldr d2, [x4]", - "dup v2.8h, v2.h[3]", + "dup v2.4h, v2.h[3]", "str d2, [x28, #752]" ] }, diff --git a/unittests/InstructionCountCI/Secondary_OpSize.json b/unittests/InstructionCountCI/Secondary_OpSize.json index 9f196fc7f..83f999e39 100644 --- a/unittests/InstructionCountCI/Secondary_OpSize.json +++ b/unittests/InstructionCountCI/Secondary_OpSize.json @@ -1146,7 +1146,7 @@ "Optimal": "Yes", "Comment": "0x66 0x0f 0xd0", "ExpectedArm64ASM": [ - "ldr q2, [x28, #1952]", + "ldr q2, [x28, #1968]", "eor v2.16b, v17.16b, v2.16b", "fadd v16.2d, v16.2d, v2.2d" ] diff --git a/unittests/InstructionCountCI/Secondary_REPNE.json b/unittests/InstructionCountCI/Secondary_REPNE.json index c9be66a7f..24fba7429 100644 --- a/unittests/InstructionCountCI/Secondary_REPNE.json +++ b/unittests/InstructionCountCI/Secondary_REPNE.json @@ -495,7 +495,7 @@ "Optimal": "Yes", "Comment": "0xf2 0x0f 0xd0", "ExpectedArm64ASM": [ - "ldr q2, [x28, #1920]", + "ldr q2, [x28, #1936]", "eor v2.16b, v17.16b, v2.16b", "fadd v16.4s, v16.4s, v2.4s" ] diff --git a/unittests/InstructionCountCI/VEX_map1.json b/unittests/InstructionCountCI/VEX_map1.json index 81b4c3f1e..990eb596d 100644 --- a/unittests/InstructionCountCI/VEX_map1.json +++ b/unittests/InstructionCountCI/VEX_map1.json @@ -4366,7 +4366,7 @@ "Map 1 0b01 0xd0 128-bit" ], "ExpectedArm64ASM": [ - "ldr q2, [x28, #1952]", + "ldr q2, [x28, #1968]", "eor v2.16b, v18.16b, v2.16b", "fadd v16.2d, v17.2d, v2.2d" ] @@ -4391,7 +4391,7 @@ "Map 1 0b11 0xd0 128-bit" ], "ExpectedArm64ASM": [ - "ldr q2, [x28, #1920]", + "ldr q2, [x28, #1936]", "eor v2.16b, v18.16b, v2.16b", "fadd v16.4s, v17.4s, v2.4s" ] diff --git a/unittests/InstructionCountCI/VEX_map2.json b/unittests/InstructionCountCI/VEX_map2.json index 6545a7654..6c7c5b1b2 100644 --- a/unittests/InstructionCountCI/VEX_map2.json +++ b/unittests/InstructionCountCI/VEX_map2.json @@ -1712,7 +1712,7 @@ "Map 2 0b01 0x41 256-bit" ], "ExpectedArm64ASM": [ - "ldr q2, [x28, #1888]", + "ldr q2, [x28, #1904]", "zip1 v3.8h, v2.8h, v17.8h", "zip2 v2.8h, v2.8h, v17.8h", "umin v2.4s, v3.4s, v2.4s", diff --git a/unittests/InstructionCountCI/VEX_map3.json b/unittests/InstructionCountCI/VEX_map3.json index 0dff7ab78..878961930 100644 --- a/unittests/InstructionCountCI/VEX_map3.json +++ b/unittests/InstructionCountCI/VEX_map3.json @@ -3136,25 +3136,13 @@ ] }, "vdpps xmm0, xmm1, xmm2, 00001111b": { - "ExpectedInstructionCount": 13, + "ExpectedInstructionCount": 1, "Optimal": "No", "Comment": [ "Map 3 0b01 0x40 128-bit" ], "ExpectedArm64ASM": [ - "movi v2.2d, #0x0", - "fmul v3.4s, v17.4s, v18.4s", - "mov v3.s[0], v2.s[0]", - "mov v3.s[1], v2.s[0]", - "mov v3.s[2], v2.s[0]", - "mov v3.s[3], v2.s[0]", - "faddp v3.4s, v3.4s, v2.4s", - "faddp v3.4s, v3.4s, v2.4s", - "mov v2.s[0], v3.s[0]", - "mov v2.s[1], v3.s[0]", - "mov v2.s[2], v3.s[0]", - "mov v16.16b, v2.16b", - "mov v16.s[3], v3.s[0]" + "movi v16.2d, #0x0" ] }, "vdpps xmm0, xmm1, xmm2, 11110000b": { @@ -3168,21 +3156,16 @@ ] }, "vdpps xmm0, xmm1, xmm2, 11111111b": { - "ExpectedInstructionCount": 9, + "ExpectedInstructionCount": 4, "Optimal": "No", "Comment": [ "Map 3 0b01 0x40 128-bit" ], "ExpectedArm64ASM": [ - "movi v2.2d, #0x0", - "fmul v3.4s, v17.4s, v18.4s", - "faddp v3.4s, v3.4s, v2.4s", - "faddp v3.4s, v3.4s, v2.4s", - "mov v2.s[0], v3.s[0]", - "mov v2.s[1], v3.s[0]", - "mov v2.s[2], v3.s[0]", - "mov v16.16b, v2.16b", - "mov v16.s[3], v3.s[0]" + "fmul v2.4s, v17.4s, v18.4s", + "faddp v2.4s, v2.4s, v2.4s", + "faddp s2, v2.2s", + "dup v16.4s, v2.s[0]" ] }, "vdpps ymm0, ymm1, ymm2, 00000000b": { @@ -3356,20 +3339,13 @@ ] }, "vdppd xmm0, xmm1, xmm2, 00001111b": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 1, "Optimal": "No", "Comment": [ "Map 3 0b01 0x41 128-bit" ], "ExpectedArm64ASM": [ - "movi v2.2d, #0x0", - "fmul v3.2d, v17.2d, v18.2d", - "mov v3.d[0], v2.d[0]", - "mov v3.d[1], v2.d[0]", - "faddp v3.2d, v3.2d, v2.2d", - "mov v2.d[0], v3.d[0]", - "mov v16.16b, v2.16b", - "mov v16.d[1], v3.d[0]" + "movi v16.2d, #0x0" ] }, "vdppd xmm0, xmm1, xmm2, 11110000b": { @@ -3383,18 +3359,15 @@ ] }, "vdppd xmm0, xmm1, xmm2, 11111111b": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 3, "Optimal": "No", "Comment": [ "Map 3 0b01 0x41 128-bit" ], "ExpectedArm64ASM": [ - "movi v2.2d, #0x0", - "fmul v3.2d, v17.2d, v18.2d", - "faddp v3.2d, v3.2d, v2.2d", - "mov v2.d[0], v3.d[0]", - "mov v16.16b, v2.16b", - "mov v16.d[1], v3.d[0]" + "fmul v2.2d, v17.2d, v18.2d", + "faddp d2, v2.2d", + "dup v16.2d, v2.d[0]" ] }, "vmpsadbw xmm0, xmm1, xmm2, 000b": { @@ -4698,7 +4671,7 @@ "Map 3 0b01 0xdf 128-bit" ], "ExpectedArm64ASM": [ - "ldr q2, [x28, #2000]", + "ldr q2, [x28, #2016]", "movi v3.2d, #0x0", "mov v16.16b, v17.16b", "unimplemented (Unimplemented)", @@ -4712,7 +4685,7 @@ "Map 3 0b01 0xdf 128-bit" ], "ExpectedArm64ASM": [ - "ldr q2, [x28, #2000]", + "ldr q2, [x28, #2016]", "movi v3.2d, #0x0", "mov v16.16b, v17.16b", "unimplemented (Unimplemented)", From 826e15aea91ca335c0c89fbe51a367f86768e7b5 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Fri, 20 Oct 2023 17:53:30 +0200 Subject: [PATCH 6/6] unittests/ASM: Adds dpps/dppd broadcast mask tests Ensures that the optimization around the broadcast mask is correct. --- unittests/ASM/H0F38/66_40_2.asm | 63 +++++++++++++++++++++++++++++++++ unittests/ASM/H0F3A/66_41_2.asm | 63 +++++++++++++++++++++++++++++++++ 2 files changed, 126 insertions(+) create mode 100644 unittests/ASM/H0F38/66_40_2.asm create mode 100644 unittests/ASM/H0F3A/66_41_2.asm diff --git a/unittests/ASM/H0F38/66_40_2.asm b/unittests/ASM/H0F38/66_40_2.asm new file mode 100644 index 000000000..784a235d7 --- /dev/null +++ b/unittests/ASM/H0F38/66_40_2.asm @@ -0,0 +1,63 @@ +%ifdef CONFIG +{ + "RegData": { + "XMM0": ["0", "0"], + "XMM1": ["0x00000000c197f874", "0"], + "XMM2": ["0xff80000000000000", "0x0000000000000000"], + "XMM3": ["0x7a147e317a147e31", "0x0000000000000000"], + "XMM4": ["0x0000000000000000", "0x000000006cd0f887"], + "XMM5": ["0x000000007f800000", "0x000000007f800000"], + "XMM6": ["0xff80000000000000", "0x00000000ff800000"], + "XMM7": ["0xfc944256fc944256", "0x00000000fc944256"], + "XMM8": ["0x0000000000000000", "0xc3ac072e00000000"], + "XMM9": ["0x000000005c5c09a3", "0x5c5c09a300000000"], + "XMM10": ["0xdc34227c00000000", "0xdc34227c00000000"], + "XMM11": ["0xda1627d2da1627d2", "0xda1627d200000000"], + "XMM12": ["0x0000000000000000", "0x7f8000007f800000"], + "XMM13": ["0x000000005f30e9d3", "0x5f30e9d35f30e9d3"], + "XMM14": ["0xda3f264a00000000", "0xda3f264ada3f264a"], + "XMM15": ["0x7f8000007f800000", "0x7f8000007f800000"] + } +} +%endif + +movaps xmm0, [rel .data + 16 * 0] +movaps xmm1, [rel .data + 16 * 1] +movaps xmm2, [rel .data + 16 * 2] +movaps xmm3, [rel .data + 16 * 3] +movaps xmm4, [rel .data + 16 * 4] +movaps xmm5, [rel .data + 16 * 5] +movaps xmm6, [rel .data + 16 * 6] +movaps xmm7, [rel .data + 16 * 7] +movaps xmm8, [rel .data + 16 * 8] +movaps xmm9, [rel .data + 16 * 9] +movaps xmm10, [rel .data + 16 * 10] +movaps xmm11, [rel .data + 16 * 11] +movaps xmm12, [rel .data + 16 * 12] +movaps xmm13, [rel .data + 16 * 13] +movaps xmm14, [rel .data + 16 * 14] +movaps xmm15, [rel .data + 16 * 15] + +; Full source mask but different broadcast tests +dpps xmm0, [rel .data + 16 * 16], 1111_0000b +dpps xmm1, [rel .data + 16 * 16], 1111_0001b +dpps xmm2, [rel .data + 16 * 16], 1111_0010b +dpps xmm3, [rel .data + 16 * 16], 1111_0011b +dpps xmm4, [rel .data + 16 * 16], 1111_0100b +dpps xmm5, [rel .data + 16 * 16], 1111_0101b +dpps xmm6, [rel .data + 16 * 16], 1111_0110b +dpps xmm7, [rel .data + 16 * 16], 1111_0111b +dpps xmm8, [rel .data + 16 * 16], 1111_1000b +dpps xmm9, [rel .data + 16 * 16], 1111_1001b +dpps xmm10, [rel .data + 16 * 16], 1111_1010b +dpps xmm11, [rel .data + 16 * 16], 1111_1011b +dpps xmm12, [rel .data + 16 * 16], 1111_1100b +dpps xmm13, [rel .data + 16 * 16], 1111_1101b +dpps xmm14, [rel .data + 16 * 16], 1111_1110b +dpps xmm15, [rel .data + 16 * 16], 1111_1111b + +hlt +align 16 +; 512bytes of random data +.data: +dq 83.0999,69.50512,41.02678,13.05881,5.35242,21.9932,9.67383,5.32372,29.02872,66.50151,19.30764,91.3633,40.45086,50.96153,32.64489,23.97574,90.64316,24.22547,98.9394,91.21715,90.80143,99.48407,64.97245,74.39838,35.22761,25.35321,5.8732,90.19956,33.03133,52.02952,58.38554,10.17531,47.84703,84.04831,90.02965,65.81329,96.27991,6.64479,25.58971,95.00694,88.1929,37.16964,49.52602,10.27223,77.70605,20.21439,9.8056,41.29389,15.4071,57.54286,9.61117,55.54302,52.90745,4.88086,72.52882,3.0201,56.55091,71.22749,61.84736,88.74295,47.72641,24.17404,33.70564,96.71303 diff --git a/unittests/ASM/H0F3A/66_41_2.asm b/unittests/ASM/H0F3A/66_41_2.asm new file mode 100644 index 000000000..f2eed3db1 --- /dev/null +++ b/unittests/ASM/H0F3A/66_41_2.asm @@ -0,0 +1,63 @@ +%ifdef CONFIG +{ + "RegData": { + "XMM0": ["0", "0"], + "XMM1": ["0x40a7e92935462e9e", "0"], + "XMM2": ["0", "0x40a0712d6903205c"], + "XMM3": ["0x408c728276ca7656", "0x408c728276ca7656"], + "XMM4": ["0", "0"], + "XMM5": ["0x40c0cd5f41a95ce2", "0"], + "XMM6": ["0", "0x40b84aaf198a4022"], + "XMM7": ["0x40abf229b504629d", "0x40abf229b504629d"], + "XMM8": ["0", "0"], + "XMM9": ["0x40c8384d475e602a", "0"], + "XMM10": ["0", "0x40c8d105fa49a70e"], + "XMM11": ["0x40c248e5ffd69239", "0x40c248e5ffd69239"], + "XMM12": ["0", "0"], + "XMM13": ["0x40beb622c0fe35c7", "0"], + "XMM14": ["0", "0x40b74171bb41b9ba"], + "XMM15": ["0x40ac8195a7735fbe", "0x40ac8195a7735fbe"] + } +} +%endif + +movaps xmm0, [rel .data + 16 * 0] +movaps xmm1, [rel .data + 16 * 1] +movaps xmm2, [rel .data + 16 * 2] +movaps xmm3, [rel .data + 16 * 3] +movaps xmm4, [rel .data + 16 * 4] +movaps xmm5, [rel .data + 16 * 5] +movaps xmm6, [rel .data + 16 * 6] +movaps xmm7, [rel .data + 16 * 7] +movaps xmm8, [rel .data + 16 * 8] +movaps xmm9, [rel .data + 16 * 9] +movaps xmm10, [rel .data + 16 * 10] +movaps xmm11, [rel .data + 16 * 11] +movaps xmm12, [rel .data + 16 * 12] +movaps xmm13, [rel .data + 16 * 13] +movaps xmm14, [rel .data + 16 * 14] +movaps xmm15, [rel .data + 16 * 15] + +; Full source mask but different broadcast tests +dppd xmm0, [rel .data + 16 * 16], 1111_0000b +dppd xmm1, [rel .data + 16 * 16], 1111_0001b +dppd xmm2, [rel .data + 16 * 16], 1111_0010b +dppd xmm3, [rel .data + 16 * 16], 1111_0011b +dppd xmm4, [rel .data + 16 * 16], 1111_0100b +dppd xmm5, [rel .data + 16 * 16], 1111_0101b +dppd xmm6, [rel .data + 16 * 16], 1111_0110b +dppd xmm7, [rel .data + 16 * 16], 1111_0111b +dppd xmm8, [rel .data + 16 * 16], 1111_1000b +dppd xmm9, [rel .data + 16 * 16], 1111_1001b +dppd xmm10, [rel .data + 16 * 16], 1111_1010b +dppd xmm11, [rel .data + 16 * 16], 1111_1011b +dppd xmm12, [rel .data + 16 * 16], 1111_1100b +dppd xmm13, [rel .data + 16 * 16], 1111_1101b +dppd xmm14, [rel .data + 16 * 16], 1111_1110b +dppd xmm15, [rel .data + 16 * 16], 1111_1111b + +hlt +align 16 +; 512bytes of random data +.data: +dq 83.0999,69.50512,41.02678,13.05881,5.35242,21.9932,9.67383,5.32372,29.02872,66.50151,19.30764,91.3633,40.45086,50.96153,32.64489,23.97574,90.64316,24.22547,98.9394,91.21715,90.80143,99.48407,64.97245,74.39838,35.22761,25.35321,5.8732,90.19956,33.03133,52.02952,58.38554,10.17531,47.84703,84.04831,90.02965,65.81329,96.27991,6.64479,25.58971,95.00694,88.1929,37.16964,49.52602,10.27223,77.70605,20.21439,9.8056,41.29389,15.4071,57.54286,9.61117,55.54302,52.90745,4.88086,72.52882,3.0201,56.55091,71.22749,61.84736,88.74295,47.72641,24.17404,33.70564,96.71303