From 759cc0025a6cbbd456d35946b94c150c6d010bb2 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Thu, 14 Sep 2023 21:39:40 -0700 Subject: [PATCH 1/4] OpcodeDispatcher: Add a dirty flag for tracking NZCV status Cached NZCV reads don't need to be written back at the end of the block. This will remove one instruction from the end of some blocks. --- FEXCore/Source/Interface/Core/OpcodeDispatcher.h | 9 +++++++-- FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp | 6 ++++-- 2 files changed, 11 insertions(+), 4 deletions(-) diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h index 4db854ecd..bb4ccf185 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h @@ -869,8 +869,9 @@ private: } } - OrderedNode* CachedNZCV = {}; - uint32_t PossiblySetNZCVBits = 0; + OrderedNode* CachedNZCV{}; + bool NZCVDirty{}; + uint32_t PossiblySetNZCVBits{}; fextl::map JumpTargets; bool HandledLock{false}; @@ -1115,11 +1116,13 @@ private: void SetNZCV(OrderedNode *Value) { CachedNZCV = Value; + NZCVDirty = true; } void ZeroNZCV() { CachedNZCV = _Constant(0); PossiblySetNZCVBits = 0; + NZCVDirty = true; } void ZeroCV() { @@ -1152,6 +1155,7 @@ private: // Mask off just the N bit, which now equals the sign bit CachedNZCV = _And(OpSize::i32Bit, Shifted, _Constant(1u << NBit)); PossiblySetNZCVBits = (1u << NBit); + NZCVDirty = true; } void SetNZ_ZeroCV(unsigned SrcSize, OrderedNode *Res) { @@ -1160,6 +1164,7 @@ private: if (SrcSize >= 4) { CachedNZCV = _TestNZ(SrcSize, Res); PossiblySetNZCVBits = (1u << 31) | (1u << 30); + NZCVDirty = true; } else { // N SetN_ZeroZCV(SrcSize, Res); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp index 9d314dc3c..76e1ee533 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp @@ -272,10 +272,11 @@ void OpDispatchBuilder::CalculateAF(OpSize OpSize, OrderedNode *Res, OrderedNode void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) { if (CurrentDeferredFlags.Type == FlagsGenerationType::TYPE_NONE) { // Nothing to do - if (CachedNZCV) + if (NZCVDirty && CachedNZCV) _StoreFlag(CachedNZCV, FEXCore::X86State::RFLAG_NZCV_LOC); CachedNZCV = nullptr; + NZCVDirty = false; return; } @@ -459,10 +460,11 @@ void OpDispatchBuilder::CalculateDeferredFlags(uint32_t FlagsToCalculateMask) { // Done calculating CurrentDeferredFlags.Type = FlagsGenerationType::TYPE_NONE; - if (CachedNZCV) + if (NZCVDirty && CachedNZCV) _StoreFlag(CachedNZCV, FEXCore::X86State::RFLAG_NZCV_LOC); CachedNZCV = nullptr; + NZCVDirty = false; } void OpDispatchBuilder::CalculateFlags_ADC(uint8_t SrcSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) { From 6dbbd9ecfc5b74a0f344853087443b53dea29fb5 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Thu, 14 Sep 2023 21:40:55 -0700 Subject: [PATCH 2/4] OpcodeDispatcher: Duplicate SelectCC but with Explicit result size This is a temporary measure as we are moving Select operations over to explicit sizes. Once we remove all uses of SelectCC then it will get removed. --- .../Interface/Core/OpcodeDispatcher.cpp | 227 ++++++++++++++++++ .../Source/Interface/Core/OpcodeDispatcher.h | 2 + 2 files changed, 229 insertions(+) diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index d1589230e..9652c5911 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -1061,6 +1061,233 @@ OrderedNode *OpDispatchBuilder::SelectCC(uint8_t OP, OrderedNode *TrueValue, Ord return SrcCond; } +OrderedNode *OpDispatchBuilder::SelectCCExplicitSize(uint8_t OP, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue) { + OrderedNode *SrcCond = nullptr; + + auto ZeroConst = _Constant(0); + auto OneConst = _Constant(1); + + switch (OP) { + case 0x0: { // JO - Jump if OF == 1 + auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ}, + Flag, ZeroConst, TrueValue, FalseValue); + break; + } + case 0x1:{ // JNO - Jump if OF == 0 + auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ}, + Flag, ZeroConst, TrueValue, FalseValue); + break; + } + case 0x2: { // JC - Jump if CF == 1 + auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ}, + Flag, ZeroConst, TrueValue, FalseValue); + break; + } + case 0x3: { // JNC - Jump if CF == 0 + auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ}, + Flag, ZeroConst, TrueValue, FalseValue); + break; + } + case 0x4: { // JE - Jump if ZF == 1 + auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ}, + Flag, ZeroConst, TrueValue, FalseValue); + break; + } + case 0x5: { // JNE - Jump if ZF == 0 + auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ}, + Flag, ZeroConst, TrueValue, FalseValue); + break; + } + case 0x6: { // JNA - Jump if CF == 1 || ZC == 1 + auto Flag1 = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC); + auto Flag2 = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC); + auto Check = _Or(OpSize::i32Bit, Flag1, Flag2); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ}, + Check, OneConst, TrueValue, FalseValue); + break; + } + case 0x7: { // JA - Jump if CF == 0 && ZF == 0 + auto Flag1 = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC); + auto Flag2 = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC); + auto Check = _Or(OpSize::i32Bit, Flag1, Flag2); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ}, + Check, ZeroConst, TrueValue, FalseValue); + break; + } + case 0x8: { // JS - Jump if SF == 1 + auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_SF_LOC); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ}, + Flag, ZeroConst, TrueValue, FalseValue); + break; + } + case 0x9: { // JNS - Jump if SF == 0 + auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_SF_LOC); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ}, + Flag, ZeroConst, TrueValue, FalseValue); + break; + } + case 0xA: { // JP - Jump if PF == 1 + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ}, LoadPF(), ZeroConst, TrueValue, FalseValue); + break; + } + case 0xB: { // JNP - Jump if PF == 0 + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ}, LoadPF(), ZeroConst, TrueValue, FalseValue); + break; + } + case 0xC: { // SF <> OF + auto Flag1 = GetRFLAG(FEXCore::X86State::RFLAG_SF_LOC); + auto Flag2 = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ}, + Flag1, Flag2, TrueValue, FalseValue); + break; + } + case 0xD: { // SF = OF + auto Flag1 = GetRFLAG(FEXCore::X86State::RFLAG_SF_LOC); + auto Flag2 = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ}, + Flag1, Flag2, TrueValue, FalseValue); + break; + } + case 0xE: {// ZF = 1 || SF <> OF + auto Flag1 = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC); + auto Flag2 = GetRFLAG(FEXCore::X86State::RFLAG_SF_LOC); + auto Flag3 = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC); + + auto Select1 = _Select(OpSize::i32Bit, OpSize::i32Bit, CondClassType{COND_EQ}, + Flag1, OneConst, OneConst, ZeroConst); + + auto Select2 = _Select(OpSize::i32Bit, OpSize::i32Bit, CondClassType{COND_NEQ}, + Flag2, Flag3, OneConst, ZeroConst); + + auto Check = _Or(OpSize::i32Bit, Select1, Select2); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ}, + Check, OneConst, TrueValue, FalseValue); + break; + } + case 0xF: {// ZF = 0 && SF = OF + auto Flag1 = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC); + auto Flag2 = GetRFLAG(FEXCore::X86State::RFLAG_SF_LOC); + auto Flag3 = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC); + + auto Select1 = _Select(OpSize::i32Bit, OpSize::i32Bit, CondClassType{COND_EQ}, + Flag1, ZeroConst, OneConst, ZeroConst); + + auto Select2 = _Select(OpSize::i32Bit, OpSize::i32Bit, CondClassType{COND_EQ}, + Flag2, Flag3, OneConst, ZeroConst); + + auto Check = _And(OpSize::i32Bit, Select1, Select2); + SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ}, + Check, OneConst, TrueValue, FalseValue); + break; + } + default: + LOGMAN_MSG_A_FMT("Unknown CC Op: 0x{:x}\n", OP); + return nullptr; + } + + // Try folding the flags generation in the select op + if (flagsOp == SelectionFlag::CMP) { + switch(OP) { + // SGT + case 0xF: SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_SGT}, flagsOpDestSigned, flagsOpSrcSigned, TrueValue, FalseValue); break; + // SLE + case 0xE: SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_SLE}, flagsOpDestSigned, flagsOpSrcSigned, TrueValue, FalseValue); break; + // SGE + case 0xD: SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_SGE}, flagsOpDestSigned, flagsOpSrcSigned, TrueValue, FalseValue); break; + // SL + case 0xC: SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_SLT}, flagsOpDestSigned, flagsOpSrcSigned, TrueValue, FalseValue); break; + + // not sign + //case 0x99: SrcCond = _Select(FEXCore::IR::COND_, flagsOpDestSigned, flagsOpSrcSigned, TrueValue, FalseValue, flagsOpSize); break; + // sign + //case 0x98: SrcCond = _Select(FEXCore::IR::COND_, flagsOpDestSigned, flagsOpSrcSigned, TrueValue, FalseValue, flagsOpSize); break; + + // UABove + case 0x7: SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_UGT}, flagsOpDest, flagsOpSrc, TrueValue, FalseValue); break; + // UBE + case 0x6: SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_ULE}, flagsOpDest, flagsOpSrc, TrueValue, FalseValue); break; + // NE + case 0x5: SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_NEQ}, flagsOpDest, flagsOpSrc, TrueValue, FalseValue); break; + // EQ/Zero + case 0x4: SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_EQ}, flagsOpDest, flagsOpSrc, TrueValue, FalseValue); break; + // UAE + case 0x3: SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_UGE}, flagsOpDest, flagsOpSrc, TrueValue, FalseValue); break; + // UBelow + case 0x2: SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_ULT}, flagsOpDest, flagsOpSrc, TrueValue, FalseValue); break; + + //default: printf("Missed Condition %04X OP_CMP\n", OP); break; + } + } + else if (flagsOp == SelectionFlag::AND) { + switch(OP) { + case 0x4: SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_EQ}, flagsOpDest, ZeroConst, TrueValue, FalseValue); break; + case 0x5: SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_NEQ}, flagsOpDest, ZeroConst, TrueValue, FalseValue); break; + //default: printf("Missed Condition %04X OP_AND\n", OP); break; + } + } else if (flagsOp == SelectionFlag::FCMP) { + /* + x86:ZCP + unordered { 11 1 } + greater { 00 0 } + less { 01 0 } + equal { 10 0 } + aarch64: NZCV + unordered { 0 01 1 } + greater { 0 01 0 } + less { 1 00 0 } + equal { 0 11 0 } + */ + + /* + eq = 0, // Z set Equal. + ne = 1, // Z clear Not equal. + cs = 2, // C set Carry set. + cc = 3, // C clear Carry clear. + mi = 4, // N set Negative. + pl = 5, // N clear Positive or zero. + vs = 6, // V set Overflow. + vc = 7, // V clear No overflow. + hi = 8, // C set, Z clear Unsigned higher. + ls = 9, // C clear or Z set Unsigned lower or same. + ge = 10, // N == V Greater or equal. + lt = 11, // N != V Less than. + gt = 12, // Z clear, N == V Greater than. + le = 13, // Z set or N != V Less then or equal + */ + switch(OP) { + case 0x2: // CF == 1 // less or unordered // N==1 OR V==1 // lt + SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_FLU}, flagsOpDest, flagsOpSrc, TrueValue, FalseValue); + break; + case 0x3: // CF == 0 // greater or equal (and not unordered) // N==V // ge + SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_FGE}, flagsOpDest, flagsOpSrc, TrueValue, FalseValue); + break; + case 0x6: // CF == 1 || ZF == 1 // less or equal or unordered // Z==1 OR N!=V // le + SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_FLEU}, flagsOpDest, flagsOpSrc, TrueValue, FalseValue); + break; + case 0x7: // CF == 0 && ZF == 0 // greater (and not unordered) // C==1 AND V=0 // hi + SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_FGT}, flagsOpDest, flagsOpSrc, TrueValue, FalseValue); + break; + case 0xA: // PF = 1 // unordered // V==1 // vs + SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_FU}, flagsOpDest, flagsOpSrc, TrueValue, FalseValue); + break; + case 0xB: // PF = 0 // not unordered // V==0 // vc + SrcCond = _Select(ResultSize, IR::SizeToOpSize(flagsOpSize), CondClassType{COND_FNU}, flagsOpDest, flagsOpSrc, TrueValue, FalseValue); + break; + default: + // TODO: Add more optimized cases + break; + } + } + + return SrcCond; +} + void OpDispatchBuilder::SETccOp(OpcodeArgs) { // Calculate flags early. CalculateDeferredFlags(); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h index bb4ccf185..44f8c4950 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h @@ -1279,7 +1279,9 @@ private: CachedIndexedNamedVectorConstants.clear(); } + // TODO: SelectCC is duplicated SelectCCExplicitSize. Should get removed once all users are removed. OrderedNode *SelectCC(uint8_t OP, OrderedNode *TrueValue, OrderedNode *FalseValue); + OrderedNode *SelectCCExplicitSize(uint8_t OP, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue); /** * @name Deferred RFLAG calculation and generation. From d5b58eebaf8812d6a7807581bff96e8580702bb1 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Thu, 14 Sep 2023 21:41:58 -0700 Subject: [PATCH 3/4] OpcodeDispatcher: Optimize cmov cmov was quite terrible in its implementation. Some things of note: - NZCV cache would cause store for no reason - {16,32}-bit would zero extend sources for no reason - 16-bit would zero extend result for no reason A bunch of flag testing is still doing a ubfx plus compare against zero when it could end up being a tst instead, but this is a step in the right direction and switches over to explicit sized selects. --- .../Source/Interface/Core/OpcodeDispatcher.cpp | 15 ++++++++++++--- 1 file changed, 12 insertions(+), 3 deletions(-) diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index 9652c5911..95575f235 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -1301,13 +1301,22 @@ void OpDispatchBuilder::SETccOp(OpcodeArgs) { } void OpDispatchBuilder::CMOVOp(OpcodeArgs) { + const uint8_t GPRSize = CTX->GetGPRSize(); + // Calculate flags early. CalculateDeferredFlags(); - OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); - OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1); + // Destination is always a GPR. + OrderedNode *Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags, -1); + OrderedNode *Src{}; + if (Op->Src[0].IsGPR()) { + Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], GPRSize, Op->Flags, -1); + } + else { + Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); + } - auto SrcCond = SelectCC(Op->OP & 0xF, Src, Dest); + auto SrcCond = SelectCCExplicitSize(Op->OP & 0xF, IR::SizeToOpSize(std::max(4u, GetSrcSize(Op))), Src, Dest); StoreResult(GPRClass, Op, SrcCond, -1); } From 3eb501aa27614baf46fb62f1670042b47ad20384 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Fri, 15 Sep 2023 10:11:50 -0700 Subject: [PATCH 4/4] InstCountCI: Update for optimized NZCV and cmov --- unittests/InstructionCountCI/Primary.json | 129 +- .../InstructionCountCI/Primary_32Bit.json | 13 +- unittests/InstructionCountCI/Secondary.json | 912 +++++------- unittests/InstructionCountCI/x87.json | 1296 ++++++++--------- 4 files changed, 1089 insertions(+), 1261 deletions(-) diff --git a/unittests/InstructionCountCI/Primary.json b/unittests/InstructionCountCI/Primary.json index ebce5bace..91096b444 100644 --- a/unittests/InstructionCountCI/Primary.json +++ b/unittests/InstructionCountCI/Primary.json @@ -3100,7 +3100,7 @@ "ExpectedArm64ASM": [] }, "pushf": { - "ExpectedInstructionCount": 43, + "ExpectedInstructionCount": 42, "Optimal": "No", "Comment": "0x9c", "ExpectedArm64ASM": [ @@ -3142,15 +3142,14 @@ "orr x21, x21, x22, lsl #20", "ldrb w22, [x28, #725]", "orr x21, x21, x22, lsl #21", - "and x22, x20, #0xc0000000", - "orr x21, x21, x22, lsr #24", - "orr x21, x21, #0x2", - "str x21, [x8, #-8]!", - "str w20, [x28, #728]" + "and x20, x20, #0xc0000000", + "orr x20, x21, x20, lsr #24", + "orr x20, x20, #0x2", + "str x20, [x8, #-8]!" ] }, "pushfq": { - "ExpectedInstructionCount": 43, + "ExpectedInstructionCount": 42, "Optimal": "No", "Comment": "0x9c", "ExpectedArm64ASM": [ @@ -3192,11 +3191,10 @@ "orr x21, x21, x22, lsl #20", "ldrb w22, [x28, #725]", "orr x21, x21, x22, lsl #21", - "and x22, x20, #0xc0000000", - "orr x21, x21, x22, lsr #24", - "orr x21, x21, #0x2", - "str x21, [x8, #-8]!", - "str w20, [x28, #728]" + "and x20, x20, #0xc0000000", + "orr x20, x21, x20, lsr #24", + "orr x20, x20, #0x2", + "str x20, [x8, #-8]!" ] }, "popf": { @@ -3273,7 +3271,7 @@ ] }, "lahf": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "Yes", "Comment": "0x9f", "ExpectedArm64ASM": [ @@ -3291,11 +3289,10 @@ "eor w22, w22, w23", "ubfx w22, w22, #4, #1", "orr x21, x21, x22, lsl #4", - "and x22, x20, #0xc0000000", - "orr x21, x21, x22, lsr #24", - "orr x21, x21, #0x2", - "bfi x4, x21, #8, #8", - "str w20, [x28, #728]" + "and x20, x20, #0xc0000000", + "orr x20, x21, x20, lsr #24", + "orr x20, x20, #0x2", + "bfi x4, x20, #8, #8" ] }, "db 0x48, 0xa1; dq 0x00000000e0000008": { @@ -3736,12 +3733,12 @@ "and w21, w21, w22", "ubfx w21, w21, #7, #1", "orr w21, w24, w21, lsl #28", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", "add x10, x10, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbnz w22, #-0x68" + "ubfx w21, w21, #30, #1", + "cbnz w21, #-0x68" ] }, "repz cmpsw": { @@ -3775,12 +3772,12 @@ "and w21, w21, w22", "ubfx w21, w21, #15, #1", "orr w21, w24, w21, lsl #28", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", "add x10, x10, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbnz w22, #-0x68" + "ubfx w21, w21, #30, #1", + "cbnz w21, #-0x68" ] }, "repz cmpsd": { @@ -3803,12 +3800,12 @@ "cmp w22, w21", "mrs x21, nzcv", "eor w21, w21, #0x20000000", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", "add x10, x10, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbnz w22, #-0x3c" + "ubfx w21, w21, #30, #1", + "cbnz w21, #-0x3c" ] }, "repz cmpsq": { @@ -3831,12 +3828,12 @@ "cmp x22, x21", "mrs x21, nzcv", "eor w21, w21, #0x20000000", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", "add x10, x10, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbnz w22, #-0x3c" + "ubfx w21, w21, #30, #1", + "cbnz w21, #-0x3c" ] }, "repnz cmpsb": { @@ -3870,12 +3867,12 @@ "and w21, w21, w22", "ubfx w21, w21, #7, #1", "orr w21, w24, w21, lsl #28", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", "add x10, x10, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbz w22, #-0x68" + "ubfx w21, w21, #30, #1", + "cbz w21, #-0x68" ] }, "repnz cmpsw": { @@ -3909,12 +3906,12 @@ "and w21, w21, w22", "ubfx w21, w21, #15, #1", "orr w21, w24, w21, lsl #28", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", "add x10, x10, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbz w22, #-0x68" + "ubfx w21, w21, #30, #1", + "cbz w21, #-0x68" ] }, "repnz cmpsd": { @@ -3937,12 +3934,12 @@ "cmp w22, w21", "mrs x21, nzcv", "eor w21, w21, #0x20000000", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", "add x10, x10, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbz w22, #-0x3c" + "ubfx w21, w21, #30, #1", + "cbz w21, #-0x3c" ] }, "repnz cmpsq": { @@ -3965,12 +3962,12 @@ "cmp x22, x21", "mrs x21, nzcv", "eor w21, w21, #0x20000000", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", "add x10, x10, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbz w22, #-0x3c" + "ubfx w21, w21, #30, #1", + "cbz w21, #-0x3c" ] }, "test al, 1": { @@ -4460,11 +4457,11 @@ "and w21, w22, w21", "ubfx w21, w21, #7, #1", "orr w21, w24, w21, lsl #28", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbnz w22, #-0x64" + "ubfx w21, w21, #30, #1", + "cbnz w21, #-0x64" ] }, "repz scasw": { @@ -4498,11 +4495,11 @@ "and w21, w22, w21", "ubfx w21, w21, #15, #1", "orr w21, w24, w21, lsl #28", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbnz w22, #-0x64" + "ubfx w21, w21, #30, #1", + "cbnz w21, #-0x64" ] }, "repz scasd": { @@ -4525,11 +4522,11 @@ "cmp w21, w22", "mrs x21, nzcv", "eor w21, w21, #0x20000000", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbnz w22, #-0x38" + "ubfx w21, w21, #30, #1", + "cbnz w21, #-0x38" ] }, "repz scasq": { @@ -4551,11 +4548,11 @@ "cmp x4, x21", "mrs x21, nzcv", "eor w21, w21, #0x20000000", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbnz w22, #-0x34" + "ubfx w21, w21, #30, #1", + "cbnz w21, #-0x34" ] }, "repnz scasb": { @@ -4589,11 +4586,11 @@ "and w21, w22, w21", "ubfx w21, w21, #7, #1", "orr w21, w24, w21, lsl #28", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbz w22, #-0x64" + "ubfx w21, w21, #30, #1", + "cbz w21, #-0x64" ] }, "repnz scasw": { @@ -4627,11 +4624,11 @@ "and w21, w22, w21", "ubfx w21, w21, #15, #1", "orr w21, w24, w21, lsl #28", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbz w22, #-0x64" + "ubfx w21, w21, #30, #1", + "cbz w21, #-0x64" ] }, "repnz scasd": { @@ -4654,11 +4651,11 @@ "cmp w21, w22", "mrs x21, nzcv", "eor w21, w21, #0x20000000", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbz w22, #-0x38" + "ubfx w21, w21, #30, #1", + "cbz w21, #-0x38" ] }, "repnz scasq": { @@ -4680,11 +4677,11 @@ "cmp x4, x21", "mrs x21, nzcv", "eor w21, w21, #0x20000000", + "str w21, [x28, #728]", "sub x5, x5, #0x1 (1)", "add x11, x11, x20", - "ubfx w22, w21, #30, #1", - "str w21, [x28, #728]", - "cbz w22, #-0x34" + "ubfx w21, w21, #30, #1", + "cbz w21, #-0x34" ] }, "mov al, 0xff": { diff --git a/unittests/InstructionCountCI/Primary_32Bit.json b/unittests/InstructionCountCI/Primary_32Bit.json index b958672c8..b9ccae16f 100644 --- a/unittests/InstructionCountCI/Primary_32Bit.json +++ b/unittests/InstructionCountCI/Primary_32Bit.json @@ -563,18 +563,17 @@ ] }, "salc": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 7, "Optimal": "No", "Comment": "0xd6", "ExpectedArm64ASM": [ "uxtb w20, w4", "ldr w21, [x28, #728]", - "ubfx w22, w21, #29, #1", - "add w20, w20, w22", - "uxtb w22, w4", - "sub w20, w22, w20", - "bfxil w4, w20, #0, #8", - "str w21, [x28, #728]" + "ubfx w21, w21, #29, #1", + "add w20, w20, w21", + "uxtb w21, w4", + "sub w20, w21, w20", + "bfxil w4, w20, #0, #8" ] } } diff --git a/unittests/InstructionCountCI/Secondary.json b/unittests/InstructionCountCI/Secondary.json index 09b44df72..63d1af271 100644 --- a/unittests/InstructionCountCI/Secondary.json +++ b/unittests/InstructionCountCI/Secondary.json @@ -265,460 +265,386 @@ ] }, "cmovo ax, bx": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0x0f 0x40", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #28, #1", - "cmp x23, #0x0 (0)", - "csel w20, w20, w21, ne", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w20, w20, #28, #1", + "cmp w20, #0x0 (0)", + "csel w20, w7, w4, ne", + "bfxil x4, x20, #0, #16" ] }, "cmovo eax, ebx": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 4, "Optimal": "No", "Comment": "0x0f 0x40", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #28, #1", - "cmp x23, #0x0 (0)", - "csel w4, w20, w21, ne", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w20, w20, #28, #1", + "cmp w20, #0x0 (0)", + "csel w4, w7, w4, ne" ] }, "cmovo rax, rbx": { - "ExpectedInstructionCount": 5, + "ExpectedInstructionCount": 4, "Optimal": "Yes", "Comment": "0x0f 0x40", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #28, #1", - "cmp x21, #0x0 (0)", - "csel x4, x7, x4, ne", - "str w20, [x28, #728]" + "ubfx w20, w20, #28, #1", + "cmp w20, #0x0 (0)", + "csel x4, x7, x4, ne" ] }, "cmovno ax, bx": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0x0f 0x41", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #28, #1", - "cmp x23, #0x0 (0)", - "csel w20, w20, w21, eq", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w20, w20, #28, #1", + "cmp w20, #0x0 (0)", + "csel w20, w7, w4, eq", + "bfxil x4, x20, #0, #16" ] }, "cmovno eax, ebx": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 4, "Optimal": "No", "Comment": "0x0f 0x41", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #28, #1", - "cmp x23, #0x0 (0)", - "csel w4, w20, w21, eq", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w20, w20, #28, #1", + "cmp w20, #0x0 (0)", + "csel w4, w7, w4, eq" ] }, "cmovno rax, rbx": { - "ExpectedInstructionCount": 5, + "ExpectedInstructionCount": 4, "Optimal": "Yes", "Comment": "0x0f 0x41", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #28, #1", - "cmp x21, #0x0 (0)", - "csel x4, x7, x4, eq", - "str w20, [x28, #728]" + "ubfx w20, w20, #28, #1", + "cmp w20, #0x0 (0)", + "csel x4, x7, x4, eq" ] }, "cmovb ax, bx": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0x0f 0x42", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #29, #1", - "cmp x23, #0x0 (0)", - "csel w20, w20, w21, ne", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w20, w20, #29, #1", + "cmp w20, #0x0 (0)", + "csel w20, w7, w4, ne", + "bfxil x4, x20, #0, #16" ] }, "cmovb eax, ebx": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 4, "Optimal": "No", "Comment": "0x0f 0x42", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #29, #1", - "cmp x23, #0x0 (0)", - "csel w4, w20, w21, ne", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w20, w20, #29, #1", + "cmp w20, #0x0 (0)", + "csel w4, w7, w4, ne" ] }, "cmovb rax, rbx": { - "ExpectedInstructionCount": 5, + "ExpectedInstructionCount": 4, "Optimal": "No", "Comment": "0x0f 0x42", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "cmp x21, #0x0 (0)", - "csel x4, x7, x4, ne", - "str w20, [x28, #728]" + "ubfx w20, w20, #29, #1", + "cmp w20, #0x0 (0)", + "csel x4, x7, x4, ne" ] }, "cmovnb ax, bx": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0x0f 0x43", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #29, #1", - "cmp x23, #0x0 (0)", - "csel w20, w20, w21, eq", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w20, w20, #29, #1", + "cmp w20, #0x0 (0)", + "csel w20, w7, w4, eq", + "bfxil x4, x20, #0, #16" ] }, "cmovnb eax, ebx": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 4, "Optimal": "No", "Comment": "0x0f 0x43", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #29, #1", - "cmp x23, #0x0 (0)", - "csel w4, w20, w21, eq", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w20, w20, #29, #1", + "cmp w20, #0x0 (0)", + "csel w4, w7, w4, eq" ] }, "cmovnb rax, rbx": { - "ExpectedInstructionCount": 5, + "ExpectedInstructionCount": 4, "Optimal": "Yes", "Comment": "0x0f 0x43", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "cmp x21, #0x0 (0)", - "csel x4, x7, x4, eq", - "str w20, [x28, #728]" + "ubfx w20, w20, #29, #1", + "cmp w20, #0x0 (0)", + "csel x4, x7, x4, eq" ] }, "cmovz ax, bx": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0x0f 0x44", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #30, #1", - "cmp x23, #0x0 (0)", - "csel w20, w20, w21, ne", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "cmp w20, #0x0 (0)", + "csel w20, w7, w4, ne", + "bfxil x4, x20, #0, #16" ] }, "cmovz eax, ebx": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 4, "Optimal": "No", "Comment": "0x0f 0x44", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #30, #1", - "cmp x23, #0x0 (0)", - "csel w4, w20, w21, ne", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "cmp w20, #0x0 (0)", + "csel w4, w7, w4, ne" ] }, "cmovz rax, rbx": { - "ExpectedInstructionCount": 5, + "ExpectedInstructionCount": 4, "Optimal": "Yes", "Comment": "0x0f 0x44", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #30, #1", - "cmp x21, #0x0 (0)", - "csel x4, x7, x4, ne", - "str w20, [x28, #728]" + "ubfx w20, w20, #30, #1", + "cmp w20, #0x0 (0)", + "csel x4, x7, x4, ne" ] }, "cmovnz ax, bx": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0x0f 0x45", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #30, #1", - "cmp x23, #0x0 (0)", - "csel w20, w20, w21, eq", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "cmp w20, #0x0 (0)", + "csel w20, w7, w4, eq", + "bfxil x4, x20, #0, #16" ] }, "cmovnz eax, ebx": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 4, "Optimal": "No", "Comment": "0x0f 0x45", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #30, #1", - "cmp x23, #0x0 (0)", - "csel w4, w20, w21, eq", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "cmp w20, #0x0 (0)", + "csel w4, w7, w4, eq" ] }, "cmovnz rax, rbx": { - "ExpectedInstructionCount": 5, + "ExpectedInstructionCount": 4, "Optimal": "Yes", "Comment": "0x0f 0x45", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #30, #1", - "cmp x21, #0x0 (0)", - "csel x4, x7, x4, eq", - "str w20, [x28, #728]" + "ubfx w20, w20, #30, #1", + "cmp w20, #0x0 (0)", + "csel x4, x7, x4, eq" ] }, "cmovbe ax, bx": { - "ExpectedInstructionCount": 10, + "ExpectedInstructionCount": 7, "Optimal": "No", "Comment": "0x0f 0x46", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #30, #1", - "ubfx w24, w22, #29, #1", - "orr w23, w23, w24", - "cmp x23, #0x1 (1)", - "csel w20, w20, w21, eq", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w21, w20, #30, #1", + "ubfx w20, w20, #29, #1", + "orr w20, w21, w20", + "cmp w20, #0x1 (1)", + "csel w20, w7, w4, eq", + "bfxil x4, x20, #0, #16" ] }, "cmovbe eax, ebx": { - "ExpectedInstructionCount": 9, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": "0x0f 0x46", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #30, #1", - "ubfx w24, w22, #29, #1", - "orr w23, w23, w24", - "cmp x23, #0x1 (1)", - "csel w4, w20, w21, eq", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w21, w20, #30, #1", + "ubfx w20, w20, #29, #1", + "orr w20, w21, w20", + "cmp w20, #0x1 (1)", + "csel w4, w7, w4, eq" ] }, "cmovbe rax, rbx": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": "0x0f 0x46", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #30, #1", - "ubfx w22, w20, #29, #1", - "orr w21, w21, w22", - "cmp x21, #0x1 (1)", - "csel x4, x7, x4, eq", - "str w20, [x28, #728]" + "ubfx w20, w20, #29, #1", + "orr w20, w21, w20", + "cmp w20, #0x1 (1)", + "csel x4, x7, x4, eq" ] }, "cmovnbe ax, bx": { - "ExpectedInstructionCount": 10, - "Optimal": "No", - "Comment": "0x0f 0x47", - "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #30, #1", - "ubfx w24, w22, #29, #1", - "orr w23, w23, w24", - "cmp x23, #0x0 (0)", - "csel w20, w20, w21, eq", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" - ] - }, - "cmovnbe eax, ebx": { - "ExpectedInstructionCount": 9, - "Optimal": "No", - "Comment": "0x0f 0x47", - "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #30, #1", - "ubfx w24, w22, #29, #1", - "orr w23, w23, w24", - "cmp x23, #0x0 (0)", - "csel w4, w20, w21, eq", - "str w22, [x28, #728]" - ] - }, - "cmovnbe rax, rbx": { "ExpectedInstructionCount": 7, "Optimal": "No", "Comment": "0x0f 0x47", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #30, #1", - "ubfx w22, w20, #29, #1", - "orr w21, w21, w22", - "cmp x21, #0x0 (0)", - "csel x4, x7, x4, eq", - "str w20, [x28, #728]" + "ubfx w20, w20, #29, #1", + "orr w20, w21, w20", + "cmp w20, #0x0 (0)", + "csel w20, w7, w4, eq", + "bfxil x4, x20, #0, #16" + ] + }, + "cmovnbe eax, ebx": { + "ExpectedInstructionCount": 6, + "Optimal": "No", + "Comment": "0x0f 0x47", + "ExpectedArm64ASM": [ + "ldr w20, [x28, #728]", + "ubfx w21, w20, #30, #1", + "ubfx w20, w20, #29, #1", + "orr w20, w21, w20", + "cmp w20, #0x0 (0)", + "csel w4, w7, w4, eq" + ] + }, + "cmovnbe rax, rbx": { + "ExpectedInstructionCount": 6, + "Optimal": "No", + "Comment": "0x0f 0x47", + "ExpectedArm64ASM": [ + "ldr w20, [x28, #728]", + "ubfx w21, w20, #30, #1", + "ubfx w20, w20, #29, #1", + "orr w20, w21, w20", + "cmp w20, #0x0 (0)", + "csel x4, x7, x4, eq" ] }, "cmovs ax, bx": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0x0f 0x48", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "lsr w23, w22, #31", - "cmp x23, #0x0 (0)", - "csel w20, w20, w21, ne", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "lsr w20, w20, #31", + "cmp w20, #0x0 (0)", + "csel w20, w7, w4, ne", + "bfxil x4, x20, #0, #16" ] }, "cmovs eax, ebx": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 4, "Optimal": "No", "Comment": "0x0f 0x48", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "lsr w23, w22, #31", - "cmp x23, #0x0 (0)", - "csel w4, w20, w21, ne", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "lsr w20, w20, #31", + "cmp w20, #0x0 (0)", + "csel w4, w7, w4, ne" ] }, "cmovs rax, rbx": { - "ExpectedInstructionCount": 5, + "ExpectedInstructionCount": 4, "Optimal": "Yes", "Comment": "0x0f 0x48", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "lsr w21, w20, #31", - "cmp x21, #0x0 (0)", - "csel x4, x7, x4, ne", - "str w20, [x28, #728]" + "lsr w20, w20, #31", + "cmp w20, #0x0 (0)", + "csel x4, x7, x4, ne" ] }, "cmovns ax, bx": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0x0f 0x49", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "lsr w23, w22, #31", - "cmp x23, #0x0 (0)", - "csel w20, w20, w21, eq", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "lsr w20, w20, #31", + "cmp w20, #0x0 (0)", + "csel w20, w7, w4, eq", + "bfxil x4, x20, #0, #16" ] }, "cmovns eax, ebx": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 4, "Optimal": "No", "Comment": "0x0f 0x49", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "lsr w23, w22, #31", - "cmp x23, #0x0 (0)", - "csel w4, w20, w21, eq", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "lsr w20, w20, #31", + "cmp w20, #0x0 (0)", + "csel w4, w7, w4, eq" ] }, "cmovns rax, rbx": { - "ExpectedInstructionCount": 5, + "ExpectedInstructionCount": 4, "Optimal": "Yes", "Comment": "0x0f 0x49", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "lsr w21, w20, #31", - "cmp x21, #0x0 (0)", - "csel x4, x7, x4, eq", - "str w20, [x28, #728]" + "lsr w20, w20, #31", + "cmp w20, #0x0 (0)", + "csel x4, x7, x4, eq" ] }, "cmovpe ax, bx": { - "ExpectedInstructionCount": 11, + "ExpectedInstructionCount": 9, "Optimal": "No", "Comment": "0x0f 0x4a", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldrb w22, [x28, #706]", - "eor w22, w22, #0x1", - "fmov s2, w22", + "ldrb w20, [x28, #706]", + "eor w20, w20, #0x1", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w22, v2.b[0]", - "and x22, x22, #0x1", - "cmp x22, #0x0 (0)", - "csel w20, w20, w21, ne", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "cmp w20, #0x0 (0)", + "csel w20, w7, w4, ne", "bfxil x4, x20, #0, #16" ] }, "cmovpe eax, ebx": { - "ExpectedInstructionCount": 10, + "ExpectedInstructionCount": 8, "Optimal": "No", "Comment": "0x0f 0x4a", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldrb w22, [x28, #706]", - "eor w22, w22, #0x1", - "fmov s2, w22", + "ldrb w20, [x28, #706]", + "eor w20, w20, #0x1", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w22, v2.b[0]", - "and x22, x22, #0x1", - "cmp x22, #0x0 (0)", - "csel w4, w20, w21, ne" + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "cmp w20, #0x0 (0)", + "csel w4, w7, w4, ne" ] }, "cmovpe rax, rbx": { @@ -732,43 +658,39 @@ "cnt v2.16b, v2.16b", "umov w20, v2.b[0]", "and x20, x20, #0x1", - "cmp x20, #0x0 (0)", + "cmp w20, #0x0 (0)", "csel x4, x7, x4, ne" ] }, "cmovnp ax, bx": { - "ExpectedInstructionCount": 11, + "ExpectedInstructionCount": 9, "Optimal": "No", "Comment": "0x0f 0x4b", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldrb w22, [x28, #706]", - "eor w22, w22, #0x1", - "fmov s2, w22", + "ldrb w20, [x28, #706]", + "eor w20, w20, #0x1", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w22, v2.b[0]", - "and x22, x22, #0x1", - "cmp x22, #0x0 (0)", - "csel w20, w20, w21, eq", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "cmp w20, #0x0 (0)", + "csel w20, w7, w4, eq", "bfxil x4, x20, #0, #16" ] }, "cmovnp eax, ebx": { - "ExpectedInstructionCount": 10, + "ExpectedInstructionCount": 8, "Optimal": "No", "Comment": "0x0f 0x4b", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldrb w22, [x28, #706]", - "eor w22, w22, #0x1", - "fmov s2, w22", + "ldrb w20, [x28, #706]", + "eor w20, w20, #0x1", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w22, v2.b[0]", - "and x22, x22, #0x1", - "cmp x22, #0x0 (0)", - "csel w4, w20, w21, eq" + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "cmp w20, #0x0 (0)", + "csel w4, w7, w4, eq" ] }, "cmovnp rax, rbx": { @@ -782,204 +704,140 @@ "cnt v2.16b, v2.16b", "umov w20, v2.b[0]", "and x20, x20, #0x1", - "cmp x20, #0x0 (0)", + "cmp w20, #0x0 (0)", "csel x4, x7, x4, eq" ] }, "cmovl ax, bx": { - "ExpectedInstructionCount": 9, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": "0x0f 0x4c", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "lsr w23, w22, #31", - "ubfx w24, w22, #28, #1", - "cmp w23, w24", - "csel w20, w20, w21, ne", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "lsr w21, w20, #31", + "ubfx w20, w20, #28, #1", + "cmp w21, w20", + "csel w20, w7, w4, ne", + "bfxil x4, x20, #0, #16" ] }, "cmovl eax, ebx": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0x0f 0x4c", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "lsr w23, w22, #31", - "ubfx w24, w22, #28, #1", - "cmp w23, w24", - "csel w4, w20, w21, ne", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "lsr w21, w20, #31", + "ubfx w20, w20, #28, #1", + "cmp w21, w20", + "csel w4, w7, w4, ne" ] }, "cmovl rax, rbx": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 5, "Optimal": "Yes", "Comment": "0x0f 0x4c", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "lsr w21, w20, #31", - "ubfx w22, w20, #28, #1", - "cmp w21, w22", - "csel x4, x7, x4, ne", - "str w20, [x28, #728]" + "ubfx w20, w20, #28, #1", + "cmp w21, w20", + "csel x4, x7, x4, ne" ] }, "cmovnl ax, bx": { - "ExpectedInstructionCount": 9, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": "0x0f 0x4d", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "lsr w23, w22, #31", - "ubfx w24, w22, #28, #1", - "cmp w23, w24", - "csel w20, w20, w21, eq", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "lsr w21, w20, #31", + "ubfx w20, w20, #28, #1", + "cmp w21, w20", + "csel w20, w7, w4, eq", + "bfxil x4, x20, #0, #16" ] }, "cmovnl eax, ebx": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0x0f 0x4d", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "lsr w23, w22, #31", - "ubfx w24, w22, #28, #1", - "cmp w23, w24", - "csel w4, w20, w21, eq", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "lsr w21, w20, #31", + "ubfx w20, w20, #28, #1", + "cmp w21, w20", + "csel w4, w7, w4, eq" ] }, "cmovnl rax, rbx": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 5, "Optimal": "Yes", "Comment": "0x0f 0x4d", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "lsr w21, w20, #31", - "ubfx w22, w20, #28, #1", - "cmp w21, w22", - "csel x4, x7, x4, eq", - "str w20, [x28, #728]" + "ubfx w20, w20, #28, #1", + "cmp w21, w20", + "csel x4, x7, x4, eq" ] }, "cmovle ax, bx": { - "ExpectedInstructionCount": 15, + "ExpectedInstructionCount": 12, "Optimal": "No", "Comment": "0x0f 0x4e", "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #30, #1", - "lsr w24, w22, #31", - "ubfx w25, w22, #28, #1", - "cmp x23, #0x1 (1)", - "cset x23, eq", - "cmp w24, w25", - "cset x24, ne", - "orr x23, x23, x24", - "cmp x23, #0x1 (1)", - "csel w20, w20, w21, eq", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w21, w20, #30, #1", + "lsr w22, w20, #31", + "ubfx w20, w20, #28, #1", + "cmp w21, #0x1 (1)", + "cset w21, eq", + "cmp w22, w20", + "cset w20, ne", + "orr w20, w21, w20", + "cmp w20, #0x1 (1)", + "csel w20, w7, w4, eq", + "bfxil x4, x20, #0, #16" ] }, "cmovle eax, ebx": { - "ExpectedInstructionCount": 14, + "ExpectedInstructionCount": 11, "Optimal": "No", "Comment": "0x0f 0x4e", "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #30, #1", - "lsr w24, w22, #31", - "ubfx w25, w22, #28, #1", - "cmp x23, #0x1 (1)", - "cset x23, eq", - "cmp w24, w25", - "cset x24, ne", - "orr x23, x23, x24", - "cmp x23, #0x1 (1)", - "csel w4, w20, w21, eq", - "str w22, [x28, #728]" + "ldr w20, [x28, #728]", + "ubfx w21, w20, #30, #1", + "lsr w22, w20, #31", + "ubfx w20, w20, #28, #1", + "cmp w21, #0x1 (1)", + "cset w21, eq", + "cmp w22, w20", + "cset w20, ne", + "orr w20, w21, w20", + "cmp w20, #0x1 (1)", + "csel w4, w7, w4, eq" ] }, "cmovle rax, rbx": { - "ExpectedInstructionCount": 12, + "ExpectedInstructionCount": 11, "Optimal": "No", "Comment": "0x0f 0x4e", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #30, #1", "lsr w22, w20, #31", - "ubfx w23, w20, #28, #1", - "cmp x21, #0x1 (1)", - "cset x21, eq", - "cmp w22, w23", - "cset x22, ne", - "orr x21, x21, x22", - "cmp x21, #0x1 (1)", - "csel x4, x7, x4, eq", - "str w20, [x28, #728]" + "ubfx w20, w20, #28, #1", + "cmp w21, #0x1 (1)", + "cset w21, eq", + "cmp w22, w20", + "cset w20, ne", + "orr w20, w21, w20", + "cmp w20, #0x1 (1)", + "csel x4, x7, x4, eq" ] }, "cmovnle ax, bx": { - "ExpectedInstructionCount": 15, - "Optimal": "No", - "Comment": "0x0f 0x4f", - "ExpectedArm64ASM": [ - "uxth w20, w7", - "uxth w21, w4", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #30, #1", - "lsr w24, w22, #31", - "ubfx w25, w22, #28, #1", - "cmp x23, #0x0 (0)", - "cset x23, eq", - "cmp w24, w25", - "cset x24, eq", - "and x23, x23, x24", - "cmp x23, #0x1 (1)", - "csel w20, w20, w21, eq", - "bfxil x4, x20, #0, #16", - "str w22, [x28, #728]" - ] - }, - "cmovnle eax, ebx": { - "ExpectedInstructionCount": 14, - "Optimal": "No", - "Comment": "0x0f 0x4f", - "ExpectedArm64ASM": [ - "lsr w20, w7, #0", - "lsr w21, w4, #0", - "ldr w22, [x28, #728]", - "ubfx w23, w22, #30, #1", - "lsr w24, w22, #31", - "ubfx w25, w22, #28, #1", - "cmp x23, #0x0 (0)", - "cset x23, eq", - "cmp w24, w25", - "cset x24, eq", - "and x23, x23, x24", - "cmp x23, #0x1 (1)", - "csel w4, w20, w21, eq", - "str w22, [x28, #728]" - ] - }, - "cmovnle rax, rbx": { "ExpectedInstructionCount": 12, "Optimal": "No", "Comment": "0x0f 0x4f", @@ -987,15 +845,51 @@ "ldr w20, [x28, #728]", "ubfx w21, w20, #30, #1", "lsr w22, w20, #31", - "ubfx w23, w20, #28, #1", - "cmp x21, #0x0 (0)", - "cset x21, eq", - "cmp w22, w23", - "cset x22, eq", - "and x21, x21, x22", - "cmp x21, #0x1 (1)", - "csel x4, x7, x4, eq", - "str w20, [x28, #728]" + "ubfx w20, w20, #28, #1", + "cmp w21, #0x0 (0)", + "cset w21, eq", + "cmp w22, w20", + "cset w20, eq", + "and w20, w21, w20", + "cmp w20, #0x1 (1)", + "csel w20, w7, w4, eq", + "bfxil x4, x20, #0, #16" + ] + }, + "cmovnle eax, ebx": { + "ExpectedInstructionCount": 11, + "Optimal": "No", + "Comment": "0x0f 0x4f", + "ExpectedArm64ASM": [ + "ldr w20, [x28, #728]", + "ubfx w21, w20, #30, #1", + "lsr w22, w20, #31", + "ubfx w20, w20, #28, #1", + "cmp w21, #0x0 (0)", + "cset w21, eq", + "cmp w22, w20", + "cset w20, eq", + "and w20, w21, w20", + "cmp w20, #0x1 (1)", + "csel w4, w7, w4, eq" + ] + }, + "cmovnle rax, rbx": { + "ExpectedInstructionCount": 11, + "Optimal": "No", + "Comment": "0x0f 0x4f", + "ExpectedArm64ASM": [ + "ldr w20, [x28, #728]", + "ubfx w21, w20, #30, #1", + "lsr w22, w20, #31", + "ubfx w20, w20, #28, #1", + "cmp w21, #0x0 (0)", + "cset w21, eq", + "cmp w22, w20", + "cset w20, eq", + "and w20, w21, w20", + "cmp w20, #0x1 (1)", + "csel x4, x7, x4, eq" ] }, "movmskps eax, xmm0": { @@ -1568,137 +1462,127 @@ ] }, "seto al": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 5, "Optimal": "Yes", "Comment": "0x0f 0x90", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #28, #1", - "cmp x21, #0x0 (0)", - "cset x21, ne", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "ubfx w20, w20, #28, #1", + "cmp x20, #0x0 (0)", + "cset x20, ne", + "bfxil x4, x20, #0, #8" ] }, "setno al": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 5, "Optimal": "Yes", "Comment": "0x0f 0x91", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #28, #1", - "cmp x21, #0x0 (0)", - "cset x21, eq", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "ubfx w20, w20, #28, #1", + "cmp x20, #0x0 (0)", + "cset x20, eq", + "bfxil x4, x20, #0, #8" ] }, "setb al": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 5, "Optimal": "Yes", "Comment": "0x0f 0x92", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "cmp x21, #0x0 (0)", - "cset x21, ne", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "ubfx w20, w20, #29, #1", + "cmp x20, #0x0 (0)", + "cset x20, ne", + "bfxil x4, x20, #0, #8" ] }, "setnb al": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 5, "Optimal": "Yes", "Comment": "0x0f 0x93", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "cmp x21, #0x0 (0)", - "cset x21, eq", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "ubfx w20, w20, #29, #1", + "cmp x20, #0x0 (0)", + "cset x20, eq", + "bfxil x4, x20, #0, #8" ] }, "setz al": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 5, "Optimal": "Yes", "Comment": "0x0f 0x94", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #30, #1", - "cmp x21, #0x0 (0)", - "cset x21, ne", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "ubfx w20, w20, #30, #1", + "cmp x20, #0x0 (0)", + "cset x20, ne", + "bfxil x4, x20, #0, #8" ] }, "setnz al": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 5, "Optimal": "Yes", "Comment": "0x0f 0x95", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #30, #1", - "cmp x21, #0x0 (0)", - "cset x21, eq", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "ubfx w20, w20, #30, #1", + "cmp x20, #0x0 (0)", + "cset x20, eq", + "bfxil x4, x20, #0, #8" ] }, "setbe al": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 7, "Optimal": "No", "Comment": "0x0f 0x96", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #30, #1", - "ubfx w22, w20, #29, #1", - "orr w21, w21, w22", - "cmp x21, #0x1 (1)", - "cset x21, eq", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "ubfx w20, w20, #29, #1", + "orr w20, w21, w20", + "cmp x20, #0x1 (1)", + "cset x20, eq", + "bfxil x4, x20, #0, #8" ] }, "setnbe al": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 7, "Optimal": "No", "Comment": "0x0f 0x97", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #30, #1", - "ubfx w22, w20, #29, #1", - "orr w21, w21, w22", - "cmp x21, #0x0 (0)", - "cset x21, eq", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "ubfx w20, w20, #29, #1", + "orr w20, w21, w20", + "cmp x20, #0x0 (0)", + "cset x20, eq", + "bfxil x4, x20, #0, #8" ] }, "sets al": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 5, "Optimal": "Yes", "Comment": "0x0f 0x98", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "lsr w21, w20, #31", - "cmp x21, #0x0 (0)", - "cset x21, ne", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "lsr w20, w20, #31", + "cmp x20, #0x0 (0)", + "cset x20, ne", + "bfxil x4, x20, #0, #8" ] }, "setns al": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 5, "Optimal": "Yes", "Comment": "0x0f 0x99", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "lsr w21, w20, #31", - "cmp x21, #0x0 (0)", - "cset x21, eq", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "lsr w20, w20, #31", + "cmp x20, #0x0 (0)", + "cset x20, eq", + "bfxil x4, x20, #0, #8" ] }, "setpe al": { @@ -1734,71 +1618,67 @@ ] }, "setl al": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": "0x0f 0x9c", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "lsr w21, w20, #31", - "ubfx w22, w20, #28, #1", - "cmp w21, w22", - "cset x21, ne", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "ubfx w20, w20, #28, #1", + "cmp w21, w20", + "cset x20, ne", + "bfxil x4, x20, #0, #8" ] }, "setnl al": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": "0x0f 0x9d", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "lsr w21, w20, #31", - "ubfx w22, w20, #28, #1", - "cmp w21, w22", - "cset x21, eq", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "ubfx w20, w20, #28, #1", + "cmp w21, w20", + "cset x20, eq", + "bfxil x4, x20, #0, #8" ] }, "setle al": { - "ExpectedInstructionCount": 13, + "ExpectedInstructionCount": 12, "Optimal": "No", "Comment": "0x0f 0x9e", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #30, #1", "lsr w22, w20, #31", - "ubfx w23, w20, #28, #1", + "ubfx w20, w20, #28, #1", "cmp x21, #0x1 (1)", "cset x21, eq", - "cmp w22, w23", - "cset x22, ne", - "orr x21, x21, x22", - "cmp x21, #0x1 (1)", - "cset x21, eq", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "cmp w22, w20", + "cset x20, ne", + "orr x20, x21, x20", + "cmp x20, #0x1 (1)", + "cset x20, eq", + "bfxil x4, x20, #0, #8" ] }, "setnle al": { - "ExpectedInstructionCount": 13, + "ExpectedInstructionCount": 12, "Optimal": "No", "Comment": "0x0f 0x9f", "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #30, #1", "lsr w22, w20, #31", - "ubfx w23, w20, #28, #1", + "ubfx w20, w20, #28, #1", "cmp x21, #0x0 (0)", "cset x21, eq", - "cmp w22, w23", - "cset x22, eq", - "and x21, x21, x22", - "cmp x21, #0x1 (1)", - "cset x21, eq", - "bfxil x4, x21, #0, #8", - "str w20, [x28, #728]" + "cmp w22, w20", + "cset x20, eq", + "and x20, x21, x20", + "cmp x20, #0x1 (1)", + "cset x20, eq", + "bfxil x4, x20, #0, #8" ] }, "push fs": { diff --git a/unittests/InstructionCountCI/x87.json b/unittests/InstructionCountCI/x87.json index e148931d7..6659f4c10 100644 --- a/unittests/InstructionCountCI/x87.json +++ b/unittests/InstructionCountCI/x87.json @@ -8308,223 +8308,215 @@ ] }, "fcmovb st0, st0": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xda 11b 0xc0 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x0 (0)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x0 (0)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovb st0, st1": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xda 11b 0xc1 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x1 (1)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x1 (1)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovb st0, st2": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xda 11b 0xc2 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x2 (2)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x2 (2)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovb st0, st3": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xda 11b 0xc3 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x3 (3)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x3 (3)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovb st0, st4": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xda 11b 0xc4 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x4 (4)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x4 (4)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovb st0, st5": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xda 11b 0xc5 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x5 (5)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x5 (5)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovb st0, st6": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xda 11b 0xc6 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x6 (6)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x6 (6)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovb st0, st7": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xda 11b 0xc7 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x7 (7)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x7 (7)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmove st0, st0": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xc8 /1" @@ -8532,28 +8524,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x0 (0)", - "and w22, w22, #0x7", + "add w21, w20, #0x0 (0)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmove st0, st1": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xc9 /1" @@ -8561,28 +8552,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x1 (1)", - "and w22, w22, #0x7", + "add w21, w20, #0x1 (1)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmove st0, st2": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xca /1" @@ -8590,28 +8580,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x2 (2)", - "and w22, w22, #0x7", + "add w21, w20, #0x2 (2)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmove st0, st3": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xcb /1" @@ -8619,28 +8608,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x3 (3)", - "and w22, w22, #0x7", + "add w21, w20, #0x3 (3)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmove st0, st4": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xcc /1" @@ -8648,28 +8636,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x4 (4)", - "and w22, w22, #0x7", + "add w21, w20, #0x4 (4)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmove st0, st5": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xcd /1" @@ -8677,28 +8664,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x5 (5)", - "and w22, w22, #0x7", + "add w21, w20, #0x5 (5)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmove st0, st6": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xce /1" @@ -8706,28 +8692,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x6 (6)", - "and w22, w22, #0x7", + "add w21, w20, #0x6 (6)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmove st0, st7": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xcf /1" @@ -8735,28 +8720,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x7 (7)", - "and w22, w22, #0x7", + "add w21, w20, #0x7 (7)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmovbe st0, st0": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xd0 /0" @@ -8764,28 +8748,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x0 (0)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x0 (0)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovbe st0, st1": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xd1 /0" @@ -8793,28 +8776,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x1 (1)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x1 (1)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovbe st0, st2": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xd2 /0" @@ -8822,28 +8804,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x2 (2)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x2 (2)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovbe st0, st3": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xd3 /0" @@ -8851,28 +8832,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x3 (3)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x3 (3)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovbe st0, st4": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xd4 /0" @@ -8880,28 +8860,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x4 (4)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x4 (4)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovbe st0, st5": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xd5 /0" @@ -8909,28 +8888,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x5 (5)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x5 (5)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovbe st0, st6": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xd6 /0" @@ -8938,28 +8916,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x6 (6)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x6 (6)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovbe st0, st7": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xda 11b 0xd7 /0" @@ -8967,24 +8944,23 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov w22, #0x0", - "mov x23, #0xffffffffffffffff", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x7 (7)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov w21, #0x0", + "mov x22, #0xffffffffffffffff", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x7 (7)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovu st0, st0": { @@ -9603,223 +9579,215 @@ ] }, "fcmovnb st0, st0": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xdb 11b 0xc0 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x0 (0)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x0 (0)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnb st0, st1": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xdb 11b 0xc1 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x1 (1)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x1 (1)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnb st0, st2": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xdb 11b 0xc2 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x2 (2)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x2 (2)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnb st0, st3": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xdb 11b 0xc3 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x3 (3)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x3 (3)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnb st0, st4": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xdb 11b 0xc4 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x4 (4)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x4 (4)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnb st0, st5": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xdb 11b 0xc5 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x5 (5)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x5 (5)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnb st0, st6": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xdb 11b 0xc6 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x6 (6)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x6 (6)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnb st0, st7": { - "ExpectedInstructionCount": 18, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": [ "0xdb 11b 0xc7 /0" ], "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", - "ubfx w21, w20, #29, #1", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x7 (7)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #29, #1", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x7 (7)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovne st0, st0": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xc8 /1" @@ -9827,28 +9795,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x0 (0)", - "and w22, w22, #0x7", + "add w21, w20, #0x0 (0)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmovne st0, st1": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xc9 /1" @@ -9856,28 +9823,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x1 (1)", - "and w22, w22, #0x7", + "add w21, w20, #0x1 (1)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmovne st0, st2": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xca /1" @@ -9885,28 +9851,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x2 (2)", - "and w22, w22, #0x7", + "add w21, w20, #0x2 (2)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmovne st0, st3": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xcb /1" @@ -9914,28 +9879,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x3 (3)", - "and w22, w22, #0x7", + "add w21, w20, #0x3 (3)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmovne st0, st4": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xcc /1" @@ -9943,28 +9907,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x4 (4)", - "and w22, w22, #0x7", + "add w21, w20, #0x4 (4)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmovne st0, st5": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xcd /1" @@ -9972,28 +9935,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x5 (5)", - "and w22, w22, #0x7", + "add w21, w20, #0x5 (5)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmovne st0, st6": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xce /1" @@ -10001,28 +9963,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x6 (6)", - "and w22, w22, #0x7", + "add w21, w20, #0x6 (6)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmovne st0, st7": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xcf /1" @@ -10030,28 +9991,27 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "ldr w21, [x28, #728]", - "ubfx w22, w21, #30, #1", - "orr x20, x20, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", + "ubfx w21, w21, #30, #1", + "orr x20, x20, x21, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", "cmp x20, #0x0 (0)", - "csel x20, x22, x23, eq", + "csel x20, x21, x22, eq", "dup v2.2d, x20", "ldrb w20, [x28, #747]", - "add w22, w20, #0x7 (7)", - "and w22, w22, #0x7", + "add w21, w20, #0x7 (7)", + "and w21, w21, #0x7", "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", "add x0, x28, x20, lsl #4", - "str q2, [x0, #752]", - "str w21, [x28, #728]" + "str q2, [x0, #752]" ] }, "fcmovnbe st0, st0": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xd0 /2" @@ -10059,28 +10019,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x0 (0)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x0 (0)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnbe st0, st1": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xd1 /2" @@ -10088,28 +10047,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x1 (1)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x1 (1)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnbe st0, st2": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xd2 /2" @@ -10117,28 +10075,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x2 (2)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x2 (2)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnbe st0, st3": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xd3 /2" @@ -10146,28 +10103,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x3 (3)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x3 (3)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnbe st0, st4": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xd4 /2" @@ -10175,28 +10131,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x4 (4)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x4 (4)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnbe st0, st5": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xd5 /2" @@ -10204,28 +10159,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x5 (5)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x5 (5)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnbe st0, st6": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xd6 /2" @@ -10233,28 +10187,27 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x6 (6)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x6 (6)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnbe st0, st7": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": [ "0xdb 11b 0xd7 /2" @@ -10262,24 +10215,23 @@ "ExpectedArm64ASM": [ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", - "ubfx w22, w20, #30, #1", - "orr x21, x21, x22, lsl #6", - "mov x22, #0xffffffffffffffff", - "mov w23, #0x0", - "cmp x21, #0x0 (0)", - "csel x21, x22, x23, eq", - "dup v2.2d, x21", - "ldrb w21, [x28, #747]", - "add w22, w21, #0x7 (7)", - "and w22, w22, #0x7", - "add x0, x28, x21, lsl #4", + "ubfx w20, w20, #30, #1", + "orr x20, x21, x20, lsl #6", + "mov x21, #0xffffffffffffffff", + "mov w22, #0x0", + "cmp x20, #0x0 (0)", + "csel x20, x21, x22, eq", + "dup v2.2d, x20", + "ldrb w20, [x28, #747]", + "add w21, w20, #0x7 (7)", + "and w21, w21, #0x7", + "add x0, x28, x20, lsl #4", "ldr q3, [x0, #752]", - "add x0, x28, x22, lsl #4", + "add x0, x28, x21, lsl #4", "ldr q4, [x0, #752]", "bsl v2.16b, v4.16b, v3.16b", - "add x0, x28, x21, lsl #4", - "str q2, [x0, #752]", - "str w20, [x28, #728]" + "add x0, x28, x20, lsl #4", + "str q2, [x0, #752]" ] }, "fcmovnu st0, st0": {