diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index 483dec5f0..a85621b89 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -906,11 +906,11 @@ OrderedNode *OpDispatchBuilder::SelectCC(uint8_t OP, OrderedNode *TrueValue, Ord break; } case 0xA: { // JP - Jump if PF == 1 - SrcCond = _Select(FEXCore::IR::COND_NEQ, LoadPF(), ZeroConst, TrueValue, FalseValue); + SrcCond = _Select(FEXCore::IR::COND_EQ, LoadPFInverted(), ZeroConst, TrueValue, FalseValue); break; } case 0xB: { // JNP - Jump if PF == 0 - SrcCond = _Select(FEXCore::IR::COND_EQ, LoadPF(), ZeroConst, TrueValue, FalseValue); + SrcCond = _Select(FEXCore::IR::COND_NEQ, LoadPFInverted(), ZeroConst, TrueValue, FalseValue); break; } case 0xC: { // SF <> OF diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h index 44f8c4950..ccb77ad25 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h @@ -1387,6 +1387,7 @@ private: * @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs. * @{ */ OrderedNode *LoadPF(); + OrderedNode *LoadPFInverted(); OrderedNode *LoadAF(); void FixupAF(); void CalculatePF(OrderedNode *Res, OrderedNode *condition = nullptr); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp index 76e1ee533..80cf216c2 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp @@ -140,6 +140,7 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) { CalculateDeferredFlags(); OrderedNode *Original = _Constant(0); + bool Nonzero = false; // SF/ZF and N/Z are together on both arm64 and x86_64, so we special case that. bool GetNZ = (FlagsMask & (1 << FEXCore::X86State::RFLAG_SF_LOC)) && @@ -149,6 +150,7 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) { if (FlagsMask & (1 << FEXCore::X86State::RFLAG_CF_LOC)) { static_assert(FEXCore::X86State::RFLAG_CF_LOC == 0); Original = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC); + Nonzero = true; } for (size_t i = 0; i < FlagOffsets.size(); ++i) { @@ -174,7 +176,12 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) { else Flag = GetRFLAG(FlagOffset); - Original = _Orlshl(OpSize::i64Bit, Original, Flag, FlagOffset); + if (Nonzero) + Original = _Orlshl(OpSize::i64Bit, Original, Flag, FlagOffset); + else + Original = _Lshl(OpSize::i64Bit, Flag, _Constant(FlagOffset)); + + Nonzero = true; } // OR in the SF/ZF flags at the end, allowing the lshr to fold with the OR @@ -201,15 +208,10 @@ void OpDispatchBuilder::CalculateOF_Add(uint8_t SrcSize, OrderedNode *Res, Order SetRFLAG(AndOp1); } -OrderedNode *OpDispatchBuilder::LoadPF() { +OrderedNode *OpDispatchBuilder::LoadPFInverted() { // Read the stored byte. This is the original 8-bit result, it needs parity calculated. auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC); - // We will use the bottom bit of the popcount, set if an odd number of bits are set. - // But the x86 parity flag is supposed to be set for an even number of bits. - // Simply invert any bit of the input GPR and that will invert the bottom bit of the - PFByte = _Xor(OpSize::i32Bit, PFByte, _Constant(1)); - // Cast the input to a 32-bit FPR. Logically we only need 8-bit, but that would // generate unwanted an ubfx instruction. VPopcount will ignore the upper bits anyway. auto InputFPR = _VCastFromGPR(4, 4, PFByte); @@ -222,6 +224,10 @@ OrderedNode *OpDispatchBuilder::LoadPF() { return _And(OpSize::i64Bit, Parity, _Constant(1)); } +OrderedNode *OpDispatchBuilder::LoadPF() { + return _Xor(OpSize::i32Bit, LoadPFInverted(), _Constant(1)); +} + OrderedNode *OpDispatchBuilder::LoadAF() { // Read the stored byte. This is the XOR of the arguments. auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC); diff --git a/unittests/InstructionCountCI/Primary.json b/unittests/InstructionCountCI/Primary.json index 728e719e3..433e3bcb5 100644 --- a/unittests/InstructionCountCI/Primary.json +++ b/unittests/InstructionCountCI/Primary.json @@ -3107,11 +3107,11 @@ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", "ldrb w22, [x28, #706]", - "eor w22, w22, #0x1", "fmov s2, w22", "cnt v2.16b, v2.16b", "umov w22, v2.b[0]", "and x22, x22, #0x1", + "eor w22, w22, #0x1", "orr x21, x21, x22, lsl #2", "ldrb w22, [x28, #708]", "ldrb w23, [x28, #706]", @@ -3156,11 +3156,11 @@ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", "ldrb w22, [x28, #706]", - "eor w22, w22, #0x1", "fmov s2, w22", "cnt v2.16b, v2.16b", "umov w22, v2.b[0]", "and x22, x22, #0x1", + "eor w22, w22, #0x1", "orr x21, x21, x22, lsl #2", "ldrb w22, [x28, #708]", "ldrb w23, [x28, #706]", @@ -3278,11 +3278,11 @@ "ldr w20, [x28, #728]", "ubfx w21, w20, #29, #1", "ldrb w22, [x28, #706]", - "eor w22, w22, #0x1", "fmov s2, w22", "cnt v2.16b, v2.16b", "umov w22, v2.b[0]", "and x22, x22, #0x1", + "eor w22, w22, #0x1", "orr x21, x21, x22, lsl #2", "ldrb w22, [x28, #708]", "ldrb w23, [x28, #706]", diff --git a/unittests/InstructionCountCI/Secondary.json b/unittests/InstructionCountCI/Secondary.json index 44f133fee..6f6fc4ac3 100644 --- a/unittests/InstructionCountCI/Secondary.json +++ b/unittests/InstructionCountCI/Secondary.json @@ -622,11 +622,11 @@ "Comment": "0x0f 0x4a", "ExpectedArm64ASM": [ "ldrb w20, [x28, #706]", - "eor w20, w20, #0x1", "fmov s2, w20", "cnt v2.16b, v2.16b", "umov w20, v2.b[0]", "and x20, x20, #0x1", + "eor w20, w20, #0x1", "cmp w20, #0x0 (0)", "csel w20, w7, w4, ne", "bfxil x4, x20, #0, #16" @@ -638,11 +638,11 @@ "Comment": "0x0f 0x4a", "ExpectedArm64ASM": [ "ldrb w20, [x28, #706]", - "eor w20, w20, #0x1", "fmov s2, w20", "cnt v2.16b, v2.16b", "umov w20, v2.b[0]", "and x20, x20, #0x1", + "eor w20, w20, #0x1", "cmp w20, #0x0 (0)", "csel w4, w7, w4, ne" ] @@ -653,11 +653,11 @@ "Comment": "0x0f 0x4a", "ExpectedArm64ASM": [ "ldrb w20, [x28, #706]", - "eor w20, w20, #0x1", "fmov s2, w20", "cnt v2.16b, v2.16b", "umov w20, v2.b[0]", "and x20, x20, #0x1", + "eor w20, w20, #0x1", "cmp w20, #0x0 (0)", "csel x4, x7, x4, ne" ] @@ -668,11 +668,11 @@ "Comment": "0x0f 0x4b", "ExpectedArm64ASM": [ "ldrb w20, [x28, #706]", - "eor w20, w20, #0x1", "fmov s2, w20", "cnt v2.16b, v2.16b", "umov w20, v2.b[0]", "and x20, x20, #0x1", + "eor w20, w20, #0x1", "cmp w20, #0x0 (0)", "csel w20, w7, w4, eq", "bfxil x4, x20, #0, #16" @@ -684,11 +684,11 @@ "Comment": "0x0f 0x4b", "ExpectedArm64ASM": [ "ldrb w20, [x28, #706]", - "eor w20, w20, #0x1", "fmov s2, w20", "cnt v2.16b, v2.16b", "umov w20, v2.b[0]", "and x20, x20, #0x1", + "eor w20, w20, #0x1", "cmp w20, #0x0 (0)", "csel w4, w7, w4, eq" ] @@ -699,11 +699,11 @@ "Comment": "0x0f 0x4b", "ExpectedArm64ASM": [ "ldrb w20, [x28, #706]", - "eor w20, w20, #0x1", "fmov s2, w20", "cnt v2.16b, v2.16b", "umov w20, v2.b[0]", "and x20, x20, #0x1", + "eor w20, w20, #0x1", "cmp w20, #0x0 (0)", "csel x4, x7, x4, eq" ] @@ -1586,28 +1586,11 @@ ] }, "setpe al": { - "ExpectedInstructionCount": 9, + "ExpectedInstructionCount": 8, "Optimal": "Yes", "Comment": "0x0f 0x9a", "ExpectedArm64ASM": [ "ldrb w20, [x28, #706]", - "eor w20, w20, #0x1", - "fmov s2, w20", - "cnt v2.16b, v2.16b", - "umov w20, v2.b[0]", - "and x20, x20, #0x1", - "cmp x20, #0x0 (0)", - "cset x20, ne", - "bfxil x4, x20, #0, #8" - ] - }, - "setnp al": { - "ExpectedInstructionCount": 9, - "Optimal": "Yes", - "Comment": "0x0f 0x9b", - "ExpectedArm64ASM": [ - "ldrb w20, [x28, #706]", - "eor w20, w20, #0x1", "fmov s2, w20", "cnt v2.16b, v2.16b", "umov w20, v2.b[0]", @@ -1617,6 +1600,21 @@ "bfxil x4, x20, #0, #8" ] }, + "setnp al": { + "ExpectedInstructionCount": 8, + "Optimal": "Yes", + "Comment": "0x0f 0x9b", + "ExpectedArm64ASM": [ + "ldrb w20, [x28, #706]", + "fmov s2, w20", + "cnt v2.16b, v2.16b", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "cmp x20, #0x0 (0)", + "cset x20, ne", + "bfxil x4, x20, #0, #8" + ] + }, "setl al": { "ExpectedInstructionCount": 6, "Optimal": "No", diff --git a/unittests/InstructionCountCI/x87.json b/unittests/InstructionCountCI/x87.json index 6659f4c10..f2f89c9da 100644 --- a/unittests/InstructionCountCI/x87.json +++ b/unittests/InstructionCountCI/x87.json @@ -8516,16 +8516,15 @@ ] }, "fcmove st0, st0": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xda 11b 0xc8 /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -8544,16 +8543,15 @@ ] }, "fcmove st0, st1": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xda 11b 0xc9 /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -8572,16 +8570,15 @@ ] }, "fcmove st0, st2": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xda 11b 0xca /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -8600,16 +8597,15 @@ ] }, "fcmove st0, st3": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xda 11b 0xcb /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -8628,16 +8624,15 @@ ] }, "fcmove st0, st4": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xda 11b 0xcc /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -8656,16 +8651,15 @@ ] }, "fcmove st0, st5": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xda 11b 0xcd /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -8684,16 +8678,15 @@ ] }, "fcmove st0, st6": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xda 11b 0xce /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -8712,16 +8705,15 @@ ] }, "fcmove st0, st7": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xda 11b 0xcf /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -8964,20 +8956,19 @@ ] }, "fcmovu st0, st0": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xda 11b 0xd8 /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -8996,20 +8987,19 @@ ] }, "fcmovu st0, st1": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xda 11b 0xd9 /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -9028,20 +9018,19 @@ ] }, "fcmovu st0, st2": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xda 11b 0xda /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -9060,20 +9049,19 @@ ] }, "fcmovu st0, st3": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xda 11b 0xdb /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -9092,20 +9080,19 @@ ] }, "fcmovu st0, st4": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xda 11b 0xdc /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -9124,20 +9111,19 @@ ] }, "fcmovu st0, st5": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xda 11b 0xdd /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -9156,20 +9142,19 @@ ] }, "fcmovu st0, st6": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xda 11b 0xde /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -9188,20 +9173,19 @@ ] }, "fcmovu st0, st7": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xda 11b 0xdf /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov w21, #0x0", "mov x22, #0xffffffffffffffff", "cmp x20, #0x0 (0)", @@ -9787,16 +9771,15 @@ ] }, "fcmovne st0, st0": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xdb 11b 0xc8 /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -9815,16 +9798,15 @@ ] }, "fcmovne st0, st1": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xdb 11b 0xc9 /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -9843,16 +9825,15 @@ ] }, "fcmovne st0, st2": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xdb 11b 0xca /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -9871,16 +9852,15 @@ ] }, "fcmovne st0, st3": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xdb 11b 0xcb /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -9899,16 +9879,15 @@ ] }, "fcmovne st0, st4": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xdb 11b 0xcc /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -9927,16 +9906,15 @@ ] }, "fcmovne st0, st5": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xdb 11b 0xcd /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -9955,16 +9933,15 @@ ] }, "fcmovne st0, st6": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xdb 11b 0xce /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -9983,16 +9960,15 @@ ] }, "fcmovne st0, st7": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": [ "0xdb 11b 0xcf /1" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldr w21, [x28, #728]", - "ubfx w21, w21, #30, #1", - "orr x20, x20, x21, lsl #6", + "ldr w20, [x28, #728]", + "ubfx w20, w20, #30, #1", + "lsl x20, x20, #6", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -10235,20 +10211,19 @@ ] }, "fcmovnu st0, st0": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xdb 11b 0xd8 /3" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -10267,20 +10242,19 @@ ] }, "fcmovnu st0, st1": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xdb 11b 0xd9 /3" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -10299,20 +10273,19 @@ ] }, "fcmovnu st0, st2": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xdb 11b 0xda /3" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -10331,20 +10304,19 @@ ] }, "fcmovnu st0, st3": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xdb 11b 0xdb /3" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -10363,20 +10335,19 @@ ] }, "fcmovnu st0, st4": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xdb 11b 0xdc /3" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -10395,20 +10366,19 @@ ] }, "fcmovnu st0, st5": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xdb 11b 0xdd /3" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -10427,20 +10397,19 @@ ] }, "fcmovnu st0, st6": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xdb 11b 0xde /3" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)", @@ -10459,20 +10428,19 @@ ] }, "fcmovnu st0, st7": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ "0xdb 11b 0xdf /3" ], "ExpectedArm64ASM": [ - "mov w20, #0x0", - "ldrb w21, [x28, #706]", - "eor w21, w21, #0x1", - "fmov s2, w21", + "ldrb w20, [x28, #706]", + "fmov s2, w20", "cnt v2.16b, v2.16b", - "umov w21, v2.b[0]", - "and x21, x21, #0x1", - "orr x20, x20, x21, lsl #2", + "umov w20, v2.b[0]", + "and x20, x20, #0x1", + "eor w20, w20, #0x1", + "lsl x20, x20, #2", "mov x21, #0xffffffffffffffff", "mov w22, #0x0", "cmp x20, #0x0 (0)",