Merge pull request #3116 from alyssarosenzweig/minor/flag-opts

Minor/flag opts
This commit is contained in:
Ryan Houdek authored and GitHub committed 2023-09-18 10:28:51 -07:00
commit 8b523082af
6 files changed
+217 -244

No files matched your search

@@ -906,11 +906,11 @@ OrderedNode *OpDispatchBuilder::SelectCC(uint8_t OP, OrderedNode *TrueValue, Ord
break;
}
case 0xA: { // JP - Jump if PF == 1
SrcCond = _Select(FEXCore::IR::COND_NEQ, LoadPF(), ZeroConst, TrueValue, FalseValue);
SrcCond = _Select(FEXCore::IR::COND_EQ, LoadPFInverted(), ZeroConst, TrueValue, FalseValue);
break;
}
case 0xB: { // JNP - Jump if PF == 0
SrcCond = _Select(FEXCore::IR::COND_EQ, LoadPF(), ZeroConst, TrueValue, FalseValue);
SrcCond = _Select(FEXCore::IR::COND_NEQ, LoadPFInverted(), ZeroConst, TrueValue, FalseValue);
break;
}
case 0xC: { // SF <> OF
@@ -1387,6 +1387,7 @@ private:
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
* @{ */
OrderedNode *LoadPF();
OrderedNode *LoadPFInverted();
OrderedNode *LoadAF();
void FixupAF();
void CalculatePF(OrderedNode *Res, OrderedNode *condition = nullptr);
@@ -140,6 +140,7 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
CalculateDeferredFlags();
OrderedNode *Original = _Constant(0);
bool Nonzero = false;
// SF/ZF and N/Z are together on both arm64 and x86_64, so we special case that.
bool GetNZ = (FlagsMask & (1 << FEXCore::X86State::RFLAG_SF_LOC)) &&
@@ -149,6 +150,7 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_CF_LOC)) {
static_assert(FEXCore::X86State::RFLAG_CF_LOC == 0);
Original = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
Nonzero = true;
}
for (size_t i = 0; i < FlagOffsets.size(); ++i) {
@@ -174,7 +176,12 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
else
Flag = GetRFLAG(FlagOffset);
Original = _Orlshl(OpSize::i64Bit, Original, Flag, FlagOffset);
if (Nonzero)
Original = _Orlshl(OpSize::i64Bit, Original, Flag, FlagOffset);
else
Original = _Lshl(OpSize::i64Bit, Flag, _Constant(FlagOffset));
Nonzero = true;
}
// OR in the SF/ZF flags at the end, allowing the lshr to fold with the OR
@@ -201,15 +208,10 @@ void OpDispatchBuilder::CalculateOF_Add(uint8_t SrcSize, OrderedNode *Res, Order
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(AndOp1);
}
OrderedNode *OpDispatchBuilder::LoadPF() {
OrderedNode *OpDispatchBuilder::LoadPFInverted() {
// Read the stored byte. This is the original 8-bit result, it needs parity calculated.
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC);
// We will use the bottom bit of the popcount, set if an odd number of bits are set.
// But the x86 parity flag is supposed to be set for an even number of bits.
// Simply invert any bit of the input GPR and that will invert the bottom bit of the
PFByte = _Xor(OpSize::i32Bit, PFByte, _Constant(1));
// Cast the input to a 32-bit FPR. Logically we only need 8-bit, but that would
// generate unwanted an ubfx instruction. VPopcount will ignore the upper bits anyway.
auto InputFPR = _VCastFromGPR(4, 4, PFByte);
@@ -222,6 +224,10 @@ OrderedNode *OpDispatchBuilder::LoadPF() {
return _And(OpSize::i64Bit, Parity, _Constant(1));
}
OrderedNode *OpDispatchBuilder::LoadPF() {
return _Xor(OpSize::i32Bit, LoadPFInverted(), _Constant(1));
}
OrderedNode *OpDispatchBuilder::LoadAF() {
// Read the stored byte. This is the XOR of the arguments.
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
+3 -3
View File
@@ -3107,11 +3107,11 @@
"ldr w20, [x28, #728]",
"ubfx w21, w20, #29, #1",
"ldrb w22, [x28, #706]",
"eor w22, w22, #0x1",
"fmov s2, w22",
"cnt v2.16b, v2.16b",
"umov w22, v2.b[0]",
"and x22, x22, #0x1",
"eor w22, w22, #0x1",
"orr x21, x21, x22, lsl #2",
"ldrb w22, [x28, #708]",
"ldrb w23, [x28, #706]",
@@ -3156,11 +3156,11 @@
"ldr w20, [x28, #728]",
"ubfx w21, w20, #29, #1",
"ldrb w22, [x28, #706]",
"eor w22, w22, #0x1",
"fmov s2, w22",
"cnt v2.16b, v2.16b",
"umov w22, v2.b[0]",
"and x22, x22, #0x1",
"eor w22, w22, #0x1",
"orr x21, x21, x22, lsl #2",
"ldrb w22, [x28, #708]",
"ldrb w23, [x28, #706]",
@@ -3278,11 +3278,11 @@
"ldr w20, [x28, #728]",
"ubfx w21, w20, #29, #1",
"ldrb w22, [x28, #706]",
"eor w22, w22, #0x1",
"fmov s2, w22",
"cnt v2.16b, v2.16b",
"umov w22, v2.b[0]",
"and x22, x22, #0x1",
"eor w22, w22, #0x1",
"orr x21, x21, x22, lsl #2",
"ldrb w22, [x28, #708]",
"ldrb w23, [x28, #706]",
+22 -24
View File
@@ -622,11 +622,11 @@
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"eor w20, w20, #0x1",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"cmp w20, #0x0 (0)",
"csel w20, w7, w4, ne",
"bfxil x4, x20, #0, #16"
@@ -638,11 +638,11 @@
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"eor w20, w20, #0x1",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"cmp w20, #0x0 (0)",
"csel w4, w7, w4, ne"
]
@@ -653,11 +653,11 @@
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"eor w20, w20, #0x1",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"cmp w20, #0x0 (0)",
"csel x4, x7, x4, ne"
]
@@ -668,11 +668,11 @@
"Comment": "0x0f 0x4b",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"eor w20, w20, #0x1",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"cmp w20, #0x0 (0)",
"csel w20, w7, w4, eq",
"bfxil x4, x20, #0, #16"
@@ -684,11 +684,11 @@
"Comment": "0x0f 0x4b",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"eor w20, w20, #0x1",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"cmp w20, #0x0 (0)",
"csel w4, w7, w4, eq"
]
@@ -699,11 +699,11 @@
"Comment": "0x0f 0x4b",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"eor w20, w20, #0x1",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"cmp w20, #0x0 (0)",
"csel x4, x7, x4, eq"
]
@@ -1586,28 +1586,11 @@
]
},
"setpe al": {
"ExpectedInstructionCount": 9,
"ExpectedInstructionCount": 8,
"Optimal": "Yes",
"Comment": "0x0f 0x9a",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"eor w20, w20, #0x1",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"cmp x20, #0x0 (0)",
"cset x20, ne",
"bfxil x4, x20, #0, #8"
]
},
"setnp al": {
"ExpectedInstructionCount": 9,
"Optimal": "Yes",
"Comment": "0x0f 0x9b",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"eor w20, w20, #0x1",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
@@ -1617,6 +1600,21 @@
"bfxil x4, x20, #0, #8"
]
},
"setnp al": {
"ExpectedInstructionCount": 8,
"Optimal": "Yes",
"Comment": "0x0f 0x9b",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"cmp x20, #0x0 (0)",
"cset x20, ne",
"bfxil x4, x20, #0, #8"
]
},
"setl al": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
+176 -208
View File
@@ -8516,16 +8516,15 @@
]
},
"fcmove st0, st0": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xda 11b 0xc8 /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -8544,16 +8543,15 @@
]
},
"fcmove st0, st1": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xda 11b 0xc9 /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -8572,16 +8570,15 @@
]
},
"fcmove st0, st2": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xda 11b 0xca /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -8600,16 +8597,15 @@
]
},
"fcmove st0, st3": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xda 11b 0xcb /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -8628,16 +8624,15 @@
]
},
"fcmove st0, st4": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xda 11b 0xcc /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -8656,16 +8651,15 @@
]
},
"fcmove st0, st5": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xda 11b 0xcd /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -8684,16 +8678,15 @@
]
},
"fcmove st0, st6": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xda 11b 0xce /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -8712,16 +8705,15 @@
]
},
"fcmove st0, st7": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xda 11b 0xcf /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -8964,20 +8956,19 @@
]
},
"fcmovu st0, st0": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xda 11b 0xd8 /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -8996,20 +8987,19 @@
]
},
"fcmovu st0, st1": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xda 11b 0xd9 /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -9028,20 +9018,19 @@
]
},
"fcmovu st0, st2": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xda 11b 0xda /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -9060,20 +9049,19 @@
]
},
"fcmovu st0, st3": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xda 11b 0xdb /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -9092,20 +9080,19 @@
]
},
"fcmovu st0, st4": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xda 11b 0xdc /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -9124,20 +9111,19 @@
]
},
"fcmovu st0, st5": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xda 11b 0xdd /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -9156,20 +9142,19 @@
]
},
"fcmovu st0, st6": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xda 11b 0xde /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -9188,20 +9173,19 @@
]
},
"fcmovu st0, st7": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xda 11b 0xdf /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov w21, #0x0",
"mov x22, #0xffffffffffffffff",
"cmp x20, #0x0 (0)",
@@ -9787,16 +9771,15 @@
]
},
"fcmovne st0, st0": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xc8 /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -9815,16 +9798,15 @@
]
},
"fcmovne st0, st1": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xc9 /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -9843,16 +9825,15 @@
]
},
"fcmovne st0, st2": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xca /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -9871,16 +9852,15 @@
]
},
"fcmovne st0, st3": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xcb /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -9899,16 +9879,15 @@
]
},
"fcmovne st0, st4": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xcc /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -9927,16 +9906,15 @@
]
},
"fcmovne st0, st5": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xcd /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -9955,16 +9933,15 @@
]
},
"fcmovne st0, st6": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xce /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -9983,16 +9960,15 @@
]
},
"fcmovne st0, st7": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 18,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xcf /1"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldr w21, [x28, #728]",
"ubfx w21, w21, #30, #1",
"orr x20, x20, x21, lsl #6",
"ldr w20, [x28, #728]",
"ubfx w20, w20, #30, #1",
"lsl x20, x20, #6",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -10235,20 +10211,19 @@
]
},
"fcmovnu st0, st0": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xd8 /3"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -10267,20 +10242,19 @@
]
},
"fcmovnu st0, st1": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xd9 /3"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -10299,20 +10273,19 @@
]
},
"fcmovnu st0, st2": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xda /3"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -10331,20 +10304,19 @@
]
},
"fcmovnu st0, st3": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xdb /3"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -10363,20 +10335,19 @@
]
},
"fcmovnu st0, st4": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xdc /3"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -10395,20 +10366,19 @@
]
},
"fcmovnu st0, st5": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xdd /3"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -10427,20 +10397,19 @@
]
},
"fcmovnu st0, st6": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xde /3"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",
@@ -10459,20 +10428,19 @@
]
},
"fcmovnu st0, st7": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": [
"0xdb 11b 0xdf /3"
],
"ExpectedArm64ASM": [
"mov w20, #0x0",
"ldrb w21, [x28, #706]",
"eor w21, w21, #0x1",
"fmov s2, w21",
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w21, v2.b[0]",
"and x21, x21, #0x1",
"orr x20, x20, x21, lsl #2",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"lsl x20, x20, #2",
"mov x21, #0xffffffffffffffff",
"mov w22, #0x0",
"cmp x20, #0x0 (0)",