From 5a590a9b113b29fb206d1085f02beeb79db8ab9d Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Tue, 27 Aug 2024 09:37:24 -0400 Subject: [PATCH 01/19] InstCountCI: add 8-bit test cases Signed-off-by: Alyssa Rosenzweig --- .../InstructionCountCI/FlagM/FlagOpts.json | 54 +++++++++++++++++++ 1 file changed, 54 insertions(+) diff --git a/unittests/InstructionCountCI/FlagM/FlagOpts.json b/unittests/InstructionCountCI/FlagM/FlagOpts.json index 28551c669..ea7a2c7a4 100644 --- a/unittests/InstructionCountCI/FlagM/FlagOpts.json +++ b/unittests/InstructionCountCI/FlagM/FlagOpts.json @@ -300,6 +300,60 @@ "mov x26, x7" ] }, + "Test use only zero - self 16-bit": { + "x86InstructionCount": 3, + "ExpectedInstructionCount": 6, + "x86Insts": [ + "test ax, ax", + "setz al", + "test cl, cl" + ], + "ExpectedArm64ASM": [ + "cmn wzr, w4, lsl #16", + "cset x20, eq", + "bfxil x4, x20, #0, #8", + "cmn wzr, w7, lsl #24", + "cfinv", + "mov x26, x7" + ] + }, + "Test use only zero - non constant 16-bit": { + "x86InstructionCount": 3, + "ExpectedInstructionCount": 7, + "x86Insts": [ + "test ax, bx", + "setz al", + "test cl, cl" + ], + "ExpectedArm64ASM": [ + "and w20, w4, w6", + "cmn wzr, w20, lsl #16", + "cset x21, eq", + "bfxil x4, x21, #0, #8", + "cmn wzr, w7, lsl #24", + "cfinv", + "mov x26, x7" + ] + }, + "Test use only zero - constant 8-bit": { + "x86InstructionCount": 3, + "ExpectedInstructionCount": 8, + "x86Insts": [ + "test al, 137", + "setnz al", + "test cl, cl" + ], + "ExpectedArm64ASM": [ + "mov w20, #0x89", + "and w20, w4, w20", + "cmn wzr, w20, lsl #24", + "cset x21, ne", + "bfxil x4, x21, #0, #8", + "cmn wzr, w7, lsl #24", + "cfinv", + "mov x26, x7" + ] + }, "Dead cmpxchg flags": { "x86InstructionCount": 2, "ExpectedInstructionCount": 10, From 9fe7bc5818862c224a68bf848591d7a2a6bbb336 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Tue, 3 Sep 2024 07:04:38 -0400 Subject: [PATCH 02/19] InstCountCI: add ucomiss+pf opt case Signed-off-by: Alyssa Rosenzweig --- .../InstructionCountCI/FlagM/FlagOpts.json | 19 +++++++++++++++++++ 1 file changed, 19 insertions(+) diff --git a/unittests/InstructionCountCI/FlagM/FlagOpts.json b/unittests/InstructionCountCI/FlagM/FlagOpts.json index ea7a2c7a4..d8919c5fc 100644 --- a/unittests/InstructionCountCI/FlagM/FlagOpts.json +++ b/unittests/InstructionCountCI/FlagM/FlagOpts.json @@ -300,6 +300,25 @@ "mov x26, x7" ] }, + "UCOMISS use only PF": { + "x86InstructionCount": 3, + "ExpectedInstructionCount": 8, + "x86Insts": [ + "ucomiss xmm0, xmm1", + "setnp cl", + "test rax, rax" + ], + "ExpectedArm64ASM": [ + "fcmp s16, s17", + "cset w26, vc", + "eor w20, w26, w26, lsr #4", + "eor w20, w20, w20, lsr #2", + "eor w20, w20, w20, lsr #1", + "and x20, x20, #0x1", + "bfxil x7, x20, #0, #8", + "subs x26, x4, #0x0 (0)" + ] + }, "Test use only zero - self 16-bit": { "x86InstructionCount": 3, "ExpectedInstructionCount": 6, From 8d9f19bd73ed2848c028f141f62449a0eabc93d4 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Tue, 27 Aug 2024 09:34:29 -0400 Subject: [PATCH 03/19] IR: add TestZ op more optimized than TestNZ if we don't care about the sign bit. Signed-off-by: Alyssa Rosenzweig --- .../Interface/Core/JIT/Arm64/ALUOps.cpp | 24 +++++++++++++++++++ FEXCore/Source/Interface/IR/IR.json | 5 ++++ 2 files changed, 29 insertions(+) diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp b/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp index 30268973c..815dc91c7 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp @@ -190,6 +190,30 @@ DEF_OP(TestNZ) { } } +DEF_OP(TestZ) { + auto Op = IROp->C(); + LOGMAN_THROW_AA_FMT(IROp->Size < 4, "TestNZ used at higher sizes"); + const auto EmitSize = ARMEmitter::Size::i32Bit; + + uint64_t Const; + uint64_t Mask = IROp->Size == 8 ? ~0ULL : ((1ull << (IROp->Size * 8)) - 1); + auto Src1 = GetReg(Op->Src1.ID()); + + if (IsInlineConstant(Op->Src2, &Const)) { + // We can promote 8/16-bit tests to 32-bit since the constant is masked. + LOGMAN_THROW_AA_FMT(!(Const & ~Mask), "constant is already masked"); + tst(EmitSize, Src1, Const); + } else { + const auto Src2 = GetReg(Op->Src2.ID()); + if (Src1 == Src2) { + tst(EmitSize, Src1 /* Src2 */, Mask); + } else { + and_(EmitSize, TMP1, Src1, Src2); + tst(EmitSize, TMP1, Mask); + } + } +} + DEF_OP(SubShift) { auto Op = IROp->C(); diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index 63c49dbde..ee0c60c4c 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -1331,6 +1331,11 @@ "DestSize": "Size", "HasSideEffects": true }, + "TestZ OpSize:#Size, GPR:$Src1, GPR:$Src2": { + "Desc": ["Set NZCV for the binary AND of two GPRs, setting Z accordingly and zeroing C and V. N is undefined."], + "DestSize": "Size", + "HasSideEffects": true + }, "GPR = Lshl OpSize:#Size, GPR:$Src1, GPR:$Src2": { "Desc": ["Integer logical shift left" ], From 4eb0948451f35a4aa9991d0a22b30b34cc845093 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Wed, 28 Aug 2024 13:06:48 -0400 Subject: [PATCH 04/19] IR: push down AXFLAG lowering so we can get the new axflag optimizations on billy's x13s. Signed-off-by: Alyssa Rosenzweig --- .../Interface/Core/JIT/Arm64/ALUOps.cpp | 19 +++++++++++++++-- .../Source/Interface/Core/OpcodeDispatcher.h | 21 ++++--------------- FEXCore/Source/Interface/IR/IR.json | 5 +++-- 3 files changed, 24 insertions(+), 21 deletions(-) diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp b/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp index 815dc91c7..940035646 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp @@ -5,6 +5,7 @@ tags: backend|arm64 $end_info$ */ +#include "CodeEmitter/Emitter.h" #include "FEXCore/IR/IR.h" #include "Interface/Context/Context.h" #include "Interface/Core/JIT/Arm64/JITClass.h" @@ -296,8 +297,22 @@ DEF_OP(SetSmallNZV) { } DEF_OP(AXFlag) { - LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM2, "Unsupported flagm2 op"); - axflag(); + if (CTX->HostFeatures.SupportsFlagM2) { + axflag(); + } else { + // AXFLAG is defined in the Arm spec as + // + // gt: nzCv -> nzCv + // lt: Nzcv -> nzcv <==> 1 + 0 + // eq: nZCv -> nZCv <==> 1 + (~0) + // un: nzCV -> nZcv <==> 0 + 0 + // + // For the latter 3 cases, we therefore get the right NZCV by adding V_inv + // to (eq ? ~0 : 0). The remaining case is forced with ccmn. + auto V_inv = GetReg(IROp->Args[0].ID()); + csetm(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Condition::CC_EQ); + ccmn(ARMEmitter::Size::i64Bit, V_inv, TMP1, ARMEmitter::StatusFlags {0x2} /* nzCv */, ARMEmitter::Condition::CC_LE); + } } DEF_OP(CondAddNZCV) { diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h index 117345caf..bd09ee63d 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h @@ -2106,22 +2106,9 @@ private: // Convert NZCV from the Arm representation to an eXternal representation // that's totally not a euphemism for x86, nuh-uh. But maps to exactly we // need, what a coincidence! - if (CTX->HostFeatures.SupportsFlagM2) { - _AXFlag(); - } else { - // AXFLAG is defined in the Arm spec as - // - // gt: nzCv -> nzCv - // lt: Nzcv -> nzcv <==> 1 + 0 - // eq: nZCv -> nZCv <==> 1 + (~0) - // un: nzCV -> nZcv <==> 0 + 0 - // - // For the latter 3 cases, we therefore get the right NZCV by adding V_inv - // to (eq ? ~0 : 0). The remaining case is forced with ccmn. - Ref Eq = NZCVSelect(OpSize::i64Bit, {COND_EQ}, _Constant(~0ull), _Constant(0)); - _CondAddNZCV(OpSize::i64Bit, V_inv, Eq, {COND_FLEU}, 0x2 /* nzCv */); - } - + // + // Our AXFlag emulation on FlagM2-less systems needs V_inv passed. + _AXFlag(CTX->HostFeatures.SupportsFlagM2 ? Invalid() : V_inv); PossiblySetNZCVBits = ~0; CFInverted = true; } @@ -2135,7 +2122,7 @@ private: LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp"); // Convert to x86 flags, saves us from or'ing after. - _AXFlag(); + _AXFlag(Invalid()); PossiblySetNZCVBits = ~0; CFInverted = true; diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index ee0c60c4c..b17c23f89 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -1139,8 +1139,9 @@ "Desc": ["Invert carry flag in NZCV"], "HasSideEffects": true }, - "AXFlag": { - "Desc": ["After an FCmp, converts NZCV flags from the Arm format to a mysterious eXternal format"], + "AXFlag GPR:$V_inv": { + "Desc": ["After an FCmp, converts NZCV flags from the Arm format to a mysterious eXternal format", + "On FlagM2-less platforms, takes the inverted 1/0 overflow flag"], "HasSideEffects": true }, "RmifNZCV GPR:$Src, u8:$Rotate, u8:$Mask": { From e9ab514962dc99538ccac66acc95c4683d419e77 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Wed, 28 Aug 2024 17:01:15 -0400 Subject: [PATCH 05/19] IR: push parity evaluation down so we can optimize it globally Signed-off-by: Alyssa Rosenzweig --- .../Interface/Core/JIT/Arm64/ALUOps.cpp | 21 +++++++++++++++++++ .../Interface/Core/OpcodeDispatcher.cpp | 19 ++++++++--------- .../Source/Interface/Core/OpcodeDispatcher.h | 3 ++- .../Interface/Core/OpcodeDispatcher/Flags.cpp | 21 ++++--------------- FEXCore/Source/Interface/IR/IR.json | 4 ++++ 5 files changed, 40 insertions(+), 28 deletions(-) diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp b/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp index 940035646..c1511f141 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp @@ -315,6 +315,27 @@ DEF_OP(AXFlag) { } } +DEF_OP(Parity) { + auto Op = IROp->C(); + auto Raw = GetReg(Op->Raw.ID()); + auto Dest = GetReg(Node); + + // Cascade to calculate parity of bottom 8-bits to bottom bit. + eor(ARMEmitter::Size::i32Bit, TMP1, Raw, Raw, ARMEmitter::ShiftType::LSR, 4); + eor(ARMEmitter::Size::i32Bit, TMP1, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 2); + + if (Op->Invert) { + eon(ARMEmitter::Size::i32Bit, Dest, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 1); + } else { + eor(ARMEmitter::Size::i32Bit, Dest, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 1); + } + + // The above sequence leaves garbage in the upper bits. + if (Op->Mask) { + and_(ARMEmitter::Size::i32Bit, Dest, Dest, 1); + } +} + DEF_OP(CondAddNZCV) { auto Op = IROp->C(); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index 4bdd94901..8ac4c5587 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -598,18 +598,19 @@ void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) { ExitFunction(JMPPCOffset); // If we get here then leave the function now } -Ref OpDispatchBuilder::SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue) { +Ref OpDispatchBuilder::SelectPF(bool Invert, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue) { uint64_t TrueConst, FalseConst; if (IsValueConstant(WrapNode(TrueValue), &TrueConst) && IsValueConstant(WrapNode(FalseValue), &FalseConst) && FalseConst == 0) { if (TrueConst == 1) { - return _And(ResultSize, Cmp, _Constant(1)); + return LoadPFRaw(true, Invert); } else if (TrueConst == 0xffffffff) { - return _Sbfe(OpSize::i32Bit, 1, 0, Cmp); + return _Sbfe(OpSize::i32Bit, 1, 0, LoadPFRaw(false, Invert)); } else if (TrueConst == 0xffffffffffffffffull) { - return _Sbfe(OpSize::i64Bit, 1, 0, Cmp); + return _Sbfe(OpSize::i64Bit, 1, 0, LoadPFRaw(false, Invert)); } } + Ref Cmp = LoadPFRaw(false, Invert); SaveNZCV(); // Because we're only clobbering NZCV internally, we ignore all carry flag @@ -681,12 +682,10 @@ Ref OpDispatchBuilder::SelectCC(uint8_t OP, IR::OpSize ResultSize, Ref TrueValue } switch (OP) { - case 0xA: { // JP - Jump if PF == 1 - // Raw value contains inverted PF in bottom bit - return SelectBit(LoadPFRaw(true), ResultSize, TrueValue, FalseValue); - } + case 0xA: // JP - Jump if PF == 1 case 0xB: { // JNP - Jump if PF == 0 - return SelectBit(LoadPFRaw(false), ResultSize, TrueValue, FalseValue); + // Raw value contains inverted PF in bottom bit + return SelectPF(OP == 0xA, ResultSize, TrueValue, FalseValue); } default: LOGMAN_MSG_A_FMT("Unknown CC Op: 0x{:x}\n", OP); return nullptr; } @@ -759,7 +758,7 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) { auto [Complex, SimpleCond] = DecodeNZCVCondition(OP); if (Complex) { LOGMAN_THROW_AA_FMT(OP == 0xA || OP == 0xB, "only PF left"); - CondJump_ = CondJumpBit(LoadPFRaw(false), 0, OP == 0xB); + CondJump_ = CondJumpBit(LoadPFRaw(false, false), 0, OP == 0xB); } else { CondJump_ = CondJumpNZCV(SimpleCond); } diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h index bd09ee63d..cc020ba7c 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h @@ -2308,7 +2308,8 @@ private: /** * @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs. * @{ */ - Ref LoadPFRaw(bool Invert); + Ref LoadPFRaw(bool Mask, bool Invert); + Ref SelectPF(bool Invert, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue); Ref LoadAF(); void FixupAF(); void SetAFAndFixup(Ref AF); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp index 1afef3bce..7bc9e4328 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp @@ -117,7 +117,7 @@ Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) { // instead. if (FlagsMask & (1 << FEXCore::X86State::RFLAG_PF_RAW_LOC)) { // Set every bit except the bottommost. - auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(false), _Constant(~1ull)); + auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(false, false), _Constant(~1ull)); // Rotate the bottom bit to the appropriate location for PF, so we get // something like 111P1111. Then invert that to get 000p0000. Then OR that @@ -178,22 +178,9 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2 SetRFLAG(Anded, SrcSize * 8 - 1, true); } -Ref OpDispatchBuilder::LoadPFRaw(bool Invert) { - // Read the stored byte. This is the original result (up to 64-bits), it needs - // parity calculated. - auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC); - - // Cascade to calculate parity of bottom 8-bits to bottom bit. - Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 4); - Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 2); - - if (Invert) { - Result = _XornShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1); - } else { - Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1); - } - - return Result; +Ref OpDispatchBuilder::LoadPFRaw(bool Mask, bool Invert) { + // Evaluate parity on the deferred raw value. + return _Parity(GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC), Mask, Invert); } Ref OpDispatchBuilder::LoadAF() { diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index b17c23f89..6f36e40ee 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -1144,6 +1144,10 @@ "On FlagM2-less platforms, takes the inverted 1/0 overflow flag"], "HasSideEffects": true }, + "GPR = Parity GPR:$Raw, i1:$Mask, i1:$Invert": { + "Desc": ["Calculates PF"], + "DestSize": "4" + }, "RmifNZCV GPR:$Src, u8:$Rotate, u8:$Mask": { "Desc": ["Rotate, mask, and insert into NZCV on FlagM platforms"], "HasSideEffects": true From 8745455a5b6a653107fc9b5db5584c36917d1744 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 6 Sep 2024 10:56:30 -0400 Subject: [PATCH 06/19] IR: track whether parity is read so we can gate optimizations efficiently Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp | 3 +++ FEXCore/Source/Interface/IR/IR.json | 2 +- 2 files changed, 4 insertions(+), 1 deletion(-) diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp index 7bc9e4328..3d7992385 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp @@ -179,6 +179,9 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2 } Ref OpDispatchBuilder::LoadPFRaw(bool Mask, bool Invert) { + // Most blocks do not read parity, so PF optimization is gated on this flag. + CurrentHeader->ReadsParity = true; + // Evaluate parity on the deferred raw value. return _Parity(GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC), Mask, Invert); } diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index 6f36e40ee..f9711fa5a 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -168,7 +168,7 @@ "SwitchGen": false, "JITDispatchOverride": "NoOp" }, - "IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions, i1:$HasX87{false}": { + "IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions, i1:$HasX87{false}, i1:$ReadsParity{false}": { "SwitchGen": false, "JITDispatchOverride": "NoOp" }, From 50a3ca0d6dfaa715281c8f791d885085fcf30cc4 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Sat, 7 Sep 2024 08:58:12 -0400 Subject: [PATCH 07/19] IR: introduce dedicated PF/AF instructions this makes reasoning about them a little easier, e.g. for flags. about 1% win in nodejs. Signed-off-by: Alyssa Rosenzweig --- .../Interface/Core/JIT/Arm64/MemoryOps.cpp | 54 ++++++++++++++++--- .../Source/Interface/Core/OpcodeDispatcher.h | 10 +++- FEXCore/Source/Interface/IR/IR.json | 24 ++++++++- .../RedundantFlagCalculationElimination.cpp | 36 +++---------- .../IR/Passes/RegisterAllocationPass.cpp | 17 +++--- 5 files changed, 97 insertions(+), 44 deletions(-) diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/MemoryOps.cpp b/FEXCore/Source/Interface/Core/JIT/Arm64/MemoryOps.cpp index 3560ce741..fc7292287 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/MemoryOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/MemoryOps.cpp @@ -136,12 +136,8 @@ DEF_OP(LoadRegister) { const auto OpSize = IROp->Size; if (Op->Class == IR::GPRClass) { - unsigned Reg = Op->Reg == Core::CPUState::PF_AS_GREG ? (StaticRegisters.size() - 2) : - Op->Reg == Core::CPUState::AF_AS_GREG ? (StaticRegisters.size() - 1) : - Op->Reg; - - LOGMAN_THROW_A_FMT(Reg < StaticRegisters.size(), "out of range reg"); - const auto reg = StaticRegisters[Reg]; + LOGMAN_THROW_A_FMT(Op->Reg < StaticRegisters.size(), "out of range reg"); + const auto reg = StaticRegisters[Op->Reg]; if (GetReg(Node).Idx() != reg.Idx()) { if (OpSize == 4) { @@ -170,6 +166,30 @@ DEF_OP(LoadRegister) { } } +DEF_OP(LoadPF) { + const auto reg = StaticRegisters[StaticRegisters.size() - 2]; + + if (GetReg(Node).Idx() != reg.Idx()) { + if (IROp->Size == 4) { + mov(GetReg(Node).W(), reg.W()); + } else { + mov(GetReg(Node).X(), reg.X()); + } + } +} + +DEF_OP(LoadAF) { + const auto reg = StaticRegisters[StaticRegisters.size() - 1]; + + if (GetReg(Node).Idx() != reg.Idx()) { + if (IROp->Size == 4) { + mov(GetReg(Node).W(), reg.W()); + } else { + mov(GetReg(Node).X(), reg.X()); + } + } +} + DEF_OP(StoreRegister) { const auto Op = IROp->C(); @@ -206,6 +226,28 @@ DEF_OP(StoreRegister) { } } +DEF_OP(StorePF) { + const auto Op = IROp->C(); + const auto reg = StaticRegisters[StaticRegisters.size() - 2]; + const auto Src = GetReg(Op->Value.ID()); + + if (Src.Idx() != reg.Idx()) { + // Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode. + mov(ARMEmitter::Size::i64Bit, reg, Src); + } +} + +DEF_OP(StoreAF) { + const auto Op = IROp->C(); + const auto reg = StaticRegisters[StaticRegisters.size() - 1]; + const auto Src = GetReg(Op->Value.ID()); + + if (Src.Idx() != reg.Idx()) { + // Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode. + mov(ARMEmitter::Size::i64Bit, reg, Src); + } +} + DEF_OP(LoadContextIndexed) { const auto Op = IROp->C(); const auto OpSize = IROp->Size; diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h index cc020ba7c..40f14fb3b 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h @@ -1236,8 +1236,12 @@ public: uint32_t Index = 63 - std::countl_zero(Bits); Ref Value = RegCache.Value[Index]; - if (Index >= GPR0Index && Index <= AFIndex) { + if (Index >= GPR0Index && Index <= GPR15Index) { _StoreRegister(Value, Index - GPR0Index, GPRClass, GPRSize); + } else if (Index == PFIndex) { + _StorePF(Value, GPRSize); + } else if (Index == AFIndex) { + _StoreAF(Value, GPRSize); } else if (Index >= FPR0Index && Index <= FPR15Index) { _StoreRegister(Value, Index - FPR0Index, FPRClass, VectorSize); } else if (Index == DFIndex) { @@ -1896,6 +1900,10 @@ private: if (Size == 8) { RegCache.Partial |= Bit; } + } else if (Index == PFIndex) { + RegCache.Value[Index] = _LoadPF(Size); + } else if (Index == AFIndex) { + RegCache.Value[Index] = _LoadAF(Size); } else { RegCache.Value[Index] = _LoadRegister(Offset, RegClass, Size); } diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index f9711fa5a..c061f407d 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -366,14 +366,36 @@ "DestSize": "Size" }, + "GPR = LoadPF u8:#Size": { + "Desc": ["Loads raw PF"], + "DestSize": "Size" + }, + + "GPR = LoadAF u8:#Size": { + "Desc": ["Loads raw PF"], + "DestSize": "Size" + }, + "StoreRegister SSA:$Value, u32:$Reg, RegisterClass:$Class, u8:#Size": { - "HasSideEffects": true, + "HasSideEffects": true, "Desc": ["Stores a value to a given register.", "Size must match the execution mode."], "DestSize": "Size", "EmitValidation": [ "WalkFindRegClass($Value) == $Class" ] + }, + + "StorePF GPR:$Value, u8:#Size": { + "HasSideEffects": true, + "Desc": ["Stores raw PF"], + "DestSize": "Size" + }, + + "StoreAF GPR:$Value, u8:#Size": { + "HasSideEffects": true, + "Desc": ["Stores raw AF"], + "DestSize": "Size" } }, "Memory": { diff --git a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index 55d2c3fba..dd98d5826 100644 --- a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -23,8 +23,8 @@ $end_info$ #define FLAG_C (1U << 1) #define FLAG_Z (1U << 2) #define FLAG_N (1U << 3) -#define FLAG_A (1U << 4) -#define FLAG_P (1U << 5) +#define FLAG_P (1U << 4) +#define FLAG_A (1U << 5) #define FLAG_ZCV (FLAG_Z | FLAG_C | FLAG_V) #define FLAG_NZCV (FLAG_N | FLAG_ZCV) @@ -68,10 +68,6 @@ private: bool EliminateDeadCode(IREmitter* IREmit, Ref CodeNode, IROp_Header* IROp); }; -unsigned DeadFlagCalculationEliminination::FlagForReg(unsigned Reg) { - return Reg == Core::CPUState::PF_AS_GREG ? FLAG_P : Reg == Core::CPUState::AF_AS_GREG ? FLAG_A : 0; -}; - unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType Cond) { switch (Cond) { case COND_AL: return 0; @@ -233,6 +229,11 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { .CanEliminate = true, }; + case OP_LOADPF: return {.Read = FLAG_P}; + case OP_LOADAF: return {.Read = FLAG_A}; + case OP_STOREPF: return {.Write = FLAG_P, .CanEliminate = true}; + case OP_STOREAF: return {.Write = FLAG_A, .CanEliminate = true}; + case OP_NZCVSELECT: case OP_NZCVSELECTINCREMENT: { auto Op = IROp->CW(); @@ -315,29 +316,6 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { }; } - case OP_LOADREGISTER: { - auto Op = IROp->CW(); - if (Op->Class != GPRClass) { - break; - } - - return {.Read = FlagForReg(Op->Reg)}; - } - - case OP_STOREREGISTER: { - auto Op = IROp->CW(); - if (Op->Class != GPRClass) { - break; - } - - unsigned Flag = FlagForReg(Op->Reg); - - return { - .Write = Flag, - .CanEliminate = Flag != 0, - }; - } - default: break; } diff --git a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp index 48812ab42..0707abf03 100644 --- a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp @@ -228,11 +228,14 @@ private: }; Ref DecodeSRANode(const IROp_Header* IROp, Ref Node) { - if (IROp->Op == OP_LOADREGISTER) { + if (IROp->Op == OP_LOADREGISTER || IROp->Op == OP_LOADPF || IROp->Op == OP_LOADAF) { return Node; } else if (IROp->Op == OP_STOREREGISTER) { const IROp_StoreRegister* Op = IROp->C(); return IR->GetNode(Op->Value); + } else if (IROp->Op == OP_STOREPF || IROp->Op == OP_STOREAF) { + const IROp_StorePF* Op = IROp->C(); + return IR->GetNode(Op->Value); } return nullptr; @@ -242,6 +245,8 @@ private: RegisterClassType Class; uint8_t Reg; + uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2; + if (IROp->Op == OP_LOADREGISTER) { const IROp_LoadRegister* Op = IROp->C(); @@ -253,17 +258,15 @@ private: Class = Op->Class; Reg = Op->Reg; + } else if (IROp->Op == OP_LOADPF || IROp->Op == OP_STOREPF) { + return PhysicalRegister {GPRFixedClass, FlagOffset}; + } else if (IROp->Op == OP_LOADAF || IROp->Op == OP_STOREAF) { + return PhysicalRegister {GPRFixedClass, (uint8_t)(FlagOffset + 1)}; } LOGMAN_THROW_A_FMT(Class == GPRClass || Class == FPRClass, "SRA classes"); - uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2; - if (Class == FPRClass) { return PhysicalRegister {FPRFixedClass, Reg}; - } else if (Reg == Core::CPUState::PF_AS_GREG) { - return PhysicalRegister {GPRFixedClass, FlagOffset}; - } else if (Reg == Core::CPUState::AF_AS_GREG) { - return PhysicalRegister {GPRFixedClass, (uint8_t)(FlagOffset + 1)}; } else { return PhysicalRegister {GPRFixedClass, Reg}; } From d17f33a92238603f928bc33cc89e4870c80cccb2 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Mon, 26 Aug 2024 13:26:31 -0400 Subject: [PATCH 08/19] OpcodeDispatcher: don't emit fake 0 for condjump not needed and getting in the way Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/Interface/Core/OpcodeDispatcher.h | 6 +----- 1 file changed, 1 insertion(+), 5 deletions(-) diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h index 40f14fb3b..09220ac6b 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h @@ -151,11 +151,7 @@ public: } IRPair CondJumpNZCV(CondClassType Cond) { FlushRegisterCache(); - - // The jump will ignore the sources, so it doesn't matter what we put here. - // Put an inline constant so RA+codegen will ignore altogether. - auto Placeholder = _InlineConstant(0); - return _CondJump(Placeholder, Placeholder, InvalidNode, InvalidNode, Cond, 0, true); + return _CondJump(InvalidNode, InvalidNode, InvalidNode, InvalidNode, Cond, 0, true); } IRPair CondJumpBit(Ref Src, unsigned Bit, bool Set) { FlushRegisterCache(); From f23bef1653d235915eecb104cb2e007adbe8ee7d Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Mon, 26 Aug 2024 11:30:32 -0400 Subject: [PATCH 09/19] RedundantFlagCalculationElimination: add missing testnz case Signed-off-by: Alyssa Rosenzweig --- .../Interface/IR/Passes/RedundantFlagCalculationElimination.cpp | 2 ++ 1 file changed, 2 insertions(+) diff --git a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index dd98d5826..3e6ad2691 100644 --- a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -110,6 +110,8 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { .Write = FLAG_NZCV, .CanReplace = true, .Replacement = OP_AND, + .CanReplaceWrite = true, + .ReplacementNoWrite = OP_TESTNZ, }; case OP_ADDWITHFLAGS: From c350888d7033638c1b2b492b7bfcbc228cc2a93e Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Mon, 26 Aug 2024 13:00:24 -0400 Subject: [PATCH 10/19] RedundantFlagCalculationElimination: add missing adczero case Signed-off-by: Alyssa Rosenzweig --- .../Interface/IR/Passes/RedundantFlagCalculationElimination.cpp | 1 + 1 file changed, 1 insertion(+) diff --git a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index 3e6ad2691..5a39038c2 100644 --- a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -221,6 +221,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { case OP_LOADNZCV: return {.Read = FLAG_NZCV}; case OP_ADC: + case OP_ADCZERO: case OP_SBB: return {.Read = FLAG_C}; case OP_ADCNZCV: From 249de7c7589d818da46bd369ddfa79f5ad08b142 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Mon, 26 Aug 2024 11:32:50 -0400 Subject: [PATCH 11/19] RedundantFlagCalculationElimination: drop unneeded bools nonzero <==> not dummy Signed-off-by: Alyssa Rosenzweig --- .../RedundantFlagCalculationElimination.cpp | 27 +++++-------------- 1 file changed, 6 insertions(+), 21 deletions(-) diff --git a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index 5a39038c2..2b5f8c961 100644 --- a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -46,14 +46,10 @@ struct FlagInfo { // eliminated. bool CanEliminate; - // If true, the opcode can be replaced with Replacement if its flag writes can - // all be eliminated. - bool CanReplace; + // If set, the opcode can be replaced with Replacement if its flag writes can + // all be eliminated, or ReplacementNoWrite if its register write can be + // eliminated. IROps Replacement; - - // If true, the opcode can be replaced with ReplacementNoWrite if its register - // write is unused but its flags are still needed. - bool CanReplaceWrite; IROps ReplacementNoWrite; }; @@ -108,27 +104,21 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { case OP_ANDWITHFLAGS: return { .Write = FLAG_NZCV, - .CanReplace = true, .Replacement = OP_AND, - .CanReplaceWrite = true, .ReplacementNoWrite = OP_TESTNZ, }; case OP_ADDWITHFLAGS: return { .Write = FLAG_NZCV, - .CanReplace = true, .Replacement = OP_ADD, - .CanReplaceWrite = true, .ReplacementNoWrite = OP_ADDNZCV, }; case OP_SUBWITHFLAGS: return { .Write = FLAG_NZCV, - .CanReplace = true, .Replacement = OP_SUB, - .CanReplaceWrite = true, .ReplacementNoWrite = OP_SUBNZCV, }; @@ -136,9 +126,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { return { .Read = FLAG_C, .Write = FLAG_NZCV, - .CanReplace = true, .Replacement = OP_ADC, - .CanReplaceWrite = true, .ReplacementNoWrite = OP_ADCNZCV, }; @@ -146,7 +134,6 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { return { .Read = FLAG_C, .Write = FLAG_NZCV, - .CanReplace = true, .Replacement = OP_ADCZERO, }; @@ -154,9 +141,7 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { return { .Read = FLAG_C, .Write = FLAG_NZCV, - .CanReplace = true, .Replacement = OP_SBB, - .CanReplaceWrite = true, .ReplacementNoWrite = OP_SBBNZCV, }; @@ -413,16 +398,16 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) { bool Eliminated = false; if ((FlagsRead & Info.Write) == 0) { - if ((Info.CanEliminate || Info.CanReplace) && CodeNode->GetUses() == 0) { + if ((Info.CanEliminate || Info.Replacement) && CodeNode->GetUses() == 0) { IREmit->Remove(CodeNode); Eliminated = true; - } else if (Info.CanReplace) { + } else if (Info.Replacement) { IROp->Op = Info.Replacement; } } else { FlagsRead &= ~Info.Write; - if (Info.CanReplaceWrite && CodeNode->GetUses() == 0) { + if (Info.ReplacementNoWrite && CodeNode->GetUses() == 0) { IROp->Op = Info.ReplacementNoWrite; } } From 50d6ddd59133e8239925f0242491520d249cc5b4 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Mon, 26 Aug 2024 12:31:38 -0400 Subject: [PATCH 12/19] RedundantFlagCalculationElimination: extract per-block logic nfc Signed-off-by: Alyssa Rosenzweig --- .../RedundantFlagCalculationElimination.cpp | 139 +++++++++--------- 1 file changed, 72 insertions(+), 67 deletions(-) diff --git a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index 2b5f8c961..7acc64b22 100644 --- a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -62,6 +62,7 @@ private: unsigned FlagForReg(unsigned Reg); unsigned FlagsForCondClassType(CondClassType Cond); bool EliminateDeadCode(IREmitter* IREmit, Ref CodeNode, IROp_Header* IROp); + void ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, IROp_Header* BlockHeader); }; unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType Cond) { @@ -354,79 +355,83 @@ bool DeadFlagCalculationEliminination::EliminateDeadCode(IREmitter* IREmit, Ref /** * @brief This pass removes dead code locally. */ +void DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, IROp_Header* BlockHeader) { + // We model all flags as read at the end of the block, since this pass is + // presently purely local. Optimizing this requires global anslysis. + uint32_t FlagsRead = FLAG_ALL; + + // Reverse iteration is not yet working with the iterators + auto BlockIROp = BlockHeader->CW(); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + + // Iterate the block in reverse + while (true) { + auto [CodeNode, IROp] = CodeLast(); + + // Optimizing flags can cause earlier flag reads to become dead but dead + // flag reads should not impede optimiation of earlier dead flag writes. + // We must DCE as we go to ensure we converge in a single iteration. + if (!EliminateDeadCode(IREmit, CodeNode, IROp)) { + // Optimiation algorithm: For each flag written... + // + // If the flag has a later read (per FlagsRead), remove the flag from + // FlagsRead, since the reader is covered by this write. + // + // Else, there is no later read, so remove the flag write (if we can). + // This is the active part of the optimization. + // + // Then, add each flag read to FlagsRead. + // + // This order is important: instructions that read-modify-write flags + // (like adcs) first read flags, then write flags. Since we're iterating + // the block backwards, that means we handle the write first. + struct FlagInfo Info = Classify(IROp); + + if (!Info.Trivial) { + bool Eliminated = false; + + if ((FlagsRead & Info.Write) == 0) { + if ((Info.CanEliminate || Info.Replacement) && CodeNode->GetUses() == 0) { + IREmit->Remove(CodeNode); + Eliminated = true; + } else if (Info.Replacement) { + IROp->Op = Info.Replacement; + } + } else { + FlagsRead &= ~Info.Write; + + if (Info.ReplacementNoWrite && CodeNode->GetUses() == 0) { + IROp->Op = Info.ReplacementNoWrite; + } + } + + // If we eliminated the instruction, we eliminate its read too. This + // check is required to ensure the pass converges locally in a single + // iteration. + if (!Eliminated) { + FlagsRead |= Info.Read; + } + } + } + + // Iterate in reverse + if (CodeLast == CodeBegin) { + break; + } + --CodeLast; + } +} + void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) { FEXCORE_PROFILE_SCOPED("PassManager::DFE"); auto CurrentIR = IREmit->ViewIR(); for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) { - // We model all flags as read at the end of the block, since this pass is - // presently purely local. Optimizing this requires global anslysis. - uint32_t FlagsRead = FLAG_ALL; - - // Reverse iteration is not yet working with the iterators - auto BlockIROp = BlockHeader->CW(); - - // We grab these nodes this way so we can iterate easily - auto CodeBegin = CurrentIR.at(BlockIROp->Begin); - auto CodeLast = CurrentIR.at(BlockIROp->Last); - - // Iterate the block in reverse - while (1) { - auto [CodeNode, IROp] = CodeLast(); - - // Optimizing flags can cause earlier flag reads to become dead but dead - // flag reads should not impede optimiation of earlier dead flag writes. - // We must DCE as we go to ensure we converge in a single iteration. - if (!EliminateDeadCode(IREmit, CodeNode, IROp)) { - // Optimiation algorithm: For each flag written... - // - // If the flag has a later read (per FlagsRead), remove the flag from - // FlagsRead, since the reader is covered by this write. - // - // Else, there is no later read, so remove the flag write (if we can). - // This is the active part of the optimization. - // - // Then, add each flag read to FlagsRead. - // - // This order is important: instructions that read-modify-write flags - // (like adcs) first read flags, then write flags. Since we're iterating - // the block backwards, that means we handle the write first. - struct FlagInfo Info = Classify(IROp); - - if (!Info.Trivial) { - bool Eliminated = false; - - if ((FlagsRead & Info.Write) == 0) { - if ((Info.CanEliminate || Info.Replacement) && CodeNode->GetUses() == 0) { - IREmit->Remove(CodeNode); - Eliminated = true; - } else if (Info.Replacement) { - IROp->Op = Info.Replacement; - } - } else { - FlagsRead &= ~Info.Write; - - if (Info.ReplacementNoWrite && CodeNode->GetUses() == 0) { - IROp->Op = Info.ReplacementNoWrite; - } - } - - // If we eliminated the instruction, we eliminate its read too. This - // check is required to ensure the pass converges locally in a single - // iteration. - if (!Eliminated) { - FlagsRead |= Info.Read; - } - } - } - - // Iterate in reverse - if (CodeLast == CodeBegin) { - break; - } - --CodeLast; - } + ProcessBlock(IREmit, CurrentIR, BlockHeader); } } From eb88366614ae474094037db1987867ed0b481428 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 6 Sep 2024 10:57:45 -0400 Subject: [PATCH 13/19] RedundantFlagCalculationElimination: globalize Gather a control flow graph and use it to propagate flags throughout the program. Signed-off-by: Alyssa Rosenzweig --- .../RedundantFlagCalculationElimination.cpp | 117 ++++++++++++++++-- 1 file changed, 108 insertions(+), 9 deletions(-) diff --git a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index 7acc64b22..e81ec6afd 100644 --- a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -7,6 +7,8 @@ $end_info$ */ #include "FEXCore/Core/X86Enums.h" +#include "FEXCore/Utils/MathUtils.h" +#include "FEXCore/fextl/deque.h" #include "Interface/IR/IR.h" #include "Interface/IR/IREmitter.h" @@ -15,9 +17,6 @@ $end_info$ #include "Interface/IR/PassManager.h" -#include -#include - // Flag bit flags #define FLAG_V (1U << 0) #define FLAG_C (1U << 1) @@ -53,6 +52,55 @@ struct FlagInfo { IROps ReplacementNoWrite; }; +struct BlockInfo { + fextl::vector Predecessors; + uint8_t Flags; + bool InWorklist; +}; + +struct ControlFlowGraph { + fextl::unordered_map BlockMap; + IRListView& IR; + + void AddBlock(fextl::deque& Worklist, Ref Block) { + uint32_t ID = IR.GetID(Block).Value; + + // Add the block with conservative flags and already in the worklist. + auto Info = &BlockMap.emplace(ID, BlockInfo {{}, FLAG_ALL, true}).first->second; + + // Add some initial capacity + Info->Predecessors.reserve(2); + + // Add to worklist + Worklist.push_back(Block); + } + + BlockInfo* Get(uint32_t Block) { + return &BlockMap.try_emplace(Block).first->second; + } + + BlockInfo* Get(Ref Block) { + return Get(IR.GetID(Block).Value); + } + + BlockInfo* Get(OrderedNodeWrapper Block) { + return Get(Block.ID().Value); + } + + void RecordEdge(Ref From, Ref To) { + auto Info = Get(To); + Info->Predecessors.push_back(From); + } + + void AddWorklist(fextl::deque& Worklist, Ref Block) { + auto Info = Get(Block); + if (!Info->InWorklist) { + Info->InWorklist = true; + Worklist.push_front(Block); + } + } +}; + class DeadFlagCalculationEliminination final : public FEXCore::IR::Pass { public: void Run(IREmitter* IREmit) override; @@ -62,7 +110,7 @@ private: unsigned FlagForReg(unsigned Reg); unsigned FlagsForCondClassType(CondClassType Cond); bool EliminateDeadCode(IREmitter* IREmit, Ref CodeNode, IROp_Header* IROp); - void ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, IROp_Header* BlockHeader); + bool ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG); }; unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType Cond) { @@ -355,18 +403,28 @@ bool DeadFlagCalculationEliminination::EliminateDeadCode(IREmitter* IREmit, Ref /** * @brief This pass removes dead code locally. */ -void DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, IROp_Header* BlockHeader) { - // We model all flags as read at the end of the block, since this pass is - // presently purely local. Optimizing this requires global anslysis. +bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG) { uint32_t FlagsRead = FLAG_ALL; // Reverse iteration is not yet working with the iterators - auto BlockIROp = BlockHeader->CW(); + auto BlockIROp = CurrentIR.GetOp(Block); // We grab these nodes this way so we can iterate easily auto CodeBegin = CurrentIR.at(BlockIROp->Begin); auto CodeLast = CurrentIR.at(BlockIROp->Last); + // Advance past EndBlock to get at the exit. + --CodeLast; + + // Initialize the FlagsRead mask according to the exit instruction. + auto [ExitNode, ExitOp] = CodeLast(); + if (ExitOp->Op == IR::OP_CONDJUMP) { + auto Op = ExitOp->CW(); + FlagsRead = CFG.Get(Op->TrueBlock)->Flags | CFG.Get(Op->FalseBlock)->Flags; + } else if (ExitOp->Op == IR::OP_JUMP) { + FlagsRead = CFG.Get(ExitOp->Args[0])->Flags; + } + // Iterate the block in reverse while (true) { auto [CodeNode, IROp] = CodeLast(); @@ -423,15 +481,56 @@ void DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListVie } --CodeLast; } + + // For the purposes of global propagation, the content of our progress doesn't + // matter -- only the difference in our final FlagsRead contributes to changes + // in the predecessors. + uint32_t OldFlagsRead = CFG.Get(Block)->Flags; + CFG.Get(Block)->Flags = FlagsRead; + return (OldFlagsRead != FlagsRead); } void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) { FEXCORE_PROFILE_SCOPED("PassManager::DFE"); auto CurrentIR = IREmit->ViewIR(); + fextl::deque Worklist; + ControlFlowGraph CFG {.IR = CurrentIR}; + + // Gather blocks for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) { - ProcessBlock(IREmit, CurrentIR, BlockHeader); + CFG.AddBlock(Worklist, BlockNode); + } + + // Gather CFG + for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) { + auto CodeLast = CurrentIR.at(BlockHeader->C()->Last); + --CodeLast; + auto [ExitNode, ExitOp] = CodeLast(); + if (ExitOp->Op == IR::OP_CONDJUMP) { + auto Op = ExitOp->CW(); + + CFG.RecordEdge(BlockNode, CurrentIR.GetNode(Op->TrueBlock)); + CFG.RecordEdge(BlockNode, CurrentIR.GetNode(Op->FalseBlock)); + } else if (ExitOp->Op == IR::OP_JUMP) { + CFG.RecordEdge(BlockNode, CurrentIR.GetNode(ExitOp->Args[0])); + } + } + + // After processing a block, if we made progress, we must process its + // predecessors to propagate globally. A block will be reprocessed only if + // there is a loop backedge. + for (; !Worklist.empty(); Worklist.pop_back()) { + auto Block = Worklist.back(); + auto Info = CFG.Get(Block); + Info->InWorklist = false; + + if (ProcessBlock(IREmit, CurrentIR, Block, CFG)) { + for (auto Pred : Info->Predecessors) { + CFG.AddWorklist(Worklist, Pred); + } + } } } From 6e6d640fec80e5003695b4084f087c8ebfc53894 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 6 Sep 2024 10:58:23 -0400 Subject: [PATCH 14/19] RedundantFlagCalculationElimination: fold compares/axflag into branches now that we know whether flags are killed on the edge, we can improve branch isel Signed-off-by: Alyssa Rosenzweig --- .../RedundantFlagCalculationElimination.cpp | 85 +++++++++++++++++++ 1 file changed, 85 insertions(+) diff --git a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index e81ec6afd..a4b9e455b 100644 --- a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -110,6 +110,8 @@ private: unsigned FlagForReg(unsigned Reg); unsigned FlagsForCondClassType(CondClassType Cond); bool EliminateDeadCode(IREmitter* IREmit, Ref CodeNode, IROp_Header* IROp); + void FoldBranch(IREmitter* IREmit, IRListView& CurrentIR, IROp_CondJump* Op, Ref CodeNode); + CondClassType X86ToArmFloatCond(CondClassType X86); bool ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG); }; @@ -400,6 +402,69 @@ bool DeadFlagCalculationEliminination::EliminateDeadCode(IREmitter* IREmit, Ref return true; } +CondClassType DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClassType X86) { + // Table of x86 condition codes that map to arm64 condition codes, in the + // sense that fcmp+axflag+branch(x86) is equivalent to fcmp+branch(arm). + // + // E would be "equal or unordered", no condition code. + // G would be "greater than or less than", no condition code. + // + // SF/OF conditions are trivial and therefore shouldn't actually be generated + switch (X86) { + case COND_UGE /* A */: return {COND_FGE} /* GE */; + case COND_UGT /* AE */: return {COND_FGT} /* GT */; + case COND_ULT /* B */: return {COND_SLT} /* LT */; + case COND_ULE /* BE */: return {COND_SLE} /* LE */; + case COND_SLE /* LE */: return {COND_SLE} /* LE */; + default: return {COND_AL}; + } +} + +void DeadFlagCalculationEliminination::FoldBranch(IREmitter* IREmit, IRListView& CurrentIR, IROp_CondJump* Op, Ref CodeNode) { + // Skip past StoreRegisters at the end -- they don't touch flags. + auto PrevWrap = CodeNode->Header.Previous; + while (CurrentIR.GetOp(PrevWrap)->Op == OP_STOREREGISTER || + CurrentIR.GetOp(PrevWrap)->Op == OP_STOREPF || CurrentIR.GetOp(PrevWrap)->Op == OP_STOREAF) { + PrevWrap = CurrentIR.GetNode(PrevWrap)->Header.Previous; + } + + auto Prev = CurrentIR.GetOp(PrevWrap); + if (Prev->Op == OP_AXFLAG) { + // Pattern match a branch fed by AXFLAG. + CondClassType ArmCond = X86ToArmFloatCond(Op->Cond); + if (ArmCond == COND_AL) { + return; + } + + Op->Cond = ArmCond; + } else if (Prev->Op == OP_SUBNZCV) { + // Pattern match a branch fed by a compare. We could also handle bit tests + // here, but tbz/tbnz has a limited offset range which we don't have a way to + // deal with yet. Let's hope that's not a big deal. + if (!(Op->Cond == COND_NEQ || Op->Cond == COND_EQ) || (Prev->Size < 4)) { + return; + } + + auto SecondArg = CurrentIR.GetOp(Prev->Args[1]); + if (SecondArg->Op != OP_INLINECONSTANT || SecondArg->C()->Constant != 0) { + return; + } + + // We've matched. Fold the compare into branch. + IREmit->ReplaceNodeArgument(CodeNode, 0, CurrentIR.GetNode(Prev->Args[0])); + IREmit->ReplaceNodeArgument(CodeNode, 1, CurrentIR.GetNode(Prev->Args[1])); + Op->FromNZCV = false; + Op->CompareSize = Prev->Size; + } else { + return; + } + + // The compare/test/axflag sets flags but does not write registers. Flags are + // dead after the jump. The jump does not read flags anymore. There is no + // intervening instruction. Therefore the compare is dead. + IREmit->Remove(CurrentIR.GetNode(PrevWrap)); +} + /** * @brief This pass removes dead code locally. */ @@ -532,6 +597,26 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) { } } } + + // Fold compares into branches now that we're otherwise optimized. This needs + // to run after eliminating carries etc and it needs the global flag metadata. + // But it only needs to run once, we don't do it in the loop. + for (auto [Block, _] : CurrentIR.GetBlocks()) { + // Grab the jump + auto BlockIROp = CurrentIR.GetOp(Block); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + --CodeLast; + + auto [ExitNode, ExitOp] = CodeLast(); + if (ExitOp->Op == IR::OP_CONDJUMP) { + auto Op = ExitOp->CW(); + uint32_t FlagsOut = CFG.Get(Op->TrueBlock)->Flags | CFG.Get(Op->FalseBlock)->Flags; + + if ((FlagsOut & FLAG_NZCV) == 0 && Op->FromNZCV) { + FoldBranch(IREmit, CurrentIR, Op, ExitNode); + } + } + } } fextl::unique_ptr CreateDeadFlagCalculationEliminination() { From 47ac9edd7ab7929f529c01484474a74a4a76d5c0 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 6 Sep 2024 11:00:47 -0400 Subject: [PATCH 15/19] RedundantFlagCalculationElimination: select testz this saves uops. Signed-off-by: Alyssa Rosenzweig --- .../RedundantFlagCalculationElimination.cpp | 19 +++++++++++++------ 1 file changed, 13 insertions(+), 6 deletions(-) diff --git a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index a4b9e455b..f0b7a7d6b 100644 --- a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -523,14 +523,21 @@ bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListVie } else if (Info.Replacement) { IROp->Op = Info.Replacement; } - } else { - FlagsRead &= ~Info.Write; - - if (Info.ReplacementNoWrite && CodeNode->GetUses() == 0) { - IROp->Op = Info.ReplacementNoWrite; - } + } else if (Info.ReplacementNoWrite && CodeNode->GetUses() == 0) { + IROp->Op = Info.ReplacementNoWrite; } + // If we don't care about the sign or carry, we can optimize testnz. + // Carry is inverted between testz and testnz so we check that too. Note + // this flag is outside of the if, since the TestNZ might result from + // optimizing AndWithFlags, and we need to converge locally in a single + // iteration. + if (IROp->Op == OP_TESTNZ && IROp->Size < 4 && !(FlagsRead & (FLAG_N | FLAG_C))) { + IROp->Op = OP_TESTZ; + } + + FlagsRead &= ~Info.Write; + // If we eliminated the instruction, we eliminate its read too. This // check is required to ensure the pass converges locally in a single // iteration. From dd7c66db2e3866e89333f699aed99f1f189ef05e Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Sat, 7 Sep 2024 10:56:34 -0400 Subject: [PATCH 16/19] RedundantFlagCalculationElimination: use LUT for flags Signed-off-by: Alyssa Rosenzweig --- .../RedundantFlagCalculationElimination.cpp | 197 ++++++++++++------ 1 file changed, 135 insertions(+), 62 deletions(-) diff --git a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index f0b7a7d6b..b2e25430e 100644 --- a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -7,6 +7,7 @@ $end_info$ */ #include "FEXCore/Core/X86Enums.h" +#include "FEXCore/Utils/CompilerDefs.h" #include "FEXCore/Utils/MathUtils.h" #include "FEXCore/fextl/deque.h" #include "Interface/IR/IR.h" @@ -31,10 +32,7 @@ $end_info$ namespace FEXCore::IR { -struct FlagInfo { - // If set, all following fields are zero, used for a quick exit. - bool Trivial; - +struct FlagInfoUnpacked { // Set of flags read by the instruction. unsigned Read; @@ -50,6 +48,52 @@ struct FlagInfo { // eliminated. IROps Replacement; IROps ReplacementNoWrite; + + // Needs speical handling + bool Special; +}; + +struct FlagInfo { + uint64_t Raw; + + static constexpr struct FlagInfo Pack(struct FlagInfoUnpacked F) { + uint64_t R = F.Read | (F.Write << 8) | (F.CanEliminate << 16) | (((uint64_t)F.Replacement) << 32) | + ((uint64_t)F.ReplacementNoWrite << 48) | (F.Special ? (1ull << 63) : 0); + return {.Raw = R}; + } + + bool Trivial() { + return Raw == 0; + } + + unsigned Read() { + return Bits(0, 8); + } + + unsigned Write() { + return Bits(8, 8); + } + + bool CanEliminate() { + return Bits(16, 1); + } + + bool Special() { + return Bits(63, 1); + } + + IROps Replacement() { + return (IROps)Bits(32, 16); + } + + IROps ReplacementNoWrite() { + return (IROps)Bits(48, 16); + } + +private: + unsigned Bits(unsigned Start, unsigned Count) { + return (Raw >> Start) & ((1u << Count) - 1); + } }; struct BlockInfo { @@ -150,157 +194,186 @@ unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType C } } -FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { - switch (IROp->Op) { +constexpr FlagInfo ClassifyConst(IROps Op) { + switch (Op) { case OP_ANDWITHFLAGS: - return { + return FlagInfo::Pack({ .Write = FLAG_NZCV, .Replacement = OP_AND, .ReplacementNoWrite = OP_TESTNZ, - }; + }); case OP_ADDWITHFLAGS: - return { + return FlagInfo::Pack({ .Write = FLAG_NZCV, .Replacement = OP_ADD, .ReplacementNoWrite = OP_ADDNZCV, - }; + }); case OP_SUBWITHFLAGS: - return { + return FlagInfo::Pack({ .Write = FLAG_NZCV, .Replacement = OP_SUB, .ReplacementNoWrite = OP_SUBNZCV, - }; + }); case OP_ADCWITHFLAGS: - return { + return FlagInfo::Pack({ .Read = FLAG_C, .Write = FLAG_NZCV, .Replacement = OP_ADC, .ReplacementNoWrite = OP_ADCNZCV, - }; + }); case OP_ADCZEROWITHFLAGS: - return { + return FlagInfo::Pack({ .Read = FLAG_C, .Write = FLAG_NZCV, .Replacement = OP_ADCZERO, - }; + }); case OP_SBBWITHFLAGS: - return { + return FlagInfo::Pack({ .Read = FLAG_C, .Write = FLAG_NZCV, .Replacement = OP_SBB, .ReplacementNoWrite = OP_SBBNZCV, - }; + }); case OP_SHIFTFLAGS: // _ShiftFlags conditionally sets NZCV+PF, which we model here as a // read-modify-write. Logically, it also conditionally makes AF undefined, // which we model by omitting AF from both Read and Write sets (since // "cond ? AF : undef" may be optimized to "AF"). - return { + return FlagInfo::Pack({ .Read = FLAG_NZCV | FLAG_P, .Write = FLAG_NZCV | FLAG_P, .CanEliminate = true, - }; + }); case OP_ROTATEFLAGS: // _RotateFlags conditionally sets CV, again modeled as RMW. - return { + return FlagInfo::Pack({ .Read = FLAG_C | FLAG_V, .Write = FLAG_C | FLAG_V, .CanEliminate = true, - }; + }); - case OP_RDRAND: return {.Write = FLAG_NZCV}; + case OP_RDRAND: return FlagInfo::Pack({.Write = FLAG_NZCV}); case OP_ADDNZCV: case OP_SUBNZCV: case OP_TESTNZ: case OP_FCMP: case OP_STORENZCV: - return { + return FlagInfo::Pack({ .Write = FLAG_NZCV, .CanEliminate = true, - }; + }); case OP_AXFLAG: // Per the Arm spec, axflag reads Z/V/C but not N. It writes all flags. - return { + return FlagInfo::Pack({ .Read = FLAG_ZCV, .Write = FLAG_NZCV, .CanEliminate = true, - }; + }); case OP_CMPPAIRZ: - return { + return FlagInfo::Pack({ .Write = FLAG_Z, .CanEliminate = true, - }; + }); case OP_CARRYINVERT: - return { + return FlagInfo::Pack({ .Read = FLAG_C, .Write = FLAG_C, .CanEliminate = true, - }; + }); case OP_SETSMALLNZV: - return { + return FlagInfo::Pack({ .Write = FLAG_N | FLAG_Z | FLAG_V, .CanEliminate = true, - }; + }); - case OP_LOADNZCV: return {.Read = FLAG_NZCV}; + case OP_LOADNZCV: return FlagInfo::Pack({.Read = FLAG_NZCV}); case OP_ADC: case OP_ADCZERO: - case OP_SBB: return {.Read = FLAG_C}; + case OP_SBB: return FlagInfo::Pack({.Read = FLAG_C}); case OP_ADCNZCV: case OP_SBBNZCV: - return { + return FlagInfo::Pack({ .Read = FLAG_C, .Write = FLAG_NZCV, .CanEliminate = true, - }; + }); - case OP_LOADPF: return {.Read = FLAG_P}; - case OP_LOADAF: return {.Read = FLAG_A}; - case OP_STOREPF: return {.Write = FLAG_P, .CanEliminate = true}; - case OP_STOREAF: return {.Write = FLAG_A, .CanEliminate = true}; + case OP_LOADPF: return FlagInfo::Pack({.Read = FLAG_P}); + case OP_LOADAF: return FlagInfo::Pack({.Read = FLAG_A}); + case OP_STOREPF: return FlagInfo::Pack({.Write = FLAG_P, .CanEliminate = true}); + case OP_STOREAF: return FlagInfo::Pack({.Write = FLAG_A, .CanEliminate = true}); + case OP_NZCVSELECT: + case OP_NZCVSELECTINCREMENT: + case OP_NEG: + case OP_CONDJUMP: + case OP_CONDSUBNZCV: + case OP_CONDADDNZCV: + case OP_RMIFNZCV: + case OP_INVALIDATEFLAGS: return FlagInfo::Pack({.Special = true}); + default: return FlagInfo::Pack({}); + } +} + +constexpr auto FlagInfos = std::invoke([] { + std::array ret = {}; + + for (unsigned i = 0; i < OP_LAST; ++i) { + ret[i] = ClassifyConst((IROps)i); + } + + return ret; +}); + +FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { + FlagInfo Info = FlagInfos[IROp->Op]; + if (!Info.Special()) { + return Info; + } + + switch (IROp->Op) { case OP_NZCVSELECT: case OP_NZCVSELECTINCREMENT: { auto Op = IROp->CW(); - return {.Read = FlagsForCondClassType(Op->Cond)}; + return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)}); } case OP_NEG: { auto Op = IROp->CW(); - return {.Read = FlagsForCondClassType(Op->Cond)}; + return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)}); } case OP_CONDJUMP: { auto Op = IROp->CW(); if (!Op->FromNZCV) { - break; + return FlagInfo::Pack({}); } - return {.Read = FlagsForCondClassType(Op->Cond)}; + return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)}); } case OP_CONDSUBNZCV: case OP_CONDADDNZCV: { auto Op = IROp->CW(); - return { + return FlagInfo::Pack({ .Read = FlagsForCondClassType(Op->Cond), .Write = FLAG_NZCV, .CanEliminate = true, - }; + }); } case OP_RMIFNZCV: { @@ -311,10 +384,10 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { static_assert(FLAG_C == (1 << 1), "rmif mask lines up with our bits"); static_assert(FLAG_V == (1 << 0), "rmif mask lines up with our bits"); - return { + return FlagInfo::Pack({ .Write = Op->Mask, .CanEliminate = true, - }; + }); } case OP_INVALIDATEFLAGS: { @@ -349,16 +422,16 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { // The mental model of InvalidateFlags is writing undefined values to all // of the selected flags, allowing the write-after-write optimizations to // optimize invalidate-after-write for free. - return { + return FlagInfo::Pack({ .Write = Flags, .CanEliminate = true, - }; + }); } - default: break; + default: LOGMAN_THROW_AA_FMT(false, "invalid special op"); FEX_UNREACHABLE; } - return {.Trivial = true}; + FEX_UNREACHABLE; } // General purpose dead code elimination. Returns whether flag handling should @@ -513,18 +586,18 @@ bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListVie // the block backwards, that means we handle the write first. struct FlagInfo Info = Classify(IROp); - if (!Info.Trivial) { + if (!Info.Trivial()) { bool Eliminated = false; - if ((FlagsRead & Info.Write) == 0) { - if ((Info.CanEliminate || Info.Replacement) && CodeNode->GetUses() == 0) { + if ((FlagsRead & Info.Write()) == 0) { + if ((Info.CanEliminate() || Info.Replacement()) && CodeNode->GetUses() == 0) { IREmit->Remove(CodeNode); Eliminated = true; - } else if (Info.Replacement) { - IROp->Op = Info.Replacement; + } else if (Info.Replacement()) { + IROp->Op = Info.Replacement(); } - } else if (Info.ReplacementNoWrite && CodeNode->GetUses() == 0) { - IROp->Op = Info.ReplacementNoWrite; + } else if (Info.ReplacementNoWrite() && CodeNode->GetUses() == 0) { + IROp->Op = Info.ReplacementNoWrite(); } // If we don't care about the sign or carry, we can optimize testnz. @@ -536,13 +609,13 @@ bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListVie IROp->Op = OP_TESTZ; } - FlagsRead &= ~Info.Write; + FlagsRead &= ~Info.Write(); // If we eliminated the instruction, we eliminate its read too. This // check is required to ensure the pass converges locally in a single // iteration. if (!Eliminated) { - FlagsRead |= Info.Read; + FlagsRead |= Info.Read(); } } } From b36d1f7e7b208ce72eddcb1755db03b0eaafc911 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 6 Sep 2024 10:58:51 -0400 Subject: [PATCH 17/19] RedundantFlagCalculationElimination: optimize parity another global CFG-based optimization -- if we know that the raw PF is already 1-bit we can skip parity evaluation, saving work with floating point compares. Signed-off-by: Alyssa Rosenzweig --- .../RedundantFlagCalculationElimination.cpp | 68 +++++++++++++++++++ 1 file changed, 68 insertions(+) diff --git a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index b2e25430e..19d6b137b 100644 --- a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -157,6 +157,7 @@ private: void FoldBranch(IREmitter* IREmit, IRListView& CurrentIR, IROp_CondJump* Op, Ref CodeNode); CondClassType X86ToArmFloatCond(CondClassType X86); bool ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG); + void OptimizeParity(IREmitter* IREmit, IRListView& CurrentIR, ControlFlowGraph& CFG); }; unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType Cond) { @@ -635,6 +636,69 @@ bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListVie return (OldFlagsRead != FlagsRead); } +void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListView& CurrentIR, ControlFlowGraph& CFG) { + // Mapping for flags inside this pass. + const uint8_t PARTIAL = 0; + const uint8_t FULL = 1; + + // Initialize conservatively: all blocks need full parity. This initialization + // matters for proper handling of backedges. + for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) { + CFG.Get(Block)->Flags = FULL; + } + + for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) { + bool Full = false; + auto Predecessors = CFG.Get(Block)->Predecessors; + + if (Predecessors.empty()) { + // Conservatively assume there was full parity before the start block + Full = true; + } else { + // If any predecessor needs full parity at the end, we need full parity. + for (auto Pred : Predecessors) { + Full |= (CFG.Get(Pred)->Flags == FULL); + } + } + + for (auto [CodeNode, IROp] : CurrentIR.GetCode(Block)) { + if (IROp->Op == OP_STOREPF) { + auto Op = IROp->CW(); + auto Generator = CurrentIR.GetOp(Op->Value); + + // Determine if we only write 0/1 to the parity flag. + Full = true; + if (Generator->Op == OP_NZCVSELECT) { + auto C0 = CurrentIR.GetOp(Generator->Args[0]); + auto C1 = CurrentIR.GetOp(Generator->Args[1]); + if (C0->Op == C1->Op && C0->Op == OP_INLINECONSTANT) { + auto IC0 = CurrentIR.GetOp(Generator->Args[0]); + auto IC1 = CurrentIR.GetOp(Generator->Args[1]); + + // We need the full 8 if the constant has upper bits set. + Full = (IC0->Constant | IC1->Constant) & ~1; + } + } + } else if (IROp->Op == OP_PARITY && !Full) { + // Eliminate parity calculations if it's only 1-bit. + auto Parity = IROp->C(); + Ref Value = CurrentIR.GetNode(Parity->Raw); + + if (Parity->Invert) { + IREmit->SetWriteCursor(CodeNode); + Value = IREmit->_Xor(OpSize::i32Bit, Value, IREmit->_InlineConstant(1)); + } + + IREmit->ReplaceUsesWithAfter(CodeNode, Value, CurrentIR.at(CodeNode)); + IREmit->Remove(CodeNode); + } + } + + // Record our final state for our successors to read. + CFG.Get(Block)->Flags = Full ? FULL : PARTIAL; + } +} + void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) { FEXCORE_PROFILE_SCOPED("PassManager::DFE"); @@ -697,6 +761,10 @@ void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) { } } } + + if (CurrentIR.GetHeader()->ReadsParity) { + OptimizeParity(IREmit, CurrentIR, CFG); + } } fextl::unique_ptr CreateDeadFlagCalculationEliminination() { From 384882744ccae740ccd35b4f01062439323f87e1 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Tue, 27 Aug 2024 12:31:17 -0400 Subject: [PATCH 18/19] DeadStoreElimination: drop now that it's redundant it's only really load bearing for pf/af, which is handled as a global flag opt now. this mitigates some of the compile time hit from globalizing flag opts. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/CMakeLists.txt | 1 - FEXCore/Source/Interface/IR/PassManager.cpp | 1 - FEXCore/Source/Interface/IR/Passes.h | 1 - .../IR/Passes/DeadStoreElimination.cpp | 154 ------------------ 4 files changed, 157 deletions(-) delete mode 100644 FEXCore/Source/Interface/IR/Passes/DeadStoreElimination.cpp diff --git a/FEXCore/Source/CMakeLists.txt b/FEXCore/Source/CMakeLists.txt index 85b23af6b..971271a78 100644 --- a/FEXCore/Source/CMakeLists.txt +++ b/FEXCore/Source/CMakeLists.txt @@ -140,7 +140,6 @@ set (SRCS Interface/IR/Passes/IRValidation.cpp Interface/IR/Passes/RAValidation.cpp Interface/IR/Passes/RedundantFlagCalculationElimination.cpp - Interface/IR/Passes/DeadStoreElimination.cpp Interface/IR/Passes/RegisterAllocationPass.cpp Interface/IR/Passes/x87StackOptimizationPass.cpp Utils/Telemetry.cpp diff --git a/FEXCore/Source/Interface/IR/PassManager.cpp b/FEXCore/Source/Interface/IR/PassManager.cpp index 6f5170e6b..5072364a6 100644 --- a/FEXCore/Source/Interface/IR/PassManager.cpp +++ b/FEXCore/Source/Interface/IR/PassManager.cpp @@ -71,7 +71,6 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) { if (!DisablePasses()) { InsertPass(CreateX87StackOptimizationPass()); - InsertPass(CreateDeadStoreElimination()); InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID)); InsertPass(CreateDeadFlagCalculationEliminination()); } diff --git a/FEXCore/Source/Interface/IR/Passes.h b/FEXCore/Source/Interface/IR/Passes.h index 3b377dc6f..a8a3d1cb6 100644 --- a/FEXCore/Source/Interface/IR/Passes.h +++ b/FEXCore/Source/Interface/IR/Passes.h @@ -18,7 +18,6 @@ class RegisterAllocationData; fextl::unique_ptr CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID); fextl::unique_ptr CreateDeadFlagCalculationEliminination(); -fextl::unique_ptr CreateDeadStoreElimination(); fextl::unique_ptr CreateRegisterAllocationPass(); fextl::unique_ptr CreateX87StackOptimizationPass(); diff --git a/FEXCore/Source/Interface/IR/Passes/DeadStoreElimination.cpp b/FEXCore/Source/Interface/IR/Passes/DeadStoreElimination.cpp deleted file mode 100644 index 5a3471324..000000000 --- a/FEXCore/Source/Interface/IR/Passes/DeadStoreElimination.cpp +++ /dev/null @@ -1,154 +0,0 @@ -// SPDX-License-Identifier: MIT -/* -$info$ -tags: ir|opts -desc: Cross block store-after-store elimination -$end_info$ -*/ - -#include "Interface/IR/IREmitter.h" -#include "Interface/IR/PassManager.h" - -#include -#include -#include -#include -#include - -#include -#include -#include - -namespace FEXCore::IR { - -constexpr int PropagationRounds = 5; - -// Return a bit representing a single GPR or FPR. -static inline uint64_t RegBit(RegisterClassType Class, uint32_t Reg) { - uint32_t AdjustedReg = (Class == FPRClass) ? (32 + Reg) : Reg; - - return 1ULL << AdjustedReg; -} - -class DeadStoreElimination final : public FEXCore::IR::Pass { -public: - void Run(IREmitter* IREmit) override; -}; - -struct ReadWriteKill { - uint64_t reads {0}; - uint64_t writes {0}; - uint64_t kill {0}; -}; - -struct Info { - ReadWriteKill reg; -}; - -/** - * @brief This is a temporary pass to detect simple multiblock dead reg stores - * - * First pass computes which regs are read and written per block - * - * Second pass computes which regs are stored, but overwritten by the next block(s). - * It also propagates this information a few times to catch dead regs across multiple blocks. - * - * Third pass removes the dead stores. - * - */ -void DeadStoreElimination::Run(IREmitter* IREmit) { - FEXCORE_PROFILE_SCOPED("PassManager::DSE"); - - auto CurrentIR = IREmit->ViewIR(); - fextl::vector InfoMap(CurrentIR.GetSSACount()); - - // Pass 1 - // Compute regs read/writes per block - // This is conservative and doesn't try to be smart about loads after writes - { - for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) { - auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value]; - - for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) { - if (IROp->Op == OP_STOREREGISTER) { - auto Op = IROp->C(); - BlockInfo.reg.writes |= RegBit(Op->Class, Op->Reg); - } else if (IROp->Op == OP_LOADREGISTER) { - auto Op = IROp->C(); - BlockInfo.reg.reads |= RegBit(Op->Class, Op->Reg); - } else if (IROp->Op == OP_INVALIDATEFLAGS) { - auto Op = IROp->C(); - - if (Op->Flags & (1u << X86State::RFLAG_PF_RAW_LOC)) { - BlockInfo.reg.writes |= RegBit(GPRClass, Core::CPUState::PF_AS_GREG); - } - - if (Op->Flags & (1u << X86State::RFLAG_AF_RAW_LOC)) { - BlockInfo.reg.writes |= RegBit(GPRClass, Core::CPUState::AF_AS_GREG); - } - } - } - } - } - - // Pass 2 - // Compute flags/registers that are stored, but always ovewritten in the next blocks - // Propagate the information a few times to eliminate more - for (int i = 0; i < PropagationRounds; i++) { - for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) { - auto CodeBlock = BlockIROp->C(); - - auto IROp = CurrentIR.GetNode(CurrentIR.GetNode(CodeBlock->Last)->Header.Previous)->Op(CurrentIR.GetData()); - - if (IROp->Op == OP_JUMP) { - auto Op = IROp->C(); - auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value]; - auto& TargetInfo = InfoMap[Op->Header.Args[0].ID().Value]; - - // stores to remove are written by the next block but not read - BlockInfo.reg.kill = TargetInfo.reg.writes & ~(TargetInfo.reg.reads) & ~BlockInfo.reg.reads; - - // If written by the next block can be considered as written by this block, if not read - BlockInfo.reg.writes |= BlockInfo.reg.kill & ~BlockInfo.reg.reads; - } else if (IROp->Op == OP_CONDJUMP) { - auto Op = IROp->C(); - - auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value]; - auto& TrueTargetInfo = InfoMap[Op->TrueBlock.ID().Value]; - auto& FalseTargetInfo = InfoMap[Op->FalseBlock.ID().Value]; - - // stores to remove are written by the next blocks but not read - BlockInfo.reg.kill = TrueTargetInfo.reg.writes & ~(TrueTargetInfo.reg.reads) & ~BlockInfo.reg.reads; - BlockInfo.reg.kill &= FalseTargetInfo.reg.writes & ~(FalseTargetInfo.reg.reads) & ~BlockInfo.reg.reads; - - // if written by the next blocks can be considered as written by this block, if not read - BlockInfo.reg.writes |= BlockInfo.reg.kill & ~BlockInfo.reg.reads; - } - } - } - - // Pass 3 - // Remove the dead stores - { - for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) { - auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value]; - - for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) { - if (IROp->Op == OP_STOREREGISTER) { - auto Op = IROp->C(); - - // If this OP_STOREREGISTER is never read, remove it - if (BlockInfo.reg.kill & RegBit(Op->Class, Op->Reg)) { - IREmit->Remove(CodeNode); - } - } - } - } - } -} - -fextl::unique_ptr CreateDeadStoreElimination() { - return fextl::make_unique(); -} - -} // namespace FEXCore::IR From 46f36dc89d96448d28c6765e8b84942b35d393c8 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Sat, 7 Sep 2024 10:59:41 -0400 Subject: [PATCH 19/19] InstCountCI: Update Signed-off-by: Alyssa Rosenzweig --- .../InstructionCountCI/AVX128/VEX_map1.json | 16 +- .../InstructionCountCI/FlagM/FlagOpts.json | 44 ++-- .../InstructionCountCI/FlagM/Primary.json | 18 +- .../InstructionCountCI/FlagM/Secondary.json | 52 ++-- unittests/InstructionCountCI/FlagM/x87.json | 96 ++++---- .../InstructionCountCI/FlagM/x87_f64.json | 96 ++++---- unittests/InstructionCountCI/Primary.json | 18 +- unittests/InstructionCountCI/Secondary.json | 60 ++--- .../InstructionCountCI/Secondary_OpSize.json | 8 +- unittests/InstructionCountCI/VEX_map1.json | 16 +- unittests/InstructionCountCI/x87.json | 96 ++++---- unittests/InstructionCountCI/x87_f64.json | 224 +++++++++--------- 12 files changed, 367 insertions(+), 377 deletions(-) diff --git a/unittests/InstructionCountCI/AVX128/VEX_map1.json b/unittests/InstructionCountCI/AVX128/VEX_map1.json index 85f0f3f6c..d95a72215 100644 --- a/unittests/InstructionCountCI/AVX128/VEX_map1.json +++ b/unittests/InstructionCountCI/AVX128/VEX_map1.json @@ -2891,8 +2891,8 @@ "fcmp s16, s17", "mov w27, #0x0", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "vucomisd xmm0, xmm1": { @@ -2904,8 +2904,8 @@ "fcmp d16, d17", "mov w27, #0x0", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "vcomiss xmm0, xmm1": { @@ -2917,8 +2917,8 @@ "fcmp s16, s17", "mov w27, #0x0", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "vcomisd xmm0, xmm1": { @@ -2930,8 +2930,8 @@ "fcmp d16, d17", "mov w27, #0x0", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "vaddps xmm0, xmm1, xmm2": { diff --git a/unittests/InstructionCountCI/FlagM/FlagOpts.json b/unittests/InstructionCountCI/FlagM/FlagOpts.json index d8919c5fc..e4c373942 100644 --- a/unittests/InstructionCountCI/FlagM/FlagOpts.json +++ b/unittests/InstructionCountCI/FlagM/FlagOpts.json @@ -203,27 +203,21 @@ }, "Variable rotate-through-carry dead": { "x86InstructionCount": 2, - "ExpectedInstructionCount": 17, + "ExpectedInstructionCount": 11, "x86Insts": [ "rcr rax, cl", "test rax, rdx" ], "ExpectedArm64ASM": [ "and x20, x7, #0x3f", - "cbz x20, #+0x38", + "cbz x20, #+0x20", "lsr x20, x4, x7", "cset w21, lo", "neg x22, x7", "lsl x23, x4, x22", "orr x20, x20, x23, lsl #1", - "sub x23, x7, #0x1 (1)", - "lsr x23, x4, x23", - "eor x23, x23, #0x1", - "rmif x23, #63, #nzCv", "lsl x21, x21, x22", "orr x4, x20, x21", - "eor x20, x4, x4, lsr #1", - "rmif x20, #62, #nzcV", "ands x26, x4, x5", "cfinv" ] @@ -290,10 +284,10 @@ "ExpectedArm64ASM": [ "and w26, w4, w6", "mov x4, x26", - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", - "and x20, x20, #0x1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", + "and w20, w20, #0x1", "bfxil x7, x20, #0, #8", "cmn wzr, w7, lsl #24", "cfinv", @@ -302,7 +296,7 @@ }, "UCOMISS use only PF": { "x86InstructionCount": 3, - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 4, "x86Insts": [ "ucomiss xmm0, xmm1", "setnp cl", @@ -311,11 +305,7 @@ "ExpectedArm64ASM": [ "fcmp s16, s17", "cset w26, vc", - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", - "and x20, x20, #0x1", - "bfxil x7, x20, #0, #8", + "bfxil x7, x26, #0, #8", "subs x26, x4, #0x0 (0)" ] }, @@ -328,7 +318,7 @@ "test cl, cl" ], "ExpectedArm64ASM": [ - "cmn wzr, w4, lsl #16", + "tst w4, #0xffff", "cset x20, eq", "bfxil x4, x20, #0, #8", "cmn wzr, w7, lsl #24", @@ -345,10 +335,10 @@ "test cl, cl" ], "ExpectedArm64ASM": [ - "and w20, w4, w6", - "cmn wzr, w20, lsl #16", - "cset x21, eq", - "bfxil x4, x21, #0, #8", + "and w0, w4, w6", + "tst w0, #0xffff", + "cset x20, eq", + "bfxil x4, x20, #0, #8", "cmn wzr, w7, lsl #24", "cfinv", "mov x26, x7" @@ -364,10 +354,10 @@ ], "ExpectedArm64ASM": [ "mov w20, #0x89", - "and w20, w4, w20", - "cmn wzr, w20, lsl #24", - "cset x21, ne", - "bfxil x4, x21, #0, #8", + "and w0, w4, w20", + "tst w0, #0xff", + "cset x20, ne", + "bfxil x4, x20, #0, #8", "cmn wzr, w7, lsl #24", "cfinv", "mov x26, x7" diff --git a/unittests/InstructionCountCI/FlagM/Primary.json b/unittests/InstructionCountCI/FlagM/Primary.json index 205fe0dd1..9202ecda7 100644 --- a/unittests/InstructionCountCI/FlagM/Primary.json +++ b/unittests/InstructionCountCI/FlagM/Primary.json @@ -1728,9 +1728,9 @@ "orr x20, x20, x21, lsl #20", "ldrb w21, [x28, #997]", "orr x20, x20, x21, lsl #21", - "eor w21, w26, w26, lsr #4", - "eor w21, w21, w21, lsr #2", - "eor w21, w21, w21, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w21, w0, w0, lsr #1", "orr x21, x21, #0xfffffffffffffffe", "orn x20, x20, x21, ror #62", "mrs x21, nzcv", @@ -1773,9 +1773,9 @@ "orr x20, x20, x21, lsl #20", "ldrb w21, [x28, #997]", "orr x20, x20, x21, lsl #21", - "eor w21, w26, w26, lsr #4", - "eor w21, w21, w21, lsr #2", - "eor w21, w21, w21, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w21, w0, w0, lsr #1", "orr x21, x21, #0xfffffffffffffffe", "orn x20, x20, x21, ror #62", "mrs x21, nzcv", @@ -1847,9 +1847,9 @@ "eor w21, w27, w26", "ubfx w21, w21, #4, #1", "orr x20, x20, x21, lsl #4", - "eor w21, w26, w26, lsr #4", - "eor w21, w21, w21, lsr #2", - "eor w21, w21, w21, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w21, w0, w0, lsr #1", "orr x21, x21, #0xfffffffffffffffe", "orn x20, x20, x21, ror #62", "mrs x21, nzcv", diff --git a/unittests/InstructionCountCI/FlagM/Secondary.json b/unittests/InstructionCountCI/FlagM/Secondary.json index 313949677..423ca7757 100644 --- a/unittests/InstructionCountCI/FlagM/Secondary.json +++ b/unittests/InstructionCountCI/FlagM/Secondary.json @@ -257,9 +257,9 @@ "ExpectedInstructionCount": 8, "Comment": "0x0f 0x4a", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "mrs x21, nzcv", "tst w20, #0x1", "csel w20, w6, w4, ne", @@ -271,9 +271,9 @@ "ExpectedInstructionCount": 7, "Comment": "0x0f 0x4a", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "mrs x21, nzcv", "tst w20, #0x1", "csel w4, w6, w4, ne", @@ -284,9 +284,9 @@ "ExpectedInstructionCount": 7, "Comment": "0x0f 0x4a", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "mrs x21, nzcv", "tst w20, #0x1", "csel x4, x6, x4, ne", @@ -297,9 +297,9 @@ "ExpectedInstructionCount": 8, "Comment": "0x0f 0x4b", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "mrs x21, nzcv", "tst w20, #0x1", "csel w20, w6, w4, ne", @@ -311,9 +311,9 @@ "ExpectedInstructionCount": 7, "Comment": "0x0f 0x4b", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "mrs x21, nzcv", "tst w20, #0x1", "csel w4, w6, w4, ne", @@ -324,9 +324,9 @@ "ExpectedInstructionCount": 7, "Comment": "0x0f 0x4b", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "mrs x21, nzcv", "tst w20, #0x1", "csel x4, x6, x4, ne", @@ -505,10 +505,10 @@ "ExpectedInstructionCount": 5, "Comment": "0x0f 0x9a", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", - "and x20, x20, #0x1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", + "and w20, w20, #0x1", "bfxil x4, x20, #0, #8" ] }, @@ -516,10 +516,10 @@ "ExpectedInstructionCount": 5, "Comment": "0x0f 0x9b", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", - "and x20, x20, #0x1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", + "and w20, w20, #0x1", "bfxil x4, x20, #0, #8" ] }, diff --git a/unittests/InstructionCountCI/FlagM/x87.json b/unittests/InstructionCountCI/FlagM/x87.json index 2dc678dd7..2b9b8b118 100644 --- a/unittests/InstructionCountCI/FlagM/x87.json +++ b/unittests/InstructionCountCI/FlagM/x87.json @@ -6439,9 +6439,9 @@ "0xda 11b 0xd8 /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6460,9 +6460,9 @@ "0xda 11b 0xd9 /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6483,9 +6483,9 @@ "0xda 11b 0xda /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6506,9 +6506,9 @@ "0xda 11b 0xdb /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6529,9 +6529,9 @@ "0xda 11b 0xdc /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6552,9 +6552,9 @@ "0xda 11b 0xdd /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6575,9 +6575,9 @@ "0xda 11b 0xde /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6598,9 +6598,9 @@ "0xda 11b 0xdf /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7377,9 +7377,9 @@ "0xdb 11b 0xd8 /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7398,9 +7398,9 @@ "0xdb 11b 0xd9 /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7421,9 +7421,9 @@ "0xdb 11b 0xda /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7444,9 +7444,9 @@ "0xdb 11b 0xdb /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7467,9 +7467,9 @@ "0xdb 11b 0xdc /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7490,9 +7490,9 @@ "0xdb 11b 0xdd /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7513,9 +7513,9 @@ "0xdb 11b 0xde /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7536,9 +7536,9 @@ "0xdb 11b 0xdf /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", diff --git a/unittests/InstructionCountCI/FlagM/x87_f64.json b/unittests/InstructionCountCI/FlagM/x87_f64.json index 9d2b28fca..77a203b7c 100644 --- a/unittests/InstructionCountCI/FlagM/x87_f64.json +++ b/unittests/InstructionCountCI/FlagM/x87_f64.json @@ -3790,9 +3790,9 @@ "0xda 11b 0xd8 /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3811,9 +3811,9 @@ "0xda 11b 0xd9 /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3834,9 +3834,9 @@ "0xda 11b 0xda /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3857,9 +3857,9 @@ "0xda 11b 0xdb /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3880,9 +3880,9 @@ "0xda 11b 0xdc /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3903,9 +3903,9 @@ "0xda 11b 0xdd /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3926,9 +3926,9 @@ "0xda 11b 0xde /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3949,9 +3949,9 @@ "0xda 11b 0xdf /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4658,9 +4658,9 @@ "0xdb 11b 0xd8 /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4679,9 +4679,9 @@ "0xdb 11b 0xd9 /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4702,9 +4702,9 @@ "0xdb 11b 0xda /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4725,9 +4725,9 @@ "0xdb 11b 0xdb /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4748,9 +4748,9 @@ "0xdb 11b 0xdc /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4771,9 +4771,9 @@ "0xdb 11b 0xdd /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4794,9 +4794,9 @@ "0xdb 11b 0xde /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4817,9 +4817,9 @@ "0xdb 11b 0xdf /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", diff --git a/unittests/InstructionCountCI/Primary.json b/unittests/InstructionCountCI/Primary.json index 367d7d6cf..008fb8362 100644 --- a/unittests/InstructionCountCI/Primary.json +++ b/unittests/InstructionCountCI/Primary.json @@ -2625,9 +2625,9 @@ "orr x20, x20, x21, lsl #20", "ldrb w21, [x28, #997]", "orr x20, x20, x21, lsl #21", - "eor w21, w26, w26, lsr #4", - "eor w21, w21, w21, lsr #2", - "eor w21, w21, w21, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w21, w0, w0, lsr #1", "orr x21, x21, #0xfffffffffffffffe", "orn x20, x20, x21, ror #62", "mrs x21, nzcv", @@ -2670,9 +2670,9 @@ "orr x20, x20, x21, lsl #20", "ldrb w21, [x28, #997]", "orr x20, x20, x21, lsl #21", - "eor w21, w26, w26, lsr #4", - "eor w21, w21, w21, lsr #2", - "eor w21, w21, w21, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w21, w0, w0, lsr #1", "orr x21, x21, #0xfffffffffffffffe", "orn x20, x20, x21, ror #62", "mrs x21, nzcv", @@ -2754,9 +2754,9 @@ "eor w21, w27, w26", "ubfx w21, w21, #4, #1", "orr x20, x20, x21, lsl #4", - "eor w21, w26, w26, lsr #4", - "eor w21, w21, w21, lsr #2", - "eor w21, w21, w21, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w21, w0, w0, lsr #1", "orr x21, x21, #0xfffffffffffffffe", "orn x20, x20, x21, ror #62", "mrs x21, nzcv", diff --git a/unittests/InstructionCountCI/Secondary.json b/unittests/InstructionCountCI/Secondary.json index 6fe5947dd..92cbb0127 100644 --- a/unittests/InstructionCountCI/Secondary.json +++ b/unittests/InstructionCountCI/Secondary.json @@ -210,8 +210,8 @@ "fcmp s16, s17", "mov w27, #0x0", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "comiss xmm0, xmm1": { @@ -221,8 +221,8 @@ "fcmp s16, s17", "mov w27, #0x0", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "rdtsc": { @@ -459,9 +459,9 @@ "ExpectedInstructionCount": 8, "Comment": "0x0f 0x4a", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "mrs x21, nzcv", "tst w20, #0x1", "csel w20, w6, w4, ne", @@ -473,9 +473,9 @@ "ExpectedInstructionCount": 7, "Comment": "0x0f 0x4a", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "mrs x21, nzcv", "tst w20, #0x1", "csel w4, w6, w4, ne", @@ -486,9 +486,9 @@ "ExpectedInstructionCount": 7, "Comment": "0x0f 0x4a", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "mrs x21, nzcv", "tst w20, #0x1", "csel x4, x6, x4, ne", @@ -499,9 +499,9 @@ "ExpectedInstructionCount": 8, "Comment": "0x0f 0x4b", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "mrs x21, nzcv", "tst w20, #0x1", "csel w20, w6, w4, ne", @@ -513,9 +513,9 @@ "ExpectedInstructionCount": 7, "Comment": "0x0f 0x4b", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "mrs x21, nzcv", "tst w20, #0x1", "csel w4, w6, w4, ne", @@ -526,9 +526,9 @@ "ExpectedInstructionCount": 7, "Comment": "0x0f 0x4b", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "mrs x21, nzcv", "tst w20, #0x1", "csel x4, x6, x4, ne", @@ -1222,10 +1222,10 @@ "ExpectedInstructionCount": 5, "Comment": "0x0f 0x9a", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", - "and x20, x20, #0x1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", + "and w20, w20, #0x1", "bfxil x4, x20, #0, #8" ] }, @@ -1233,10 +1233,10 @@ "ExpectedInstructionCount": 5, "Comment": "0x0f 0x9b", "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", - "and x20, x20, #0x1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", + "and w20, w20, #0x1", "bfxil x4, x20, #0, #8" ] }, diff --git a/unittests/InstructionCountCI/Secondary_OpSize.json b/unittests/InstructionCountCI/Secondary_OpSize.json index 0d2cd86c3..fb7baad18 100644 --- a/unittests/InstructionCountCI/Secondary_OpSize.json +++ b/unittests/InstructionCountCI/Secondary_OpSize.json @@ -148,8 +148,8 @@ "fcmp d16, d17", "mov w27, #0x0", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "comisd xmm0, xmm1": { @@ -159,8 +159,8 @@ "fcmp d16, d17", "mov w27, #0x0", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "movmskpd eax, xmm0": { diff --git a/unittests/InstructionCountCI/VEX_map1.json b/unittests/InstructionCountCI/VEX_map1.json index 5eea16893..2ad2dc6fd 100644 --- a/unittests/InstructionCountCI/VEX_map1.json +++ b/unittests/InstructionCountCI/VEX_map1.json @@ -3361,8 +3361,8 @@ "fcmp s16, s17", "mov w27, #0x0", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "vucomisd xmm0, xmm1": { @@ -3374,8 +3374,8 @@ "fcmp d16, d17", "mov w27, #0x0", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "vcomiss xmm0, xmm1": { @@ -3387,8 +3387,8 @@ "fcmp s16, s17", "mov w27, #0x0", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "vcomisd xmm0, xmm1": { @@ -3400,8 +3400,8 @@ "fcmp d16, d17", "mov w27, #0x0", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "vaddps xmm0, xmm1, xmm2": { diff --git a/unittests/InstructionCountCI/x87.json b/unittests/InstructionCountCI/x87.json index 8f0540d35..f6f1067a8 100644 --- a/unittests/InstructionCountCI/x87.json +++ b/unittests/InstructionCountCI/x87.json @@ -6438,9 +6438,9 @@ "0xda 11b 0xd8 /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6459,9 +6459,9 @@ "0xda 11b 0xd9 /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6482,9 +6482,9 @@ "0xda 11b 0xda /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6505,9 +6505,9 @@ "0xda 11b 0xdb /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6528,9 +6528,9 @@ "0xda 11b 0xdc /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6551,9 +6551,9 @@ "0xda 11b 0xdd /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6574,9 +6574,9 @@ "0xda 11b 0xde /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -6597,9 +6597,9 @@ "0xda 11b 0xdf /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7376,9 +7376,9 @@ "0xdb 11b 0xd8 /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7397,9 +7397,9 @@ "0xdb 11b 0xd9 /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7420,9 +7420,9 @@ "0xdb 11b 0xda /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7443,9 +7443,9 @@ "0xdb 11b 0xdb /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7466,9 +7466,9 @@ "0xdb 11b 0xdc /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7489,9 +7489,9 @@ "0xdb 11b 0xdd /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7512,9 +7512,9 @@ "0xdb 11b 0xde /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -7535,9 +7535,9 @@ "0xdb 11b 0xdf /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", diff --git a/unittests/InstructionCountCI/x87_f64.json b/unittests/InstructionCountCI/x87_f64.json index af6213bf5..b91037d20 100644 --- a/unittests/InstructionCountCI/x87_f64.json +++ b/unittests/InstructionCountCI/x87_f64.json @@ -3810,9 +3810,9 @@ "0xda 11b 0xd8 /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3831,9 +3831,9 @@ "0xda 11b 0xd9 /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3854,9 +3854,9 @@ "0xda 11b 0xda /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3877,9 +3877,9 @@ "0xda 11b 0xdb /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3900,9 +3900,9 @@ "0xda 11b 0xdc /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3923,9 +3923,9 @@ "0xda 11b 0xdd /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3946,9 +3946,9 @@ "0xda 11b 0xde /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -3969,9 +3969,9 @@ "0xda 11b 0xdf /1" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eon w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eon w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4679,9 +4679,9 @@ "0xdb 11b 0xd8 /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4700,9 +4700,9 @@ "0xdb 11b 0xd9 /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4723,9 +4723,9 @@ "0xdb 11b 0xda /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4746,9 +4746,9 @@ "0xdb 11b 0xdb /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4769,9 +4769,9 @@ "0xdb 11b 0xdc /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4792,9 +4792,9 @@ "0xdb 11b 0xdd /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4815,9 +4815,9 @@ "0xdb 11b 0xde /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4838,9 +4838,9 @@ "0xdb 11b 0xdf /3" ], "ExpectedArm64ASM": [ - "eor w20, w26, w26, lsr #4", - "eor w20, w20, w20, lsr #2", - "eor w20, w20, w20, lsr #1", + "eor w0, w26, w26, lsr #4", + "eor w0, w0, w0, lsr #2", + "eor w20, w0, w0, lsr #1", "sbfx x20, x20, #0, #1", "dup v2.2d, x20", "ldrb w20, [x28, #1019]", @@ -4899,8 +4899,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fucomi st0, st1": { @@ -4918,8 +4918,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fucomi st0, st2": { @@ -4937,8 +4937,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fucomi st0, st3": { @@ -4956,8 +4956,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fucomi st0, st4": { @@ -4975,8 +4975,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fucomi st0, st5": { @@ -4994,8 +4994,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fucomi st0, st6": { @@ -5013,8 +5013,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fucomi st0, st7": { @@ -5032,8 +5032,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fcomi st0, st0": { @@ -5049,8 +5049,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fcomi st0, st1": { @@ -5068,8 +5068,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fcomi st0, st2": { @@ -5087,8 +5087,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fcomi st0, st3": { @@ -5106,8 +5106,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fcomi st0, st4": { @@ -5125,8 +5125,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fcomi st0, st5": { @@ -5144,8 +5144,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fcomi st0, st6": { @@ -5163,8 +5163,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fcomi st0, st7": { @@ -5182,8 +5182,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x20, eq", - "ccmn x26, x20, #nzCv, le" + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le" ] }, "fadd qword [rax]": { @@ -9693,8 +9693,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -9720,8 +9720,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -9747,8 +9747,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -9774,8 +9774,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -9801,8 +9801,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -9828,8 +9828,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -9855,8 +9855,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -9882,8 +9882,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -9907,8 +9907,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -9934,8 +9934,8 @@ "ldr d3, [x0, #1040]", "fcmp d3, d2", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -9961,8 +9961,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -9988,8 +9988,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -10015,8 +10015,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -10042,8 +10042,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -10069,8 +10069,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21", @@ -10096,8 +10096,8 @@ "fcmp d3, d2", "mov w21, #0x1", "cset w26, vc", - "csetm x22, eq", - "ccmn x26, x22, #nzCv, le", + "csetm x0, eq", + "ccmn x26, x0, #nzCv, le", "ldrb w22, [x28, #1298]", "lsl w21, w21, w20", "bic w21, w22, w21",