From 99465faf63bb53164cbae724e49c4dd8db24d083 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Mon, 23 Oct 2023 09:02:04 -0700 Subject: [PATCH 1/5] IR: Implements support for subtract with shifted register Will be used soon. --- FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp | 10 ++++++++++ FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h | 9 +++++++++ FEXCore/Source/Interface/IR/IR.json | 13 ++++++++++++- FEXCore/Source/Interface/IR/IRDumper.cpp | 10 ++++++++++ FEXCore/include/FEXCore/IR/IR.h | 7 +++++++ 5 files changed, 48 insertions(+), 1 deletion(-) diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp b/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp index 5141cf2f7..443a92608 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp @@ -187,6 +187,16 @@ DEF_OP(Sub) { } } +DEF_OP(SubShift) { + auto Op = IROp->C(); + const uint8_t OpSize = IROp->Size; + + LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize); + const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit; + + sub(EmitSize, GetReg(Node), GetReg(Op->Src1.ID()), GetReg(Op->Src2.ID()), ConvertIRShiftType(Op->Shift), Op->ShiftAmount); +} + DEF_OP(SubNZCV) { auto Op = IROp->C(); const IR::OpSize OpSize = Op->Size; diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h b/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h index 088210168..2926255f3 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h @@ -116,6 +116,15 @@ private: return PhyReg; } + // Converts IR-base shift type to ARMEmitter shift type. + // Will be a no-op, only a type conversion since the two definitions match. + [[nodiscard]] ARMEmitter::ShiftType ConvertIRShiftType(IR::ShiftType Shift) const { + return Shift == IR::ShiftType::LSL ? ARMEmitter::ShiftType::LSL : + Shift == IR::ShiftType::LSR ? ARMEmitter::ShiftType::LSR : + Shift == IR::ShiftType::ASR ? ARMEmitter::ShiftType::ASR : + ARMEmitter::ShiftType::ROR; + } + [[nodiscard]] bool IsFPR(IR::NodeID Node) const; [[nodiscard]] bool IsGPR(IR::NodeID Node) const; [[nodiscard]] bool IsGPRPair(IR::NodeID Node) const; diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index 1ce60b909..3c2361f70 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -158,7 +158,8 @@ "RoundType": "RoundType", "FloatCompareOp": "FloatCompareOp", "NamedVectorConstant": "FEXCore::IR::NamedVectorConstant", - "IndexNamedVectorConstant": "FEXCore::IR::IndexNamedVectorConstant" + "IndexNamedVectorConstant": "FEXCore::IR::IndexNamedVectorConstant", + "ShiftType": "FEXCore::IR::ShiftType" }, "Ops": { "Misc": { @@ -953,6 +954,16 @@ "Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit" ] }, + "GPR = SubShift OpSize:#Size, GPR:$Src1, GPR:$Src2, ShiftType:$Shift{ShiftType::LSL}, u8:$ShiftAmount{0}": { + "Desc": [ "Integer Sub with shifted register", + "Will truncate to 64 or 32bits" + ], + "DestSize": "Size", + "EmitValidation": [ + "Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit", + "_Shift != ShiftType::ROR" + ] + }, "GPR = SubNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, u8:$InvertCarry": { "Desc": ["Return NZCV for the difference of two GPRs. ", "If InvertCarry is nonzero, carry flag uses x86 definition, inverted from arm64.", diff --git a/FEXCore/Source/Interface/IR/IRDumper.cpp b/FEXCore/Source/Interface/IR/IRDumper.cpp index 68cd2497b..dedaf3c1a 100644 --- a/FEXCore/Source/Interface/IR/IRDumper.cpp +++ b/FEXCore/Source/Interface/IR/IRDumper.cpp @@ -257,6 +257,16 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const *out << static_cast(Arg.si_code) << "}"; } +static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const* IR, FEXCore::IR::ShiftType Arg) { + switch (Arg) { + case ShiftType::LSL: *out << "LSL"; break; + case ShiftType::LSR: *out << "LSR"; break; + case ShiftType::ASR: *out << "ASR"; break; + case ShiftType::ROR: *out << "ROR"; break; + default: *out << ""; break; + } +} + void Dump(fextl::stringstream *out, IRListView const* IR, IR::RegisterAllocationData *RAData) { auto HeaderOp = IR->GetHeader(); diff --git a/FEXCore/include/FEXCore/IR/IR.h b/FEXCore/include/FEXCore/IR/IR.h index 5c130e31c..6d1c25daf 100644 --- a/FEXCore/include/FEXCore/IR/IR.h +++ b/FEXCore/include/FEXCore/IR/IR.h @@ -564,6 +564,13 @@ enum class FloatCompareOp : uint8_t { ORD, }; +enum class ShiftType : uint8_t { + LSL = 0, + LSR, + ASR, + ROR, +}; + // Converts a size stored as an integer in to an OpSize enum. // This is a nop operation and will be eliminated by the compiler. static inline OpSize SizeToOpSize(uint8_t Size) { From 95c756b466c4c7f099c9b3d1a51e399e506169ad Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Mon, 23 Oct 2023 09:02:27 -0700 Subject: [PATCH 2/5] OpcodeDispatcher: Optimize DF pointer offset calculation Previously this moved two constant, did a compare and a csel. Four instructions in total. It also corrupts NZCV which we want to use for other things. This new codegen emits one constant and one subtract instruction, two instructions total and doesn't touch NZCV. More optimal! --- .../Interface/Core/OpcodeDispatcher.cpp | 54 ++++++------------- 1 file changed, 15 insertions(+), 39 deletions(-) diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index b6e26f661..4471a4e8c 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -3858,14 +3858,10 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) { // Store to memory where RDI points _StoreMemAutoTSO(GPRClass, Size, Dest, Src, Size); - auto SizeConst = _Constant(Size); - auto NegSizeConst = _Constant(-Size); - // Calculate direction. auto DF = GetRFLAG(FEXCore::X86State::RFLAG_DF_LOC); - auto PtrDir = _Select(FEXCore::IR::COND_EQ, - DF, _Constant(0), - SizeConst, NegSizeConst); + auto SizeConst = _Constant(Size); + auto PtrDir = _SubShift(IR::SizeToOpSize(CTX->GetGPRSize()), SizeConst, DF, ShiftType::LSL, FEXCore::ilog2(Size) + 1); // Offset the pointer OrderedNode *TailDest = LoadGPRRegister(X86State::REG_RDI); @@ -3927,8 +3923,7 @@ void OpDispatchBuilder::MOVSOp(OpcodeArgs) { } else { auto SizeConst = _Constant(Size); - auto NegSizeConst = _Constant(-Size); - auto PtrDir = _Select(FEXCore::IR::COND_EQ, DF, _Constant(0), SizeConst, NegSizeConst); + auto PtrDir = _SubShift(IR::SizeToOpSize(CTX->GetGPRSize()), SizeConst, DF, ShiftType::LSL, FEXCore::ilog2(Size) + 1); OrderedNode *RSI = LoadGPRRegister(X86State::REG_RSI); OrderedNode *RDI = LoadGPRRegister(X86State::REG_RDI); @@ -3974,9 +3969,8 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) { GenerateFlags_SUB(Op, Result, Src2, Src1); auto DF = GetRFLAG(FEXCore::X86State::RFLAG_DF_LOC); - auto PtrDir = _Select(FEXCore::IR::COND_EQ, - DF, _Constant(0), - _Constant(Size), _Constant(-Size)); + auto SizeConst = _Constant(Size); + auto PtrDir = _SubShift(IR::SizeToOpSize(CTX->GetGPRSize()), SizeConst, DF, ShiftType::LSL, FEXCore::ilog2(Size) + 1); // Offset the pointer Dest_RDI = _Add(OpSize::i64Bit, Dest_RDI, PtrDir); @@ -3994,9 +3988,8 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) { // read DF once auto DF = GetRFLAG(FEXCore::X86State::RFLAG_DF_LOC); - auto PtrDir = _Select(FEXCore::IR::COND_EQ, - DF, _Constant(0), - _Constant(Size), _Constant(-Size)); + auto SizeConst = _Constant(Size); + auto PtrDir = _SubShift(IR::SizeToOpSize(CTX->GetGPRSize()), SizeConst, DF, ShiftType::LSL, FEXCore::ilog2(Size) + 1); auto JumpStart = _Jump(); // Make sure to start a new block after ending this one @@ -4088,13 +4081,9 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) { StoreResult(GPRClass, Op, Src, -1); - auto SizeConst = _Constant(Size); - auto NegSizeConst = _Constant(-Size); - auto DF = GetRFLAG(FEXCore::X86State::RFLAG_DF_LOC); - auto PtrDir = _Select(FEXCore::IR::COND_EQ, - DF, _Constant(0), - SizeConst, NegSizeConst); + auto SizeConst = _Constant(Size); + auto PtrDir = _SubShift(IR::SizeToOpSize(CTX->GetGPRSize()), SizeConst, DF, ShiftType::LSL, FEXCore::ilog2(Size) + 1); // Offset the pointer OrderedNode *TailDest_RSI = LoadGPRRegister(X86State::REG_RSI); @@ -4113,13 +4102,9 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) { // May or may not matter // Read DF once - auto SizeConst = _Constant(Size); - auto NegSizeConst = _Constant(-Size); - auto DF = GetRFLAG(FEXCore::X86State::RFLAG_DF_LOC); - auto PtrDir = _Select(FEXCore::IR::COND_EQ, - DF, _Constant(0), - SizeConst, NegSizeConst); + auto SizeConst = _Constant(Size); + auto PtrDir = _SubShift(IR::SizeToOpSize(CTX->GetGPRSize()), SizeConst, DF, ShiftType::LSL, FEXCore::ilog2(Size) + 1); auto JumpStart = _Jump(); // Make sure to start a new block after ending this one @@ -4194,13 +4179,9 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) { OrderedNode* Result = _Sub(Size == 8 ? OpSize::i64Bit : OpSize::i32Bit, Src1, Src2); GenerateFlags_SUB(Op, Result, Src1, Src2); - auto SizeConst = _Constant(Size); - auto NegSizeConst = _Constant(-Size); - auto DF = GetRFLAG(FEXCore::X86State::RFLAG_DF_LOC); - auto PtrDir = _Select(FEXCore::IR::COND_EQ, - DF, _Constant(0), - SizeConst, NegSizeConst); + auto SizeConst = _Constant(Size); + auto PtrDir = _SubShift(IR::SizeToOpSize(CTX->GetGPRSize()), SizeConst, DF, ShiftType::LSL, FEXCore::ilog2(Size) + 1); // Offset the pointer OrderedNode *TailDest_RDI = LoadGPRRegister(X86State::REG_RDI); @@ -4215,14 +4196,9 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) { bool REPE = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX; // read DF once - - auto SizeConst = _Constant(Size); - auto NegSizeConst = _Constant(-Size); - auto DF = GetRFLAG(FEXCore::X86State::RFLAG_DF_LOC); - auto PtrDir = _Select(FEXCore::IR::COND_EQ, - DF, _Constant(0), - SizeConst, NegSizeConst); + auto SizeConst = _Constant(Size); + auto PtrDir = _SubShift(IR::SizeToOpSize(CTX->GetGPRSize()), SizeConst, DF, ShiftType::LSL, FEXCore::ilog2(Size) + 1); auto JumpStart = _Jump(); // Make sure to start a new block after ending this one From dd5ca1d3494f086cc491364609b1ec8138efbd58 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Mon, 23 Oct 2023 09:04:03 -0700 Subject: [PATCH 3/5] InstCountCI: Update for DF pointer optimization. --- .../InstructionCountCI/FEXOpt/MultiInst.json | 175 +++++++++ unittests/InstructionCountCI/Primary.json | 352 +++++++----------- 2 files changed, 307 insertions(+), 220 deletions(-) diff --git a/unittests/InstructionCountCI/FEXOpt/MultiInst.json b/unittests/InstructionCountCI/FEXOpt/MultiInst.json index 5c9f57937..d1a633deb 100644 --- a/unittests/InstructionCountCI/FEXOpt/MultiInst.json +++ b/unittests/InstructionCountCI/FEXOpt/MultiInst.json @@ -59,6 +59,181 @@ "fadd s0, s16, s18", "mov v16.s[0], v0.s[0]" ] + }, + "positive movsb": { + "ExpectedInstructionCount": 8, + "Optimal": "No", + "Comment": [ + "When direction flag is a compile time constant we can optimize", + "loads and stores can turn in to post-increment when known" + ], + "x86Insts": [ + "cld", + "movsb" + ], + "ExpectedArm64ASM": [ + "mov w20, #0x0", + "strb w20, [x28, #714]", + "mov w21, #0x1", + "sub x20, x21, x20, lsl #1", + "ldrb w21, [x10]", + "strb w21, [x11]", + "add x10, x10, x20", + "add x11, x11, x20" + ] + }, + "positive movsw": { + "ExpectedInstructionCount": 8, + "Optimal": "No", + "Comment": [ + "When direction flag is a compile time constant we can optimize", + "loads and stores can turn in to post-increment when known" + ], + "x86Insts": [ + "cld", + "movsw" + ], + "ExpectedArm64ASM": [ + "mov w20, #0x0", + "strb w20, [x28, #714]", + "mov w21, #0x2", + "sub x20, x21, x20, lsl #2", + "ldrh w21, [x10]", + "strh w21, [x11]", + "add x10, x10, x20", + "add x11, x11, x20" + ] + }, + "positive movsd": { + "ExpectedInstructionCount": 8, + "Optimal": "No", + "Comment": [ + "When direction flag is a compile time constant we can optimize", + "loads and stores can turn in to post-increment when known" + ], + "x86Insts": [ + "cld", + "movsd" + ], + "ExpectedArm64ASM": [ + "mov w20, #0x0", + "strb w20, [x28, #714]", + "mov w21, #0x4", + "sub x20, x21, x20, lsl #3", + "ldr w21, [x10]", + "str w21, [x11]", + "add x10, x10, x20", + "add x11, x11, x20" + ] + }, + "positive movsq": { + "ExpectedInstructionCount": 8, + "Optimal": "No", + "Comment": [ + "When direction flag is a compile time constant we can optimize", + "loads and stores can turn in to post-increment when known" + ], + "x86Insts": [ + "cld", + "movsq" + ], + "ExpectedArm64ASM": [ + "mov w20, #0x0", + "strb w20, [x28, #714]", + "mov w21, #0x8", + "sub x20, x21, x20, lsl #4", + "ldr x21, [x10]", + "str x21, [x11]", + "add x10, x10, x20", + "add x11, x11, x20" + ] + }, + "negative movsb": { + "ExpectedInstructionCount": 7, + "Optimal": "No", + "Comment": [ + "When direction flag is a compile time constant we can optimize", + "loads and stores can turn in to post-increment when known" + ], + "x86Insts": [ + "std", + "movsb" + ], + "ExpectedArm64ASM": [ + "mov w20, #0x1", + "strb w20, [x28, #714]", + "sub x20, x20, x20, lsl #1", + "ldrb w21, [x10]", + "strb w21, [x11]", + "add x10, x10, x20", + "add x11, x11, x20" + ] + }, + "negative movsw": { + "ExpectedInstructionCount": 8, + "Optimal": "No", + "Comment": [ + "When direction flag is a compile time constant we can optimize", + "loads and stores can turn in to post-increment when known" + ], + "x86Insts": [ + "std", + "movsw" + ], + "ExpectedArm64ASM": [ + "mov w20, #0x1", + "strb w20, [x28, #714]", + "mov w21, #0x2", + "sub x20, x21, x20, lsl #2", + "ldrh w21, [x10]", + "strh w21, [x11]", + "add x10, x10, x20", + "add x11, x11, x20" + ] + }, + "negative movsd": { + "ExpectedInstructionCount": 8, + "Optimal": "No", + "Comment": [ + "When direction flag is a compile time constant we can optimize", + "loads and stores can turn in to post-increment when known" + ], + "x86Insts": [ + "std", + "movsd" + ], + "ExpectedArm64ASM": [ + "mov w20, #0x1", + "strb w20, [x28, #714]", + "mov w21, #0x4", + "sub x20, x21, x20, lsl #3", + "ldr w21, [x10]", + "str w21, [x11]", + "add x10, x10, x20", + "add x11, x11, x20" + ] + }, + "negative movsq": { + "ExpectedInstructionCount": 8, + "Optimal": "No", + "Comment": [ + "When direction flag is a compile time constant we can optimize", + "loads and stores can turn in to post-increment when known" + ], + "x86Insts": [ + "std", + "movsq" + ], + "ExpectedArm64ASM": [ + "mov w20, #0x1", + "strb w20, [x28, #714]", + "mov w21, #0x8", + "sub x20, x21, x20, lsl #4", + "ldr x21, [x10]", + "str x21, [x11]", + "add x10, x10, x20", + "add x11, x11, x20" + ] } } } diff --git a/unittests/InstructionCountCI/Primary.json b/unittests/InstructionCountCI/Primary.json index 61d216f35..d53bc30ca 100644 --- a/unittests/InstructionCountCI/Primary.json +++ b/unittests/InstructionCountCI/Primary.json @@ -3643,18 +3643,15 @@ ] }, "movsb": { - "ExpectedInstructionCount": 9, - "Optimal": "No", + "ExpectedInstructionCount": 7, + "Optimal": "Yes", "Comment": [ - "Direction flag increment/decrement location can be made with a single sbfx", "0xa4" ], "ExpectedArm64ASM": [ "ldrb w20, [x28, #714]", "mov w21, #0x1", - "mov x22, #0xffffffffffffffff", - "cmp x20, #0x0 (0)", - "csel x20, x21, x22, eq", + "sub x20, x21, x20, lsl #1", "ldrb w21, [x10]", "strb w21, [x11]", "add x10, x10, x20", @@ -3662,18 +3659,15 @@ ] }, "movsw": { - "ExpectedInstructionCount": 9, - "Optimal": "No", + "ExpectedInstructionCount": 7, + "Optimal": "Yes", "Comment": [ - "Direction flag increment/decrement location can be made with a tst+mov+csel triple", "0xa5" ], "ExpectedArm64ASM": [ "ldrb w20, [x28, #714]", "mov w21, #0x2", - "mov x22, #0xfffffffffffffffe", - "cmp x20, #0x0 (0)", - "csel x20, x21, x22, eq", + "sub x20, x21, x20, lsl #2", "ldrh w21, [x10]", "strh w21, [x11]", "add x10, x10, x20", @@ -3681,18 +3675,15 @@ ] }, "movsd": { - "ExpectedInstructionCount": 9, - "Optimal": "No", + "ExpectedInstructionCount": 7, + "Optimal": "Yes", "Comment": [ - "Direction flag increment/decrement location can be made with a tst+mov+csel triple", "0xa5" ], "ExpectedArm64ASM": [ "ldrb w20, [x28, #714]", "mov w21, #0x4", - "mov x22, #0xfffffffffffffffc", - "cmp x20, #0x0 (0)", - "csel x20, x21, x22, eq", + "sub x20, x21, x20, lsl #3", "ldr w21, [x10]", "str w21, [x11]", "add x10, x10, x20", @@ -3700,18 +3691,15 @@ ] }, "movsq": { - "ExpectedInstructionCount": 9, - "Optimal": "No", + "ExpectedInstructionCount": 7, + "Optimal": "Yes", "Comment": [ - "Direction flag increment/decrement location can be made with a tst+mov+csel triple", "0xa5" ], "ExpectedArm64ASM": [ "ldrb w20, [x28, #714]", "mov w21, #0x8", - "mov x22, #0xfffffffffffffff8", - "cmp x20, #0x0 (0)", - "csel x20, x21, x22, eq", + "sub x20, x21, x20, lsl #4", "ldr x21, [x10]", "str x21, [x11]", "add x10, x10, x20", @@ -3863,10 +3851,9 @@ ] }, "cmpsb": { - "ExpectedInstructionCount": 24, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ - "Direction flag increment/decrement location can be made with a single sbfx", "0xa6" ], "ExpectedArm64ASM": [ @@ -3875,9 +3862,7 @@ "sub w22, w21, w20", "ldrb w23, [x28, #714]", "mov w24, #0x1", - "mov x25, #0xffffffffffffffff", - "cmp x23, #0x0 (0)", - "csel x23, x24, x25, eq", + "sub x23, x24, x23, lsl #1", "add x11, x11, x23", "add x10, x10, x23", "eor w23, w21, w20", @@ -3897,10 +3882,9 @@ ] }, "cmpsw": { - "ExpectedInstructionCount": 24, + "ExpectedInstructionCount": 22, "Optimal": "No", "Comment": [ - "Direction flag increment/decrement location can be made with a tst+mov+csel triple", "0xa7" ], "ExpectedArm64ASM": [ @@ -3909,9 +3893,7 @@ "sub w22, w21, w20", "ldrb w23, [x28, #714]", "mov w24, #0x2", - "mov x25, #0xfffffffffffffffe", - "cmp x23, #0x0 (0)", - "csel x23, x24, x25, eq", + "sub x23, x24, x23, lsl #2", "add x11, x11, x23", "add x10, x10, x23", "eor w23, w21, w20", @@ -3931,10 +3913,9 @@ ] }, "cmpsd": { - "ExpectedInstructionCount": 17, + "ExpectedInstructionCount": 15, "Optimal": "No", "Comment": [ - "Direction flag increment/decrement location can be made with a tst+mov+csel triple", "0xa7" ], "ExpectedArm64ASM": [ @@ -3943,9 +3924,7 @@ "sub w22, w21, w20", "ldrb w23, [x28, #714]", "mov w24, #0x4", - "mov x25, #0xfffffffffffffffc", - "cmp x23, #0x0 (0)", - "csel x23, x24, x25, eq", + "sub x23, x24, x23, lsl #3", "add x11, x11, x23", "add x10, x10, x23", "eor w23, w21, w20", @@ -3958,10 +3937,9 @@ ] }, "cmpsq": { - "ExpectedInstructionCount": 17, + "ExpectedInstructionCount": 15, "Optimal": "No", "Comment": [ - "Direction flag increment/decrement location can be made with a tst+mov+csel triple", "0xa7" ], "ExpectedArm64ASM": [ @@ -3970,9 +3948,7 @@ "sub x22, x21, x20", "ldrb w23, [x28, #714]", "mov w24, #0x8", - "mov x25, #0xfffffffffffffff8", - "cmp x23, #0x0 (0)", - "csel x23, x24, x25, eq", + "sub x23, x24, x23, lsl #4", "add x11, x11, x23", "add x10, x10, x23", "eor x23, x21, x20", @@ -3985,15 +3961,13 @@ ] }, "repz cmpsb": { - "ExpectedInstructionCount": 28, + "ExpectedInstructionCount": 26, "Optimal": "No", "Comment": "0xa6", "ExpectedArm64ASM": [ "ldrb w20, [x28, #714]", "mov w21, #0x1", - "mov x22, #0xffffffffffffffff", - "cmp x20, #0x0 (0)", - "csel x20, x21, x22, eq", + "sub x20, x21, x20, lsl #1", "cbz x5, #+0x5c", "ldrb w21, [x11]", "ldrb w22, [x10]", @@ -4020,15 +3994,13 @@ ] }, "repz cmpsw": { - "ExpectedInstructionCount": 28, + "ExpectedInstructionCount": 26, "Optimal": "No", "Comment": "0xa7", "ExpectedArm64ASM": [ "ldrb w20, [x28, #714]", "mov w21, #0x2", - "mov x22, #0xfffffffffffffffe", - "cmp x20, #0x0 (0)", - "csel x20, x21, x22, eq", + "sub x20, x21, x20, lsl #2", "cbz x5, #+0x5c", "ldrh w21, [x11]", "ldrh w22, [x10]", @@ -4055,15 +4027,13 @@ ] }, "repz cmpsd": { - "ExpectedInstructionCount": 21, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": "0xa7", "ExpectedArm64ASM": [ "ldrb w20, [x28, #714]", "mov w21, #0x4", - "mov x22, #0xfffffffffffffffc", - "cmp x20, #0x0 (0)", - "csel x20, x21, x22, eq", + "sub x20, x21, x20, lsl #3", "cbz x5, #+0x40", "ldr w21, [x11]", "ldr w22, [x10]", @@ -4083,15 +4053,13 @@ ] }, "repz cmpsq": { - "ExpectedInstructionCount": 21, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": "0xa7", "ExpectedArm64ASM": [ "ldrb w20, [x28, #714]", "mov w21, #0x8", - "mov x22, #0xfffffffffffffff8", - "cmp x20, #0x0 (0)", - "csel x20, x21, x22, eq", + "sub x20, x21, x20, lsl #4", "cbz x5, #+0x40", "ldr x21, [x11]", "ldr x22, [x10]", @@ -4111,15 +4079,13 @@ ] }, "repnz cmpsb": { - "ExpectedInstructionCount": 28, + "ExpectedInstructionCount": 26, "Optimal": "No", "Comment": "0xa6", "ExpectedArm64ASM": [ "ldrb w20, [x28, #714]", "mov w21, #0x1", - "mov x22, #0xffffffffffffffff", - "cmp x20, #0x0 (0)", - "csel x20, x21, x22, eq", + "sub x20, x21, x20, lsl #1", "cbz x5, #+0x5c", "ldrb w21, [x11]", "ldrb w22, [x10]", @@ -4146,15 +4112,13 @@ ] }, "repnz cmpsw": { - "ExpectedInstructionCount": 28, + "ExpectedInstructionCount": 26, "Optimal": "No", "Comment": "0xa7", "ExpectedArm64ASM": [ "ldrb w20, [x28, #714]", "mov w21, #0x2", - "mov x22, #0xfffffffffffffffe", - "cmp x20, #0x0 (0)", - "csel x20, x21, x22, eq", + "sub x20, x21, x20, lsl #2", "cbz x5, #+0x5c", "ldrh w21, [x11]", "ldrh w22, [x10]", @@ -4181,15 +4145,13 @@ ] }, "repnz cmpsd": { - "ExpectedInstructionCount": 21, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": "0xa7", "ExpectedArm64ASM": [ "ldrb w20, [x28, #714]", "mov w21, #0x4", - "mov x22, #0xfffffffffffffffc", - "cmp x20, #0x0 (0)", - "csel x20, x21, x22, eq", + "sub x20, x21, x20, lsl #3", "cbz x5, #+0x40", "ldr w21, [x11]", "ldr w22, [x10]", @@ -4209,15 +4171,13 @@ ] }, "repnz cmpsq": { - "ExpectedInstructionCount": 21, + "ExpectedInstructionCount": 19, "Optimal": "No", "Comment": "0xa7", "ExpectedArm64ASM": [ "ldrb w20, [x28, #714]", "mov w21, #0x8", - "mov x22, #0xfffffffffffffff8", - "cmp x20, #0x0 (0)", - "csel x20, x21, x22, eq", + "sub x20, x21, x20, lsl #4", "cbz x5, #+0x40", "ldr x21, [x11]", "ldr x22, [x10]", @@ -4340,61 +4300,53 @@ ] }, "stosb": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": "0xaa", "ExpectedArm64ASM": [ "uxtb w20, w4", "strb w20, [x11]", - "mov w20, #0x1", - "mov x21, #0xffffffffffffffff", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x1", + "sub x20, x21, x20, lsl #1", "add x11, x11, x20" ] }, "stosw": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": "0xab", "ExpectedArm64ASM": [ "uxth w20, w4", "strh w20, [x11]", - "mov w20, #0x2", - "mov x21, #0xfffffffffffffffe", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x2", + "sub x20, x21, x20, lsl #2", "add x11, x11, x20" ] }, "stosd": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": "0xab", "ExpectedArm64ASM": [ "mov w20, w4", "str w20, [x11]", - "mov w20, #0x4", - "mov x21, #0xfffffffffffffffc", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x4", + "sub x20, x21, x20, lsl #3", "add x11, x11, x20" ] }, "stosq": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0xab", "ExpectedArm64ASM": [ "str x4, [x11]", - "mov w20, #0x8", - "mov x21, #0xfffffffffffffff8", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x8", + "sub x20, x21, x20, lsl #4", "add x11, x11, x20" ] }, @@ -4498,73 +4450,63 @@ ] }, "lodsb": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": "0xac", "ExpectedArm64ASM": [ "ldrb w20, [x10]", "bfxil x4, x20, #0, #8", - "mov w20, #0x1", - "mov x21, #0xffffffffffffffff", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x1", + "sub x20, x21, x20, lsl #1", "add x10, x10, x20" ] }, "lodsw": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": "0xad", "ExpectedArm64ASM": [ "ldrh w20, [x10]", "bfxil x4, x20, #0, #16", - "mov w20, #0x2", - "mov x21, #0xfffffffffffffffe", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x2", + "sub x20, x21, x20, lsl #2", "add x10, x10, x20" ] }, "lodsd": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0xad", "ExpectedArm64ASM": [ "ldr w4, [x10]", - "mov w20, #0x4", - "mov x21, #0xfffffffffffffffc", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x4", + "sub x20, x21, x20, lsl #3", "add x10, x10, x20" ] }, "lodsq": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 5, "Optimal": "No", "Comment": "0xad", "ExpectedArm64ASM": [ "ldr x4, [x10]", - "mov w20, #0x8", - "mov x21, #0xfffffffffffffff8", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x8", + "sub x20, x21, x20, lsl #4", "add x10, x10, x20" ] }, "rep lodsb": { - "ExpectedInstructionCount": 11, + "ExpectedInstructionCount": 9, "Optimal": "No", "Comment": "0xac", "ExpectedArm64ASM": [ - "mov w20, #0x1", - "mov x21, #0xffffffffffffffff", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x1", + "sub x20, x21, x20, lsl #1", "cbz x5, #+0x18", "ldrb w21, [x10]", "bfxil x4, x21, #0, #8", @@ -4574,15 +4516,13 @@ ] }, "rep lodsw": { - "ExpectedInstructionCount": 11, + "ExpectedInstructionCount": 9, "Optimal": "No", "Comment": "0xad", "ExpectedArm64ASM": [ - "mov w20, #0x2", - "mov x21, #0xfffffffffffffffe", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x2", + "sub x20, x21, x20, lsl #2", "cbz x5, #+0x18", "ldrh w21, [x10]", "bfxil x4, x21, #0, #16", @@ -4592,15 +4532,13 @@ ] }, "rep lodsd": { - "ExpectedInstructionCount": 10, + "ExpectedInstructionCount": 8, "Optimal": "No", "Comment": "0xad", "ExpectedArm64ASM": [ - "mov w20, #0x4", - "mov x21, #0xfffffffffffffffc", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x4", + "sub x20, x21, x20, lsl #3", "cbz x5, #+0x14", "ldr w4, [x10]", "sub x5, x5, #0x1 (1)", @@ -4609,15 +4547,13 @@ ] }, "rep lodsq": { - "ExpectedInstructionCount": 10, + "ExpectedInstructionCount": 8, "Optimal": "No", "Comment": "0xad", "ExpectedArm64ASM": [ - "mov w20, #0x8", - "mov x21, #0xfffffffffffffff8", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x8", + "sub x20, x21, x20, lsl #4", "cbz x5, #+0x14", "ldr x4, [x10]", "sub x5, x5, #0x1 (1)", @@ -4626,18 +4562,16 @@ ] }, "scasb": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 21, "Optimal": "No", "Comment": "0xae", "ExpectedArm64ASM": [ "uxtb w20, w4", "ldrb w21, [x11]", "sub w22, w20, w21", - "mov w23, #0x1", - "mov x24, #0xffffffffffffffff", - "ldrb w25, [x28, #714]", - "cmp x25, #0x0 (0)", - "csel x23, x23, x24, eq", + "ldrb w23, [x28, #714]", + "mov w24, #0x1", + "sub x23, x24, x23, lsl #1", "add x11, x11, x23", "eor w23, w20, w21", "strb w23, [x28, #708]", @@ -4656,18 +4590,16 @@ ] }, "scasw": { - "ExpectedInstructionCount": 23, + "ExpectedInstructionCount": 21, "Optimal": "No", "Comment": "0xaf", "ExpectedArm64ASM": [ "uxth w20, w4", "ldrh w21, [x11]", "sub w22, w20, w21", - "mov w23, #0x2", - "mov x24, #0xfffffffffffffffe", - "ldrb w25, [x28, #714]", - "cmp x25, #0x0 (0)", - "csel x23, x23, x24, eq", + "ldrb w23, [x28, #714]", + "mov w24, #0x2", + "sub x23, x24, x23, lsl #2", "add x11, x11, x23", "eor w23, w20, w21", "strb w23, [x28, #708]", @@ -4686,18 +4618,16 @@ ] }, "scasd": { - "ExpectedInstructionCount": 16, + "ExpectedInstructionCount": 14, "Optimal": "No", "Comment": "0xaf", "ExpectedArm64ASM": [ "mov w20, w4", "ldr w21, [x11]", "sub w22, w20, w21", - "mov w23, #0x4", - "mov x24, #0xfffffffffffffffc", - "ldrb w25, [x28, #714]", - "cmp x25, #0x0 (0)", - "csel x23, x23, x24, eq", + "ldrb w23, [x28, #714]", + "mov w24, #0x4", + "sub x23, x24, x23, lsl #3", "add x11, x11, x23", "eor w23, w20, w21", "strb w23, [x28, #708]", @@ -4709,17 +4639,15 @@ ] }, "scasq": { - "ExpectedInstructionCount": 15, + "ExpectedInstructionCount": 13, "Optimal": "No", "Comment": "0xaf", "ExpectedArm64ASM": [ "ldr x20, [x11]", "sub x21, x4, x20", - "mov w22, #0x8", - "mov x23, #0xfffffffffffffff8", - "ldrb w24, [x28, #714]", - "cmp x24, #0x0 (0)", - "csel x22, x22, x23, eq", + "ldrb w22, [x28, #714]", + "mov w23, #0x8", + "sub x22, x23, x22, lsl #4", "add x11, x11, x22", "eor x22, x4, x20", "strb w22, [x28, #708]", @@ -4731,15 +4659,13 @@ ] }, "repz scasb": { - "ExpectedInstructionCount": 27, + "ExpectedInstructionCount": 25, "Optimal": "No", "Comment": "0xae", "ExpectedArm64ASM": [ - "mov w20, #0x1", - "mov x21, #0xffffffffffffffff", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x1", + "sub x20, x21, x20, lsl #1", "cbz x5, #+0x58", "uxtb w21, w4", "ldrb w22, [x11]", @@ -4765,15 +4691,13 @@ ] }, "repz scasw": { - "ExpectedInstructionCount": 27, + "ExpectedInstructionCount": 25, "Optimal": "No", "Comment": "0xaf", "ExpectedArm64ASM": [ - "mov w20, #0x2", - "mov x21, #0xfffffffffffffffe", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x2", + "sub x20, x21, x20, lsl #2", "cbz x5, #+0x58", "uxth w21, w4", "ldrh w22, [x11]", @@ -4799,15 +4723,13 @@ ] }, "repz scasd": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": "0xaf", "ExpectedArm64ASM": [ - "mov w20, #0x4", - "mov x21, #0xfffffffffffffffc", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x4", + "sub x20, x21, x20, lsl #3", "cbz x5, #+0x3c", "mov w21, w4", "ldr w22, [x11]", @@ -4826,15 +4748,13 @@ ] }, "repz scasq": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": "0xaf", "ExpectedArm64ASM": [ - "mov w20, #0x8", - "mov x21, #0xfffffffffffffff8", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x8", + "sub x20, x21, x20, lsl #4", "cbz x5, #+0x38", "ldr x21, [x11]", "sub x22, x4, x21", @@ -4852,15 +4772,13 @@ ] }, "repnz scasb": { - "ExpectedInstructionCount": 27, + "ExpectedInstructionCount": 25, "Optimal": "No", "Comment": "0xae", "ExpectedArm64ASM": [ - "mov w20, #0x1", - "mov x21, #0xffffffffffffffff", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x1", + "sub x20, x21, x20, lsl #1", "cbz x5, #+0x58", "uxtb w21, w4", "ldrb w22, [x11]", @@ -4886,15 +4804,13 @@ ] }, "repnz scasw": { - "ExpectedInstructionCount": 27, + "ExpectedInstructionCount": 25, "Optimal": "No", "Comment": "0xaf", "ExpectedArm64ASM": [ - "mov w20, #0x2", - "mov x21, #0xfffffffffffffffe", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x2", + "sub x20, x21, x20, lsl #2", "cbz x5, #+0x58", "uxth w21, w4", "ldrh w22, [x11]", @@ -4920,15 +4836,13 @@ ] }, "repnz scasd": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 18, "Optimal": "No", "Comment": "0xaf", "ExpectedArm64ASM": [ - "mov w20, #0x4", - "mov x21, #0xfffffffffffffffc", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x4", + "sub x20, x21, x20, lsl #3", "cbz x5, #+0x3c", "mov w21, w4", "ldr w22, [x11]", @@ -4947,15 +4861,13 @@ ] }, "repnz scasq": { - "ExpectedInstructionCount": 19, + "ExpectedInstructionCount": 17, "Optimal": "No", "Comment": "0xaf", "ExpectedArm64ASM": [ - "mov w20, #0x8", - "mov x21, #0xfffffffffffffff8", - "ldrb w22, [x28, #714]", - "cmp x22, #0x0 (0)", - "csel x20, x20, x21, eq", + "ldrb w20, [x28, #714]", + "mov w21, #0x8", + "sub x20, x21, x20, lsl #4", "cbz x5, #+0x38", "ldr x21, [x11]", "sub x22, x4, x21", From 4466c50c2bbdbdf481b5946b25280a15f08c82a8 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Mon, 23 Oct 2023 10:36:33 -0700 Subject: [PATCH 4/5] ConstProp: Optimize SubShift and Add with negative When SubShift (LSL) occurs with both sources constant then optimize away the calculation. Additionally if add is found to have one immediate constant where the inverse of the constant fits in to ImmAddSub range, then invert the constant and change it in to a sub. This optimizes the cases when direction flag is known upfront in an instruction. --- .../Source/Interface/IR/Passes/ConstProp.cpp | 38 +++++++++++++++++-- 1 file changed, 35 insertions(+), 3 deletions(-) diff --git a/FEXCore/Source/Interface/IR/Passes/ConstProp.cpp b/FEXCore/Source/Interface/IR/Passes/ConstProp.cpp index 138823025..3847f0717 100644 --- a/FEXCore/Source/Interface/IR/Passes/ConstProp.cpp +++ b/FEXCore/Source/Interface/IR/Passes/ConstProp.cpp @@ -420,7 +420,6 @@ bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& C } break; } - case OP_AND: { // if AND's arguments are imms, they are masking for (int i = 0; i < IR::GetArgs(IROp->Op); i++) { @@ -648,13 +647,31 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current auto Op = IROp->C(); uint64_t Constant1{}; uint64_t Constant2{}; + bool IsConstant1 = IREmit->IsValueConstant(Op->Header.Args[0], &Constant1); + bool IsConstant2 = IREmit->IsValueConstant(Op->Header.Args[1], &Constant2); - if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) && - IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { + if (IsConstant1 && IsConstant2) { uint64_t NewConstant = (Constant1 + Constant2) & getMask(Op) ; IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } + else if (IsConstant2 && !IsImmAddSub(Constant2) && IsImmAddSub(-Constant2)) { + // If the second argument is constant, the immediate is not ImmAddSub, but when negated is. + // This means we can convert the operation in to a subtract. + // Change the IR operation itself. + IROp->Op = OP_SUB; + // Set the write cursor to just before this operation. + auto CodeIter = CurrentIR.at(CodeNode); + --CodeIter; + IREmit->SetWriteCursor(std::get<0>(*CodeIter)); + + // Negate the constant. + auto NegConstant = IREmit->_Constant(-Constant2); + + // Replace the second source with the negated constant. + IREmit->ReplaceNodeArgument(CodeNode, Op->Src2_Index, NegConstant); + Changed = true; + } break; } case OP_SUB: { @@ -670,6 +687,21 @@ bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& Current } break; } + case OP_SUBSHIFT: { + auto Op = IROp->C(); + + uint64_t Constant1, Constant2; + if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && + IREmit->IsValueConstant(IROp->Args[1], &Constant2) && + Op->Shift == IR::ShiftType::LSL) { + // Optimize the LSL case when we know both sources are constant. + // This is a pattern that shows up with direction flag calculations if DF was set just before the operation. + uint64_t NewConstant = (Constant1 - (Constant2 << Op->ShiftAmount)) & getMask(Op); + IREmit->ReplaceWithConstant(CodeNode, NewConstant); + Changed = true; + } + break; + } case OP_AND: { auto Op = IROp->CW(); uint64_t Constant1{}; From 2956e84ead7abfe13a64d3a7ece21528afe89d2e Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Mon, 23 Oct 2023 10:39:29 -0700 Subject: [PATCH 5/5] InstCountCI: Update for constant prop improvement --- .../InstructionCountCI/FEXOpt/MultiInst.json | 95 ++++++++----------- unittests/InstructionCountCI/Primary.json | 2 +- .../InstructionCountCI/PrimaryGroup.json | 4 +- 3 files changed, 43 insertions(+), 58 deletions(-) diff --git a/unittests/InstructionCountCI/FEXOpt/MultiInst.json b/unittests/InstructionCountCI/FEXOpt/MultiInst.json index d1a633deb..a8fd40c2b 100644 --- a/unittests/InstructionCountCI/FEXOpt/MultiInst.json +++ b/unittests/InstructionCountCI/FEXOpt/MultiInst.json @@ -61,7 +61,7 @@ ] }, "positive movsb": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": [ "When direction flag is a compile time constant we can optimize", @@ -74,16 +74,14 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "strb w20, [x28, #714]", - "mov w21, #0x1", - "sub x20, x21, x20, lsl #1", - "ldrb w21, [x10]", - "strb w21, [x11]", - "add x10, x10, x20", - "add x11, x11, x20" + "ldrb w20, [x10]", + "strb w20, [x11]", + "add x10, x10, #0x1 (1)", + "add x11, x11, #0x1 (1)" ] }, "positive movsw": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": [ "When direction flag is a compile time constant we can optimize", @@ -96,16 +94,14 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "strb w20, [x28, #714]", - "mov w21, #0x2", - "sub x20, x21, x20, lsl #2", - "ldrh w21, [x10]", - "strh w21, [x11]", - "add x10, x10, x20", - "add x11, x11, x20" + "ldrh w20, [x10]", + "strh w20, [x11]", + "add x10, x10, #0x2 (2)", + "add x11, x11, #0x2 (2)" ] }, "positive movsd": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": [ "When direction flag is a compile time constant we can optimize", @@ -118,16 +114,14 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "strb w20, [x28, #714]", - "mov w21, #0x4", - "sub x20, x21, x20, lsl #3", - "ldr w21, [x10]", - "str w21, [x11]", - "add x10, x10, x20", - "add x11, x11, x20" + "ldr w20, [x10]", + "str w20, [x11]", + "add x10, x10, #0x4 (4)", + "add x11, x11, #0x4 (4)" ] }, "positive movsq": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": [ "When direction flag is a compile time constant we can optimize", @@ -140,16 +134,14 @@ "ExpectedArm64ASM": [ "mov w20, #0x0", "strb w20, [x28, #714]", - "mov w21, #0x8", - "sub x20, x21, x20, lsl #4", - "ldr x21, [x10]", - "str x21, [x11]", - "add x10, x10, x20", - "add x11, x11, x20" + "ldr x20, [x10]", + "str x20, [x11]", + "add x10, x10, #0x8 (8)", + "add x11, x11, #0x8 (8)" ] }, "negative movsb": { - "ExpectedInstructionCount": 7, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": [ "When direction flag is a compile time constant we can optimize", @@ -162,15 +154,14 @@ "ExpectedArm64ASM": [ "mov w20, #0x1", "strb w20, [x28, #714]", - "sub x20, x20, x20, lsl #1", - "ldrb w21, [x10]", - "strb w21, [x11]", - "add x10, x10, x20", - "add x11, x11, x20" + "ldrb w20, [x10]", + "strb w20, [x11]", + "sub x10, x10, #0x1 (1)", + "sub x11, x11, #0x1 (1)" ] }, "negative movsw": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": [ "When direction flag is a compile time constant we can optimize", @@ -183,16 +174,14 @@ "ExpectedArm64ASM": [ "mov w20, #0x1", "strb w20, [x28, #714]", - "mov w21, #0x2", - "sub x20, x21, x20, lsl #2", - "ldrh w21, [x10]", - "strh w21, [x11]", - "add x10, x10, x20", - "add x11, x11, x20" + "ldrh w20, [x10]", + "strh w20, [x11]", + "sub x10, x10, #0x2 (2)", + "sub x11, x11, #0x2 (2)" ] }, "negative movsd": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": [ "When direction flag is a compile time constant we can optimize", @@ -205,16 +194,14 @@ "ExpectedArm64ASM": [ "mov w20, #0x1", "strb w20, [x28, #714]", - "mov w21, #0x4", - "sub x20, x21, x20, lsl #3", - "ldr w21, [x10]", - "str w21, [x11]", - "add x10, x10, x20", - "add x11, x11, x20" + "ldr w20, [x10]", + "str w20, [x11]", + "sub x10, x10, #0x4 (4)", + "sub x11, x11, #0x4 (4)" ] }, "negative movsq": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 6, "Optimal": "No", "Comment": [ "When direction flag is a compile time constant we can optimize", @@ -227,12 +214,10 @@ "ExpectedArm64ASM": [ "mov w20, #0x1", "strb w20, [x28, #714]", - "mov w21, #0x8", - "sub x20, x21, x20, lsl #4", - "ldr x21, [x10]", - "str x21, [x11]", - "add x10, x10, x20", - "add x11, x11, x20" + "ldr x20, [x10]", + "str x20, [x11]", + "sub x10, x10, #0x8 (8)", + "sub x11, x11, #0x8 (8)" ] } } diff --git a/unittests/InstructionCountCI/Primary.json b/unittests/InstructionCountCI/Primary.json index d53bc30ca..71bd4848e 100644 --- a/unittests/InstructionCountCI/Primary.json +++ b/unittests/InstructionCountCI/Primary.json @@ -321,7 +321,7 @@ "ExpectedArm64ASM": [ "mov x20, #0xffffffffffffffff", "mov x21, x4", - "add x4, x21, x20", + "sub x4, x21, #0x1 (1)", "eor x22, x21, x20", "strb w22, [x28, #708]", "strb w4, [x28, #706]", diff --git a/unittests/InstructionCountCI/PrimaryGroup.json b/unittests/InstructionCountCI/PrimaryGroup.json index 1f564b4d9..7e52b9e5f 100644 --- a/unittests/InstructionCountCI/PrimaryGroup.json +++ b/unittests/InstructionCountCI/PrimaryGroup.json @@ -653,7 +653,7 @@ "ExpectedArm64ASM": [ "mov x20, #0xffffffffffffff00", "mov x21, x4", - "add x4, x21, x20", + "sub x4, x21, #0x100 (256)", "strb w21, [x28, #708]", "strb w4, [x28, #706]", "cmn x21, x20", @@ -1186,7 +1186,7 @@ "ExpectedArm64ASM": [ "mov x20, #0xffffffffffffffff", "mov x21, x4", - "add x4, x21, x20", + "sub x4, x21, #0x1 (1)", "eor x22, x21, x20", "strb w22, [x28, #708]", "strb w4, [x28, #706]",