Merge pull request #3216 from alyssarosenzweig/opt/nzcv-infra

Prep commits for NZCV modelling
This commit is contained in:
Ryan Houdek authored and GitHub committed 2023-10-23 07:39:47 -07:00
commit e018917f76
12 files changed
+1023 -249

No files matched your search

+19 -7
View File
@@ -46,6 +46,7 @@ class OpDefinition:
NumElements: str
OpClass: str
HasSideEffects: bool
ImplicitFlagClobber: bool
RAOverride: int
SwitchGen: bool
ArgPrinter: bool
@@ -67,6 +68,7 @@ class OpDefinition:
self.OpClass = None
self.OpSize = 0
self.HasSideEffects = False
self.ImplicitFlagClobber = False
self.RAOverride = -1
self.SwitchGen = True
self.ArgPrinter = True
@@ -219,6 +221,9 @@ def parse_ops(ops):
if "HasSideEffects" in op_val:
OpDef.HasSideEffects = bool(op_val["HasSideEffects"])
if "ImplicitFlagClobber" in op_val:
OpDef.ImplicitFlagClobber = bool(op_val["ImplicitFlagClobber"])
if "ArgPrinter" in op_val:
OpDef.ArgPrinter = bool(op_val["ArgPrinter"])
@@ -372,6 +377,7 @@ def print_ir_sizes():
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetRAArgs(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool HasSideEffects(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool ImplicitFlagClobber(IROps Op);\n")
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool GetHasDest(IROps Op);\n")
output_file.write("#undef IROP_SIZES\n")
@@ -465,15 +471,17 @@ def print_ir_getraargs():
def print_ir_hassideeffects():
output_file.write("#ifdef IROP_HASSIDEEFFECTS_IMPL\n")
output_file.write("constexpr std::array<uint8_t, OP_LAST + 1> SideEffects = {\n")
for op in IROps:
output_file.write("\t{},\n".format(("true" if op.HasSideEffects else "false")))
for array, prop in [("SideEffects", "HasSideEffects"),
("ImplicitFlagClobbers", "ImplicitFlagClobber")]:
output_file.write(f"constexpr std::array<uint8_t, OP_LAST + 1> {array} = {{\n")
for op in IROps:
output_file.write("\t{},\n".format(("true" if getattr(op, prop) else "false")))
output_file.write("};\n\n")
output_file.write("};\n\n")
output_file.write("bool HasSideEffects(IROps Op) {\n")
output_file.write(" return SideEffects[Op];\n")
output_file.write("}\n")
output_file.write(f"bool {prop}(IROps Op) {{\n")
output_file.write(f" return {array}[Op];\n")
output_file.write("}\n")
output_file.write("#undef IROP_HASSIDEEFFECTS_IMPL\n")
output_file.write("#endif\n\n")
@@ -642,6 +650,10 @@ def print_ir_allocator_helpers():
output_file.write(") {\n")
# Save NZCV if needed before clobbering NZCV
if op.ImplicitFlagClobber:
output_file.write("\t\tSaveNZCV();")
output_file.write("\t\tauto Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
if op.SSAArgNum != 0:
@@ -121,8 +121,8 @@ void Dispatcher::EmitDispatcher() {
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r3, ARMEmitter::ShiftType::LSL , 4);
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
cmp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, RipReg.R());
b(ARMEmitter::Condition::CC_NE, &FullLookup);
sub(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, RipReg);
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &FullLookup);
br(ARMEmitter::Reg::r3);
@@ -167,8 +167,8 @@ void Dispatcher::EmitDispatcher() {
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x1, ARMEmitter::Reg::r0, 0);
// If the guest address doesn't match, Compile the block.
cmp(ARMEmitter::XReg::x1, RipReg);
b(ARMEmitter::Condition::CC_NE, &NoBlock);
sub(ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, RipReg);
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, &NoBlock);
// Check the host address to see if it matches, else compile the block.
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, &NoBlock);
@@ -225,7 +225,7 @@ void Dispatcher::EmitDispatcher() {
FillStaticRegs();
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
// Trigger segfault if any deferred signals are pending
@@ -262,7 +262,7 @@ void Dispatcher::EmitDispatcher() {
FillStaticRegs();
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
// Trigger segfault if any deferred signals are pending
@@ -3694,9 +3694,16 @@ DEF_OP(VInsElement) {
} else {
const auto UpperBound = 16 >> FEXCore::ilog2(ElementSize);
const auto TargetElement = static_cast<int>(DestIdx) - UpperBound;
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
index(SubRegSize, VTMP1.Z(), -UpperBound, 1);
cmpeq(SubRegSize, Predicate, PRED_TMP_32B.Zeroing(), VTMP1.Z(), TargetElement);
mov(SubRegSize, Dst.Z(), Predicate.Merging(), VTMP2.Z());
// Restore NZCV
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
}
}
else {
@@ -1911,30 +1911,10 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) {
StoreResult(GPRClass, Op, Res, -1);
auto CondJump = _CondJump(Shift, {COND_EQ});
auto CurrentBlock = GetCurrentBlock();
// Do nothing if shift count is zero
auto JumpTarget = CreateNewCodeBlockAfter(CurrentBlock);
SetFalseJumpTarget(CondJump, JumpTarget);
SetCurrentCodeBlock(JumpTarget);
StartNewBlock();
if (Size != 64) {
Res = _Bfe(OpSize::i64Bit, Size, 0, Res);
}
GenerateFlags_ShiftLeft(Op, Res, Dest, Shift);
// Calculate flags early.
CalculateDeferredFlags();
auto Jump = _Jump();
auto NextJumpTarget = CreateNewCodeBlockAfter(JumpTarget);
SetJumpTarget(Jump, NextJumpTarget);
SetTrueJumpTarget(CondJump, NextJumpTarget);
SetCurrentCodeBlock(NextJumpTarget);
StartNewBlock();
}
void OpDispatchBuilder::SHLDImmediateOp(OpcodeArgs) {
@@ -2013,28 +1993,10 @@ void OpDispatchBuilder::SHRDOp(OpcodeArgs) {
StoreResult(GPRClass, Op, Res, -1);
auto CondJump = _CondJump(Shift, {COND_EQ});
// Do not change flags if shift count is zero
auto JumpTarget = CreateNewCodeBlockAfter(GetCurrentBlock());
SetFalseJumpTarget(CondJump, JumpTarget);
SetCurrentCodeBlock(JumpTarget);
StartNewBlock();
if (Size != 64) {
Res = _Bfe(OpSize::i64Bit, Size, 0, Res);
}
GenerateFlags_ShiftRight(Op, Res, Dest, Shift);
// Calculate deferred flags immediately.
// This block is ending so it needs to serialize
CalculateDeferredFlags();
auto Jump = _Jump();
auto NextJumpTarget = CreateNewCodeBlockAfter(JumpTarget);
SetJumpTarget(Jump, NextJumpTarget);
SetTrueJumpTarget(CondJump, NextJumpTarget);
SetCurrentCodeBlock(NextJumpTarget);
StartNewBlock();
}
void OpDispatchBuilder::SHRDImmediateOp(OpcodeArgs) {
@@ -894,6 +894,10 @@ public:
}
}
protected:
void SaveNZCV() override {
}
private:
enum class SelectionFlag {
Nothing, // must rely on x86 flags
+23
View File
@@ -202,6 +202,7 @@
"GPR = ValidateCode u64:$CodeOriginalLow, u64:$CodeOriginalhigh, i64:$Offset, u8:$CodeLength": {
"HasSideEffects": true,
"HasDest": true,
"ImplicitFlagClobber": true,
"DestSize": "8"
},
@@ -247,6 +248,7 @@
"The second GPR pair element is a bool if the number is valid",
"RNG hardware is allowed to fail early and return. Software must always check this"
],
"ImplicitFlagClobber": true,
"DestSize": "16",
"NumElements": "2"
},
@@ -499,6 +501,7 @@
"FPR = VLoadVectorMasked u8:#RegisterSize, u8:#ElementSize, FPR:$Mask, GPR:$Addr, GPR:$Offset, MemOffsetType:$OffsetType, u8:$OffsetScale": {
"Desc": ["Does a masked load similar to VPMASKMOV/VMASKMOV where the upper bit of each element",
"determines whether or not that element will be loaded from memory"],
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
@@ -506,6 +509,7 @@
"Desc": ["Does a masked store similar to VPMASKMOV/VMASKMOV where the upper bit of each element",
"determines whether or not that element will be stored to memory"],
"HasSideEffects": true,
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
@@ -597,6 +601,7 @@
],
"DestSize": "Size",
"ImplicitFlagClobber": true,
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
@@ -611,6 +616,7 @@
],
"HasDest": true,
"DestSize": "Size",
"ImplicitFlagClobber": true,
"NumElements": "2",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i64Bit || Size == FEXCore::IR::OpSize::i128Bit"
@@ -753,6 +759,7 @@
"Desc": ["Set Telemetry value if the passed in 32-bit value isn't zero.",
"Only useful for 32-bit applications."
],
"ImplicitFlagClobber": true,
"DestSize": "8"
}
},
@@ -830,6 +837,7 @@
"Will truncate to 64 or 32bits"
],
"DestSize": "Size",
"ImplicitFlagClobber": true,
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
@@ -859,6 +867,7 @@
"In the case of zero returns ~0U"
],
"DestSize": "Size",
"ImplicitFlagClobber": true,
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
@@ -914,6 +923,7 @@
"GPR = AddNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Return NZCV for the sum of two GPRs"],
"DestSize": "4",
"ImplicitFlagClobber": true,
"EmitValidation": [
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
]
@@ -921,6 +931,7 @@
"GPR = AdcNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, GPR:$NZCV": {
"Desc": ["Return NZCV for the sum of two GPRs and carry-in given as NZCV"],
"DestSize": "4",
"ImplicitFlagClobber": true,
"EmitValidation": [
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
]
@@ -928,6 +939,7 @@
"GPR = SbbNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, GPR:$NZCV": {
"Desc": ["Return NZCV for the sum of two GPRs and carry-in given as NZCV"],
"DestSize": "4",
"ImplicitFlagClobber": true,
"EmitValidation": [
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
]
@@ -946,6 +958,7 @@
"If InvertCarry is nonzero, carry flag uses x86 definition, inverted from arm64.",
""],
"DestSize": "4",
"ImplicitFlagClobber": true,
"EmitValidation": [
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
]
@@ -1007,6 +1020,7 @@
},
"GPR = TestNZ u8:$Size, GPR:$Src1": {
"Desc": ["Return NZCV for a GPR, setting N and Z accordingly and zeroing C and V"],
"ImplicitFlagClobber": true,
"DestSize": "4"
},
"GPR = Lshl OpSize:#Size, GPR:$Src1, GPR:$Src2": {
@@ -1158,6 +1172,7 @@
"Dest = Cmp1 <Cond> Cmp2 ? TrueVal : FalseVal"
],
"DestSize": "ResultSize",
"ImplicitFlagClobber": true,
"EmitValidation": [
"_CompareSize == FEXCore::IR::OpSize::i32Bit || _CompareSize == FEXCore::IR::OpSize::i64Bit || _CompareSize == FEXCore::IR::OpSize::i128Bit",
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit",
@@ -1266,6 +1281,7 @@
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
"Ordering flag result is true if either float input is NaN"
],
"ImplicitFlagClobber": true,
"DestSize": "4"
}
},
@@ -1520,6 +1536,7 @@
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VCMPEQZ u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
@@ -1528,6 +1545,7 @@
"Each element is compared, if the result is true then the resulting element is ~0, else zero",
"Compares the vector against zero"
],
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
@@ -1536,6 +1554,7 @@
"Each element is compared, if the result is true then the resulting element is ~0, else zero",
"Compares the vector against zero"
],
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
@@ -1777,10 +1796,12 @@
},
"FPR = VFMin u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFMax u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
@@ -1899,10 +1920,12 @@
"Each element is compared, if the result is true then the resulting element is ~0, else zero"
],
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
"FPR = VFCMPEQ u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
"ImplicitFlagClobber": true,
"DestSize": "RegisterSize",
"NumElements": "RegisterSize / ElementSize"
},
@@ -277,7 +277,8 @@ void ConstProp::CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& Cur
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
auto BlockOp = BlockIROp->CW<FEXCore::IR::IROp_CodeBlock>();
for (auto [UnaryOpNode, UnaryOpHdr] : CurrentIR.GetCode(BlockNode)) {
if (IR::GetArgs(UnaryOpHdr->Op) == 1 && !HasSideEffects(UnaryOpHdr->Op)) {
if (IR::GetArgs(UnaryOpHdr->Op) == 1 && !HasSideEffects(UnaryOpHdr->Op)
&& !ImplicitFlagClobber(UnaryOpHdr->Op)) {
// could be moved
auto SelectOpNode = IREmit->UnwrapNode(UnaryOpHdr->Args[0]);
auto SelectOpHdr = IREmit->GetOpHeader(UnaryOpHdr->Args[0]);
+6
View File
@@ -27,6 +27,8 @@ friend class FEXCore::IR::PassManager;
ResetWorkingList();
}
virtual ~IREmitter() = default;
void ReownOrClaimBuffer() {
DualListData.ReownOrClaimBuffer();
}
@@ -332,6 +334,10 @@ friend class FEXCore::IR::PassManager;
return Ptr;
}
virtual void SaveNZCV() {
// Overriden by dispatcher, stubbed for IR tests
}
OrderedNode *CurrentWriteCursor = nullptr;
// These could be combined with a little bit of work to be more efficient with memory usage. Isn't a big deal
+44 -51
View File
@@ -1963,7 +1963,7 @@
]
},
"shld ax, bx, cl": {
"ExpectedInstructionCount": 33,
"ExpectedInstructionCount": 31,
"Optimal": "No",
"Comment": "0x0f 0xad",
"ExpectedArm64ASM": [
@@ -1972,38 +1972,36 @@
"uxtb w22, w5",
"and x22, x22, #0x1f",
"mov w23, #0x10",
"sub x23, x23, x22",
"lsl x24, x21, x22",
"lsr w20, w20, w23",
"orr x20, x24, x20",
"sub x24, x23, x22",
"lsl x25, x21, x22",
"lsr w20, w20, w24",
"orr x20, x25, x20",
"cmp x22, #0x0 (0)",
"csel x20, x21, x20, eq",
"bfxil x4, x20, #0, #16",
"cbz x22, #+0x54",
"uxth x20, w20",
"ldr w23, [x28, #728]",
"lsl w24, w20, #16",
"tst w24, w24",
"mrs x24, nzcv",
"mov w25, #0x10",
"sub w25, w25, w22",
"lsr w25, w21, w25",
"ubfx w25, w25, #0, #1",
"orr w24, w24, w25, lsl #29",
"ldr w24, [x28, #728]",
"lsl w25, w20, #16",
"tst w25, w25",
"mrs x25, nzcv",
"sub w23, w23, w22",
"lsr w23, w21, w23",
"ubfx w23, w23, #0, #1",
"orr w23, w25, w23, lsl #29",
"ldrb w25, [x28, #706]",
"cmp x22, #0x0 (0)",
"csel x25, x25, x20, eq",
"strb w25, [x28, #706]",
"eor w20, w21, w20",
"ubfx w20, w20, #15, #1",
"orr w20, w24, w20, lsl #28",
"orr w20, w23, w20, lsl #28",
"cmp x22, #0x0 (0)",
"csel w20, w23, w20, eq",
"csel w20, w24, w20, eq",
"str w20, [x28, #728]"
]
},
"shld eax, ebx, cl": {
"ExpectedInstructionCount": 32,
"ExpectedInstructionCount": 30,
"Optimal": "No",
"Comment": "0x0f 0xad",
"ExpectedArm64ASM": [
@@ -2012,37 +2010,35 @@
"uxtb w22, w5",
"and x22, x22, #0x1f",
"mov w23, #0x20",
"sub x23, x23, x22",
"lsl x24, x21, x22",
"lsr w20, w20, w23",
"orr x20, x24, x20",
"sub x24, x23, x22",
"lsl x25, x21, x22",
"lsr w20, w20, w24",
"orr x20, x25, x20",
"cmp x22, #0x0 (0)",
"csel x20, x21, x20, eq",
"mov w4, w20",
"cbz x22, #+0x50",
"mov w20, w20",
"ldr w23, [x28, #728]",
"ldr w24, [x28, #728]",
"tst w20, w20",
"mrs x24, nzcv",
"mov w25, #0x20",
"sub w25, w25, w22",
"lsr w25, w21, w25",
"ubfx w25, w25, #0, #1",
"orr w24, w24, w25, lsl #29",
"mrs x25, nzcv",
"sub w23, w23, w22",
"lsr w23, w21, w23",
"ubfx w23, w23, #0, #1",
"orr w23, w25, w23, lsl #29",
"ldrb w25, [x28, #706]",
"cmp x22, #0x0 (0)",
"csel x25, x25, x20, eq",
"strb w25, [x28, #706]",
"eor w20, w21, w20",
"lsr w20, w20, #31",
"orr w20, w24, w20, lsl #28",
"orr w20, w23, w20, lsl #28",
"cmp x22, #0x0 (0)",
"csel w20, w23, w20, eq",
"csel w20, w24, w20, eq",
"str w20, [x28, #728]"
]
},
"shld rax, rbx, cl": {
"ExpectedInstructionCount": 30,
"ExpectedInstructionCount": 27,
"Optimal": "No",
"Comment": "0x0f 0xad",
"ExpectedArm64ASM": [
@@ -2050,29 +2046,26 @@
"uxtb w21, w5",
"and x21, x21, #0x3f",
"mov w22, #0x40",
"sub x22, x22, x21",
"lsl x23, x20, x21",
"lsr x22, x7, x22",
"orr x22, x23, x22",
"sub x23, x22, x21",
"lsl x24, x20, x21",
"lsr x23, x7, x23",
"orr x23, x24, x23",
"cmp x21, #0x0 (0)",
"csel x22, x20, x22, eq",
"mov x4, x22",
"cbz x21, #+0x4c",
"csel x4, x20, x23, eq",
"ldr w23, [x28, #728]",
"tst x22, x22",
"tst x4, x4",
"mrs x24, nzcv",
"mov w25, #0x40",
"sub x25, x25, x21",
"lsr x25, x20, x25",
"ubfx x25, x25, #0, #1",
"orr w24, w24, w25, lsl #29",
"ldrb w25, [x28, #706]",
"sub x22, x22, x21",
"lsr x22, x20, x22",
"ubfx x22, x22, #0, #1",
"orr w22, w24, w22, lsl #29",
"ldrb w24, [x28, #706]",
"cmp x21, #0x0 (0)",
"csel x25, x25, x22, eq",
"strb w25, [x28, #706]",
"eor x20, x20, x22",
"csel x24, x24, x4, eq",
"strb w24, [x28, #706]",
"eor x20, x20, x4",
"lsr x20, x20, #63",
"orr w20, w24, w20, lsl #28",
"orr w20, w22, w20, lsl #28",
"cmp x21, #0x0 (0)",
"csel w20, w23, w20, eq",
"str w20, [x28, #728]"
File diff suppressed because it is too large. Load diff
+42 -14
View File
@@ -52,7 +52,7 @@
]
},
"vphaddw ymm0, ymm1, ymm2": {
"ExpectedInstructionCount": 15,
"ExpectedInstructionCount": 19,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x01 256-bit"
@@ -65,14 +65,18 @@
"splice z2.d, p6, z2.d, z1.d",
"mov z1.d, z2.d[2]",
"mov z3.d, z2.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #-1",
"mov z3.d, p0/m, z1.d",
"msr nzcv, x0",
"mov z1.d, z2.d[1]",
"mov z16.d, z3.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #0",
"mov z16.d, p0/m, z1.d"
"mov z16.d, p0/m, z1.d",
"msr nzcv, x0"
]
},
"vphaddd xmm0, xmm1, xmm2": {
@@ -86,7 +90,7 @@
]
},
"vphaddd ymm0, ymm1, ymm2": {
"ExpectedInstructionCount": 15,
"ExpectedInstructionCount": 19,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x02 256-bit"
@@ -99,14 +103,18 @@
"splice z2.d, p6, z2.d, z1.d",
"mov z1.d, z2.d[2]",
"mov z3.d, z2.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #-1",
"mov z3.d, p0/m, z1.d",
"msr nzcv, x0",
"mov z1.d, z2.d[1]",
"mov z16.d, z3.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #0",
"mov z16.d, p0/m, z1.d"
"mov z16.d, p0/m, z1.d",
"msr nzcv, x0"
]
},
"vphaddsw xmm0, xmm1, xmm2": {
@@ -122,7 +130,7 @@
]
},
"vphaddsw ymm0, ymm1, ymm2": {
"ExpectedInstructionCount": 13,
"ExpectedInstructionCount": 17,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x03 256-bit"
@@ -133,14 +141,18 @@
"sqadd z2.h, z2.h, z3.h",
"mov z1.d, z2.d[2]",
"mov z3.d, z2.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #-1",
"mov z3.d, p0/m, z1.d",
"msr nzcv, x0",
"mov z1.d, z2.d[1]",
"mov z16.d, z3.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #0",
"mov z16.d, p0/m, z1.d"
"mov z16.d, p0/m, z1.d",
"msr nzcv, x0"
]
},
"vpmaddubsw xmm0, xmm1, xmm2": {
@@ -192,7 +204,7 @@
]
},
"vphsubw ymm0, ymm1, ymm2": {
"ExpectedInstructionCount": 13,
"ExpectedInstructionCount": 17,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x05 256-bit"
@@ -203,14 +215,18 @@
"sub z2.h, z2.h, z3.h",
"mov z1.d, z2.d[2]",
"mov z3.d, z2.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #-1",
"mov z3.d, p0/m, z1.d",
"msr nzcv, x0",
"mov z1.d, z2.d[1]",
"mov z16.d, z3.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #0",
"mov z16.d, p0/m, z1.d"
"mov z16.d, p0/m, z1.d",
"msr nzcv, x0"
]
},
"vphsubd xmm0, xmm1, xmm2": {
@@ -226,7 +242,7 @@
]
},
"vphsubd ymm0, ymm1, ymm2": {
"ExpectedInstructionCount": 13,
"ExpectedInstructionCount": 17,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x06 256-bit"
@@ -237,14 +253,18 @@
"sub z2.s, z2.s, z3.s",
"mov z1.d, z2.d[2]",
"mov z3.d, z2.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #-1",
"mov z3.d, p0/m, z1.d",
"msr nzcv, x0",
"mov z1.d, z2.d[1]",
"mov z16.d, z3.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #0",
"mov z16.d, p0/m, z1.d"
"mov z16.d, p0/m, z1.d",
"msr nzcv, x0"
]
},
"vphsubsw xmm0, xmm1, xmm2": {
@@ -260,7 +280,7 @@
]
},
"vphsubsw ymm0, ymm1, ymm2": {
"ExpectedInstructionCount": 13,
"ExpectedInstructionCount": 17,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x07 256-bit"
@@ -271,14 +291,18 @@
"sqsub z2.h, z2.h, z3.h",
"mov z1.d, z2.d[2]",
"mov z3.d, z2.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #-1",
"mov z3.d, p0/m, z1.d",
"msr nzcv, x0",
"mov z1.d, z2.d[1]",
"mov z16.d, z3.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #0",
"mov z16.d, p0/m, z1.d"
"mov z16.d, p0/m, z1.d",
"msr nzcv, x0"
]
},
"vpsignb xmm0, xmm1, xmm2": {
@@ -1060,7 +1084,7 @@
]
},
"vpackusdw ymm0, ymm1, ymm2": {
"ExpectedInstructionCount": 15,
"ExpectedInstructionCount": 19,
"Optimal": "No",
"Comment": [
"Map 2 0b01 0x2b 256-bit"
@@ -1073,14 +1097,18 @@
"splice z2.h, p6, z2.h, z1.h",
"mov z1.d, z2.d[1]",
"mov z3.d, z2.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #0",
"mov z3.d, p0/m, z1.d",
"msr nzcv, x0",
"mov z1.d, z2.d[2]",
"mov z16.d, z3.d",
"mrs x0, nzcv",
"index z0.d, #-2, #1",
"cmpeq p0.d, p7/z, z0.d, #-1",
"mov z16.d, p0/m, z1.d"
"mov z16.d, p0/m, z1.d",
"msr nzcv, x0"
]
},
"vmaskmovps xmm0, xmm1, [rax]": {
File diff suppressed because it is too large. Load diff