mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 12:00:17 +02:00
Merge pull request #3216 from alyssarosenzweig/opt/nzcv-infra
Prep commits for NZCV modelling
This commit is contained in:
12 files changed
+1023
-249
No files matched your search
@@ -46,6 +46,7 @@ class OpDefinition:
|
||||
NumElements: str
|
||||
OpClass: str
|
||||
HasSideEffects: bool
|
||||
ImplicitFlagClobber: bool
|
||||
RAOverride: int
|
||||
SwitchGen: bool
|
||||
ArgPrinter: bool
|
||||
@@ -67,6 +68,7 @@ class OpDefinition:
|
||||
self.OpClass = None
|
||||
self.OpSize = 0
|
||||
self.HasSideEffects = False
|
||||
self.ImplicitFlagClobber = False
|
||||
self.RAOverride = -1
|
||||
self.SwitchGen = True
|
||||
self.ArgPrinter = True
|
||||
@@ -219,6 +221,9 @@ def parse_ops(ops):
|
||||
if "HasSideEffects" in op_val:
|
||||
OpDef.HasSideEffects = bool(op_val["HasSideEffects"])
|
||||
|
||||
if "ImplicitFlagClobber" in op_val:
|
||||
OpDef.ImplicitFlagClobber = bool(op_val["ImplicitFlagClobber"])
|
||||
|
||||
if "ArgPrinter" in op_val:
|
||||
OpDef.ArgPrinter = bool(op_val["ArgPrinter"])
|
||||
|
||||
@@ -372,6 +377,7 @@ def print_ir_sizes():
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] uint8_t GetRAArgs(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] FEXCore::IR::RegisterClassType GetRegClass(IROps Op);\n\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool HasSideEffects(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool ImplicitFlagClobber(IROps Op);\n")
|
||||
output_file.write("[[nodiscard, gnu::const, gnu::visibility(\"default\")]] bool GetHasDest(IROps Op);\n")
|
||||
|
||||
output_file.write("#undef IROP_SIZES\n")
|
||||
@@ -465,15 +471,17 @@ def print_ir_getraargs():
|
||||
def print_ir_hassideeffects():
|
||||
output_file.write("#ifdef IROP_HASSIDEEFFECTS_IMPL\n")
|
||||
|
||||
output_file.write("constexpr std::array<uint8_t, OP_LAST + 1> SideEffects = {\n")
|
||||
for op in IROps:
|
||||
output_file.write("\t{},\n".format(("true" if op.HasSideEffects else "false")))
|
||||
for array, prop in [("SideEffects", "HasSideEffects"),
|
||||
("ImplicitFlagClobbers", "ImplicitFlagClobber")]:
|
||||
output_file.write(f"constexpr std::array<uint8_t, OP_LAST + 1> {array} = {{\n")
|
||||
for op in IROps:
|
||||
output_file.write("\t{},\n".format(("true" if getattr(op, prop) else "false")))
|
||||
|
||||
output_file.write("};\n\n")
|
||||
output_file.write("};\n\n")
|
||||
|
||||
output_file.write("bool HasSideEffects(IROps Op) {\n")
|
||||
output_file.write(" return SideEffects[Op];\n")
|
||||
output_file.write("}\n")
|
||||
output_file.write(f"bool {prop}(IROps Op) {{\n")
|
||||
output_file.write(f" return {array}[Op];\n")
|
||||
output_file.write("}\n")
|
||||
|
||||
output_file.write("#undef IROP_HASSIDEEFFECTS_IMPL\n")
|
||||
output_file.write("#endif\n\n")
|
||||
@@ -642,6 +650,10 @@ def print_ir_allocator_helpers():
|
||||
|
||||
output_file.write(") {\n")
|
||||
|
||||
# Save NZCV if needed before clobbering NZCV
|
||||
if op.ImplicitFlagClobber:
|
||||
output_file.write("\t\tSaveNZCV();")
|
||||
|
||||
output_file.write("\t\tauto Op = AllocateOp<IROp_{}, IROps::OP_{}>();\n".format(op.Name, op.Name.upper()))
|
||||
|
||||
if op.SSAArgNum != 0:
|
||||
|
||||
@@ -121,8 +121,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
and_(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, RipReg.R(), LookupCache::L1_ENTRIES_MASK);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::r0, ARMEmitter::Reg::r3, ARMEmitter::ShiftType::LSL , 4);
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x0, ARMEmitter::Reg::r0, 0);
|
||||
cmp(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, RipReg.R());
|
||||
b(ARMEmitter::Condition::CC_NE, &FullLookup);
|
||||
sub(ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, &FullLookup);
|
||||
|
||||
br(ARMEmitter::Reg::r3);
|
||||
|
||||
@@ -167,8 +167,8 @@ void Dispatcher::EmitDispatcher() {
|
||||
ldp<ARMEmitter::IndexType::OFFSET>(ARMEmitter::XReg::x3, ARMEmitter::XReg::x1, ARMEmitter::Reg::r0, 0);
|
||||
|
||||
// If the guest address doesn't match, Compile the block.
|
||||
cmp(ARMEmitter::XReg::x1, RipReg);
|
||||
b(ARMEmitter::Condition::CC_NE, &NoBlock);
|
||||
sub(ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, RipReg);
|
||||
cbnz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, &NoBlock);
|
||||
|
||||
// Check the host address to see if it matches, else compile the block.
|
||||
cbz(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r3, &NoBlock);
|
||||
@@ -225,7 +225,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1);
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
@@ -262,7 +262,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
FillStaticRegs();
|
||||
|
||||
ldr(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount));
|
||||
|
||||
// Trigger segfault if any deferred signals are pending
|
||||
|
||||
@@ -3694,9 +3694,16 @@ DEF_OP(VInsElement) {
|
||||
} else {
|
||||
const auto UpperBound = 16 >> FEXCore::ilog2(ElementSize);
|
||||
const auto TargetElement = static_cast<int>(DestIdx) - UpperBound;
|
||||
|
||||
// FIXME: We should rework this op to avoid the NZCV spill/fill dance.
|
||||
mrs(TMP1, ARMEmitter::SystemRegister::NZCV);
|
||||
|
||||
index(SubRegSize, VTMP1.Z(), -UpperBound, 1);
|
||||
cmpeq(SubRegSize, Predicate, PRED_TMP_32B.Zeroing(), VTMP1.Z(), TargetElement);
|
||||
mov(SubRegSize, Dst.Z(), Predicate.Merging(), VTMP2.Z());
|
||||
|
||||
// Restore NZCV
|
||||
msr(ARMEmitter::SystemRegister::NZCV, TMP1);
|
||||
}
|
||||
}
|
||||
else {
|
||||
|
||||
@@ -1911,30 +1911,10 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) {
|
||||
|
||||
StoreResult(GPRClass, Op, Res, -1);
|
||||
|
||||
auto CondJump = _CondJump(Shift, {COND_EQ});
|
||||
|
||||
auto CurrentBlock = GetCurrentBlock();
|
||||
|
||||
// Do nothing if shift count is zero
|
||||
auto JumpTarget = CreateNewCodeBlockAfter(CurrentBlock);
|
||||
SetFalseJumpTarget(CondJump, JumpTarget);
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
StartNewBlock();
|
||||
|
||||
if (Size != 64) {
|
||||
Res = _Bfe(OpSize::i64Bit, Size, 0, Res);
|
||||
}
|
||||
GenerateFlags_ShiftLeft(Op, Res, Dest, Shift);
|
||||
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
auto Jump = _Jump();
|
||||
auto NextJumpTarget = CreateNewCodeBlockAfter(JumpTarget);
|
||||
SetJumpTarget(Jump, NextJumpTarget);
|
||||
SetTrueJumpTarget(CondJump, NextJumpTarget);
|
||||
SetCurrentCodeBlock(NextJumpTarget);
|
||||
StartNewBlock();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHLDImmediateOp(OpcodeArgs) {
|
||||
@@ -2013,28 +1993,10 @@ void OpDispatchBuilder::SHRDOp(OpcodeArgs) {
|
||||
|
||||
StoreResult(GPRClass, Op, Res, -1);
|
||||
|
||||
auto CondJump = _CondJump(Shift, {COND_EQ});
|
||||
|
||||
// Do not change flags if shift count is zero
|
||||
auto JumpTarget = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
SetFalseJumpTarget(CondJump, JumpTarget);
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
StartNewBlock();
|
||||
|
||||
if (Size != 64) {
|
||||
Res = _Bfe(OpSize::i64Bit, Size, 0, Res);
|
||||
}
|
||||
GenerateFlags_ShiftRight(Op, Res, Dest, Shift);
|
||||
// Calculate deferred flags immediately.
|
||||
// This block is ending so it needs to serialize
|
||||
CalculateDeferredFlags();
|
||||
|
||||
auto Jump = _Jump();
|
||||
auto NextJumpTarget = CreateNewCodeBlockAfter(JumpTarget);
|
||||
SetJumpTarget(Jump, NextJumpTarget);
|
||||
SetTrueJumpTarget(CondJump, NextJumpTarget);
|
||||
SetCurrentCodeBlock(NextJumpTarget);
|
||||
StartNewBlock();
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHRDImmediateOp(OpcodeArgs) {
|
||||
|
||||
@@ -894,6 +894,10 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
protected:
|
||||
void SaveNZCV() override {
|
||||
}
|
||||
|
||||
private:
|
||||
enum class SelectionFlag {
|
||||
Nothing, // must rely on x86 flags
|
||||
|
||||
@@ -202,6 +202,7 @@
|
||||
"GPR = ValidateCode u64:$CodeOriginalLow, u64:$CodeOriginalhigh, i64:$Offset, u8:$CodeLength": {
|
||||
"HasSideEffects": true,
|
||||
"HasDest": true,
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "8"
|
||||
},
|
||||
|
||||
@@ -247,6 +248,7 @@
|
||||
"The second GPR pair element is a bool if the number is valid",
|
||||
"RNG hardware is allowed to fail early and return. Software must always check this"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "16",
|
||||
"NumElements": "2"
|
||||
},
|
||||
@@ -499,6 +501,7 @@
|
||||
"FPR = VLoadVectorMasked u8:#RegisterSize, u8:#ElementSize, FPR:$Mask, GPR:$Addr, GPR:$Offset, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a masked load similar to VPMASKMOV/VMASKMOV where the upper bit of each element",
|
||||
"determines whether or not that element will be loaded from memory"],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -506,6 +509,7 @@
|
||||
"Desc": ["Does a masked store similar to VPMASKMOV/VMASKMOV where the upper bit of each element",
|
||||
"determines whether or not that element will be stored to memory"],
|
||||
"HasSideEffects": true,
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -597,6 +601,7 @@
|
||||
],
|
||||
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -611,6 +616,7 @@
|
||||
],
|
||||
"HasDest": true,
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"NumElements": "2",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i64Bit || Size == FEXCore::IR::OpSize::i128Bit"
|
||||
@@ -753,6 +759,7 @@
|
||||
"Desc": ["Set Telemetry value if the passed in 32-bit value isn't zero.",
|
||||
"Only useful for 32-bit applications."
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "8"
|
||||
}
|
||||
},
|
||||
@@ -830,6 +837,7 @@
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -859,6 +867,7 @@
|
||||
"In the case of zero returns ~0U"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -914,6 +923,7 @@
|
||||
"GPR = AddNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Return NZCV for the sum of two GPRs"],
|
||||
"DestSize": "4",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -921,6 +931,7 @@
|
||||
"GPR = AdcNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, GPR:$NZCV": {
|
||||
"Desc": ["Return NZCV for the sum of two GPRs and carry-in given as NZCV"],
|
||||
"DestSize": "4",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -928,6 +939,7 @@
|
||||
"GPR = SbbNZCV OpSize:$Size, GPR:$Src1, GPR:$Src2, GPR:$NZCV": {
|
||||
"Desc": ["Return NZCV for the sum of two GPRs and carry-in given as NZCV"],
|
||||
"DestSize": "4",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -946,6 +958,7 @@
|
||||
"If InvertCarry is nonzero, carry flag uses x86 definition, inverted from arm64.",
|
||||
""],
|
||||
"DestSize": "4",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"_Size == FEXCore::IR::OpSize::i32Bit || _Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -1007,6 +1020,7 @@
|
||||
},
|
||||
"GPR = TestNZ u8:$Size, GPR:$Src1": {
|
||||
"Desc": ["Return NZCV for a GPR, setting N and Z accordingly and zeroing C and V"],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "4"
|
||||
},
|
||||
"GPR = Lshl OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
@@ -1158,6 +1172,7 @@
|
||||
"Dest = Cmp1 <Cond> Cmp2 ? TrueVal : FalseVal"
|
||||
],
|
||||
"DestSize": "ResultSize",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"_CompareSize == FEXCore::IR::OpSize::i32Bit || _CompareSize == FEXCore::IR::OpSize::i64Bit || _CompareSize == FEXCore::IR::OpSize::i128Bit",
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit",
|
||||
@@ -1266,6 +1281,7 @@
|
||||
"Desc": ["Does a scalar unordered compare and stores the asked for flags in to a GPR",
|
||||
"Ordering flag result is true if either float input is NaN"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "4"
|
||||
}
|
||||
},
|
||||
@@ -1520,6 +1536,7 @@
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VCMPEQZ u8:#RegisterSize, u8:#ElementSize, FPR:$Vector": {
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1528,6 +1545,7 @@
|
||||
"Each element is compared, if the result is true then the resulting element is ~0, else zero",
|
||||
"Compares the vector against zero"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1536,6 +1554,7 @@
|
||||
"Each element is compared, if the result is true then the resulting element is ~0, else zero",
|
||||
"Compares the vector against zero"
|
||||
],
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1777,10 +1796,12 @@
|
||||
},
|
||||
|
||||
"FPR = VFMin u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFMax u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
@@ -1899,10 +1920,12 @@
|
||||
"Each element is compared, if the result is true then the resulting element is ~0, else zero"
|
||||
],
|
||||
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
"FPR = VFCMPEQ u8:#RegisterSize, u8:#ElementSize, FPR:$Vector1, FPR:$Vector2": {
|
||||
"ImplicitFlagClobber": true,
|
||||
"DestSize": "RegisterSize",
|
||||
"NumElements": "RegisterSize / ElementSize"
|
||||
},
|
||||
|
||||
@@ -277,7 +277,8 @@ void ConstProp::CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& Cur
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
auto BlockOp = BlockIROp->CW<FEXCore::IR::IROp_CodeBlock>();
|
||||
for (auto [UnaryOpNode, UnaryOpHdr] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IR::GetArgs(UnaryOpHdr->Op) == 1 && !HasSideEffects(UnaryOpHdr->Op)) {
|
||||
if (IR::GetArgs(UnaryOpHdr->Op) == 1 && !HasSideEffects(UnaryOpHdr->Op)
|
||||
&& !ImplicitFlagClobber(UnaryOpHdr->Op)) {
|
||||
// could be moved
|
||||
auto SelectOpNode = IREmit->UnwrapNode(UnaryOpHdr->Args[0]);
|
||||
auto SelectOpHdr = IREmit->GetOpHeader(UnaryOpHdr->Args[0]);
|
||||
|
||||
@@ -27,6 +27,8 @@ friend class FEXCore::IR::PassManager;
|
||||
ResetWorkingList();
|
||||
}
|
||||
|
||||
virtual ~IREmitter() = default;
|
||||
|
||||
void ReownOrClaimBuffer() {
|
||||
DualListData.ReownOrClaimBuffer();
|
||||
}
|
||||
@@ -332,6 +334,10 @@ friend class FEXCore::IR::PassManager;
|
||||
return Ptr;
|
||||
}
|
||||
|
||||
virtual void SaveNZCV() {
|
||||
// Overriden by dispatcher, stubbed for IR tests
|
||||
}
|
||||
|
||||
OrderedNode *CurrentWriteCursor = nullptr;
|
||||
|
||||
// These could be combined with a little bit of work to be more efficient with memory usage. Isn't a big deal
|
||||
|
||||
@@ -1963,7 +1963,7 @@
|
||||
]
|
||||
},
|
||||
"shld ax, bx, cl": {
|
||||
"ExpectedInstructionCount": 33,
|
||||
"ExpectedInstructionCount": 31,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x0f 0xad",
|
||||
"ExpectedArm64ASM": [
|
||||
@@ -1972,38 +1972,36 @@
|
||||
"uxtb w22, w5",
|
||||
"and x22, x22, #0x1f",
|
||||
"mov w23, #0x10",
|
||||
"sub x23, x23, x22",
|
||||
"lsl x24, x21, x22",
|
||||
"lsr w20, w20, w23",
|
||||
"orr x20, x24, x20",
|
||||
"sub x24, x23, x22",
|
||||
"lsl x25, x21, x22",
|
||||
"lsr w20, w20, w24",
|
||||
"orr x20, x25, x20",
|
||||
"cmp x22, #0x0 (0)",
|
||||
"csel x20, x21, x20, eq",
|
||||
"bfxil x4, x20, #0, #16",
|
||||
"cbz x22, #+0x54",
|
||||
"uxth x20, w20",
|
||||
"ldr w23, [x28, #728]",
|
||||
"lsl w24, w20, #16",
|
||||
"tst w24, w24",
|
||||
"mrs x24, nzcv",
|
||||
"mov w25, #0x10",
|
||||
"sub w25, w25, w22",
|
||||
"lsr w25, w21, w25",
|
||||
"ubfx w25, w25, #0, #1",
|
||||
"orr w24, w24, w25, lsl #29",
|
||||
"ldr w24, [x28, #728]",
|
||||
"lsl w25, w20, #16",
|
||||
"tst w25, w25",
|
||||
"mrs x25, nzcv",
|
||||
"sub w23, w23, w22",
|
||||
"lsr w23, w21, w23",
|
||||
"ubfx w23, w23, #0, #1",
|
||||
"orr w23, w25, w23, lsl #29",
|
||||
"ldrb w25, [x28, #706]",
|
||||
"cmp x22, #0x0 (0)",
|
||||
"csel x25, x25, x20, eq",
|
||||
"strb w25, [x28, #706]",
|
||||
"eor w20, w21, w20",
|
||||
"ubfx w20, w20, #15, #1",
|
||||
"orr w20, w24, w20, lsl #28",
|
||||
"orr w20, w23, w20, lsl #28",
|
||||
"cmp x22, #0x0 (0)",
|
||||
"csel w20, w23, w20, eq",
|
||||
"csel w20, w24, w20, eq",
|
||||
"str w20, [x28, #728]"
|
||||
]
|
||||
},
|
||||
"shld eax, ebx, cl": {
|
||||
"ExpectedInstructionCount": 32,
|
||||
"ExpectedInstructionCount": 30,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x0f 0xad",
|
||||
"ExpectedArm64ASM": [
|
||||
@@ -2012,37 +2010,35 @@
|
||||
"uxtb w22, w5",
|
||||
"and x22, x22, #0x1f",
|
||||
"mov w23, #0x20",
|
||||
"sub x23, x23, x22",
|
||||
"lsl x24, x21, x22",
|
||||
"lsr w20, w20, w23",
|
||||
"orr x20, x24, x20",
|
||||
"sub x24, x23, x22",
|
||||
"lsl x25, x21, x22",
|
||||
"lsr w20, w20, w24",
|
||||
"orr x20, x25, x20",
|
||||
"cmp x22, #0x0 (0)",
|
||||
"csel x20, x21, x20, eq",
|
||||
"mov w4, w20",
|
||||
"cbz x22, #+0x50",
|
||||
"mov w20, w20",
|
||||
"ldr w23, [x28, #728]",
|
||||
"ldr w24, [x28, #728]",
|
||||
"tst w20, w20",
|
||||
"mrs x24, nzcv",
|
||||
"mov w25, #0x20",
|
||||
"sub w25, w25, w22",
|
||||
"lsr w25, w21, w25",
|
||||
"ubfx w25, w25, #0, #1",
|
||||
"orr w24, w24, w25, lsl #29",
|
||||
"mrs x25, nzcv",
|
||||
"sub w23, w23, w22",
|
||||
"lsr w23, w21, w23",
|
||||
"ubfx w23, w23, #0, #1",
|
||||
"orr w23, w25, w23, lsl #29",
|
||||
"ldrb w25, [x28, #706]",
|
||||
"cmp x22, #0x0 (0)",
|
||||
"csel x25, x25, x20, eq",
|
||||
"strb w25, [x28, #706]",
|
||||
"eor w20, w21, w20",
|
||||
"lsr w20, w20, #31",
|
||||
"orr w20, w24, w20, lsl #28",
|
||||
"orr w20, w23, w20, lsl #28",
|
||||
"cmp x22, #0x0 (0)",
|
||||
"csel w20, w23, w20, eq",
|
||||
"csel w20, w24, w20, eq",
|
||||
"str w20, [x28, #728]"
|
||||
]
|
||||
},
|
||||
"shld rax, rbx, cl": {
|
||||
"ExpectedInstructionCount": 30,
|
||||
"ExpectedInstructionCount": 27,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x0f 0xad",
|
||||
"ExpectedArm64ASM": [
|
||||
@@ -2050,29 +2046,26 @@
|
||||
"uxtb w21, w5",
|
||||
"and x21, x21, #0x3f",
|
||||
"mov w22, #0x40",
|
||||
"sub x22, x22, x21",
|
||||
"lsl x23, x20, x21",
|
||||
"lsr x22, x7, x22",
|
||||
"orr x22, x23, x22",
|
||||
"sub x23, x22, x21",
|
||||
"lsl x24, x20, x21",
|
||||
"lsr x23, x7, x23",
|
||||
"orr x23, x24, x23",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"csel x22, x20, x22, eq",
|
||||
"mov x4, x22",
|
||||
"cbz x21, #+0x4c",
|
||||
"csel x4, x20, x23, eq",
|
||||
"ldr w23, [x28, #728]",
|
||||
"tst x22, x22",
|
||||
"tst x4, x4",
|
||||
"mrs x24, nzcv",
|
||||
"mov w25, #0x40",
|
||||
"sub x25, x25, x21",
|
||||
"lsr x25, x20, x25",
|
||||
"ubfx x25, x25, #0, #1",
|
||||
"orr w24, w24, w25, lsl #29",
|
||||
"ldrb w25, [x28, #706]",
|
||||
"sub x22, x22, x21",
|
||||
"lsr x22, x20, x22",
|
||||
"ubfx x22, x22, #0, #1",
|
||||
"orr w22, w24, w22, lsl #29",
|
||||
"ldrb w24, [x28, #706]",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"csel x25, x25, x22, eq",
|
||||
"strb w25, [x28, #706]",
|
||||
"eor x20, x20, x22",
|
||||
"csel x24, x24, x4, eq",
|
||||
"strb w24, [x28, #706]",
|
||||
"eor x20, x20, x4",
|
||||
"lsr x20, x20, #63",
|
||||
"orr w20, w24, w20, lsl #28",
|
||||
"orr w20, w22, w20, lsl #28",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"csel w20, w23, w20, eq",
|
||||
"str w20, [x28, #728]"
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -52,7 +52,7 @@
|
||||
]
|
||||
},
|
||||
"vphaddw ymm0, ymm1, ymm2": {
|
||||
"ExpectedInstructionCount": 15,
|
||||
"ExpectedInstructionCount": 19,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x01 256-bit"
|
||||
@@ -65,14 +65,18 @@
|
||||
"splice z2.d, p6, z2.d, z1.d",
|
||||
"mov z1.d, z2.d[2]",
|
||||
"mov z3.d, z2.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #-1",
|
||||
"mov z3.d, p0/m, z1.d",
|
||||
"msr nzcv, x0",
|
||||
"mov z1.d, z2.d[1]",
|
||||
"mov z16.d, z3.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #0",
|
||||
"mov z16.d, p0/m, z1.d"
|
||||
"mov z16.d, p0/m, z1.d",
|
||||
"msr nzcv, x0"
|
||||
]
|
||||
},
|
||||
"vphaddd xmm0, xmm1, xmm2": {
|
||||
@@ -86,7 +90,7 @@
|
||||
]
|
||||
},
|
||||
"vphaddd ymm0, ymm1, ymm2": {
|
||||
"ExpectedInstructionCount": 15,
|
||||
"ExpectedInstructionCount": 19,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x02 256-bit"
|
||||
@@ -99,14 +103,18 @@
|
||||
"splice z2.d, p6, z2.d, z1.d",
|
||||
"mov z1.d, z2.d[2]",
|
||||
"mov z3.d, z2.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #-1",
|
||||
"mov z3.d, p0/m, z1.d",
|
||||
"msr nzcv, x0",
|
||||
"mov z1.d, z2.d[1]",
|
||||
"mov z16.d, z3.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #0",
|
||||
"mov z16.d, p0/m, z1.d"
|
||||
"mov z16.d, p0/m, z1.d",
|
||||
"msr nzcv, x0"
|
||||
]
|
||||
},
|
||||
"vphaddsw xmm0, xmm1, xmm2": {
|
||||
@@ -122,7 +130,7 @@
|
||||
]
|
||||
},
|
||||
"vphaddsw ymm0, ymm1, ymm2": {
|
||||
"ExpectedInstructionCount": 13,
|
||||
"ExpectedInstructionCount": 17,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x03 256-bit"
|
||||
@@ -133,14 +141,18 @@
|
||||
"sqadd z2.h, z2.h, z3.h",
|
||||
"mov z1.d, z2.d[2]",
|
||||
"mov z3.d, z2.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #-1",
|
||||
"mov z3.d, p0/m, z1.d",
|
||||
"msr nzcv, x0",
|
||||
"mov z1.d, z2.d[1]",
|
||||
"mov z16.d, z3.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #0",
|
||||
"mov z16.d, p0/m, z1.d"
|
||||
"mov z16.d, p0/m, z1.d",
|
||||
"msr nzcv, x0"
|
||||
]
|
||||
},
|
||||
"vpmaddubsw xmm0, xmm1, xmm2": {
|
||||
@@ -192,7 +204,7 @@
|
||||
]
|
||||
},
|
||||
"vphsubw ymm0, ymm1, ymm2": {
|
||||
"ExpectedInstructionCount": 13,
|
||||
"ExpectedInstructionCount": 17,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x05 256-bit"
|
||||
@@ -203,14 +215,18 @@
|
||||
"sub z2.h, z2.h, z3.h",
|
||||
"mov z1.d, z2.d[2]",
|
||||
"mov z3.d, z2.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #-1",
|
||||
"mov z3.d, p0/m, z1.d",
|
||||
"msr nzcv, x0",
|
||||
"mov z1.d, z2.d[1]",
|
||||
"mov z16.d, z3.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #0",
|
||||
"mov z16.d, p0/m, z1.d"
|
||||
"mov z16.d, p0/m, z1.d",
|
||||
"msr nzcv, x0"
|
||||
]
|
||||
},
|
||||
"vphsubd xmm0, xmm1, xmm2": {
|
||||
@@ -226,7 +242,7 @@
|
||||
]
|
||||
},
|
||||
"vphsubd ymm0, ymm1, ymm2": {
|
||||
"ExpectedInstructionCount": 13,
|
||||
"ExpectedInstructionCount": 17,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x06 256-bit"
|
||||
@@ -237,14 +253,18 @@
|
||||
"sub z2.s, z2.s, z3.s",
|
||||
"mov z1.d, z2.d[2]",
|
||||
"mov z3.d, z2.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #-1",
|
||||
"mov z3.d, p0/m, z1.d",
|
||||
"msr nzcv, x0",
|
||||
"mov z1.d, z2.d[1]",
|
||||
"mov z16.d, z3.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #0",
|
||||
"mov z16.d, p0/m, z1.d"
|
||||
"mov z16.d, p0/m, z1.d",
|
||||
"msr nzcv, x0"
|
||||
]
|
||||
},
|
||||
"vphsubsw xmm0, xmm1, xmm2": {
|
||||
@@ -260,7 +280,7 @@
|
||||
]
|
||||
},
|
||||
"vphsubsw ymm0, ymm1, ymm2": {
|
||||
"ExpectedInstructionCount": 13,
|
||||
"ExpectedInstructionCount": 17,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x07 256-bit"
|
||||
@@ -271,14 +291,18 @@
|
||||
"sqsub z2.h, z2.h, z3.h",
|
||||
"mov z1.d, z2.d[2]",
|
||||
"mov z3.d, z2.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #-1",
|
||||
"mov z3.d, p0/m, z1.d",
|
||||
"msr nzcv, x0",
|
||||
"mov z1.d, z2.d[1]",
|
||||
"mov z16.d, z3.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #0",
|
||||
"mov z16.d, p0/m, z1.d"
|
||||
"mov z16.d, p0/m, z1.d",
|
||||
"msr nzcv, x0"
|
||||
]
|
||||
},
|
||||
"vpsignb xmm0, xmm1, xmm2": {
|
||||
@@ -1060,7 +1084,7 @@
|
||||
]
|
||||
},
|
||||
"vpackusdw ymm0, ymm1, ymm2": {
|
||||
"ExpectedInstructionCount": 15,
|
||||
"ExpectedInstructionCount": 19,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"Map 2 0b01 0x2b 256-bit"
|
||||
@@ -1073,14 +1097,18 @@
|
||||
"splice z2.h, p6, z2.h, z1.h",
|
||||
"mov z1.d, z2.d[1]",
|
||||
"mov z3.d, z2.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #0",
|
||||
"mov z3.d, p0/m, z1.d",
|
||||
"msr nzcv, x0",
|
||||
"mov z1.d, z2.d[2]",
|
||||
"mov z16.d, z3.d",
|
||||
"mrs x0, nzcv",
|
||||
"index z0.d, #-2, #1",
|
||||
"cmpeq p0.d, p7/z, z0.d, #-1",
|
||||
"mov z16.d, p0/m, z1.d"
|
||||
"mov z16.d, p0/m, z1.d",
|
||||
"msr nzcv, x0"
|
||||
]
|
||||
},
|
||||
"vmaskmovps xmm0, xmm1, [rax]": {
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
Reference in new issue
Block a user