Merge pull request #3131 from Sonicadvance1/optimize_btr

OpcodeDispatcher: Optimize lock btr
This commit is contained in:
Ryan Houdek authored and GitHub committed 2023-09-21 15:06:55 -07:00
commit 2b7e1d10ec
9 files changed
+847 -3

No files matched your search

@@ -361,6 +361,34 @@ DEF_OP(AtomicFetchAnd) {
}
}
DEF_OP(AtomicFetchCLR) {
auto Op = IROp->C<IR::IROp_AtomicFetchCLR>();
uint8_t OpSize = IROp->Size;
LOGMAN_THROW_AA_FMT(OpSize == 8 || OpSize == 4 || OpSize == 2 || OpSize == 1, "Unexpected CAS size");
auto MemSrc = GetReg(Op->Addr.ID());
auto Src = GetReg(Op->Value.ID());
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto SubEmitSize = OpSize == 8 ? ARMEmitter::SubRegSize::i64Bit :
OpSize == 4 ? ARMEmitter::SubRegSize::i32Bit :
OpSize == 2 ? ARMEmitter::SubRegSize::i16Bit :
OpSize == 1 ? ARMEmitter::SubRegSize::i8Bit : ARMEmitter::SubRegSize::i8Bit;
if (CTX->HostFeatures.SupportsAtomics) {
ldclral(SubEmitSize, Src, GetReg(Node), MemSrc);
}
else {
ARMEmitter::BackwardLabel LoopTop;
Bind(&LoopTop);
ldaxr(SubEmitSize, TMP2, MemSrc);
bic(EmitSize, TMP3, TMP2, Src);
stlxr(SubEmitSize, TMP4, TMP3, MemSrc);
cbnz(EmitSize, TMP4, &LoopTop);
mov(EmitSize, GetReg(Node), TMP2.R());
}
}
DEF_OP(AtomicFetchOr) {
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
uint8_t OpSize = IROp->Size;
@@ -909,6 +909,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
REGISTER_OP(ATOMICFETCHADD, AtomicFetchAdd);
REGISTER_OP(ATOMICFETCHSUB, AtomicFetchSub);
REGISTER_OP(ATOMICFETCHAND, AtomicFetchAnd);
REGISTER_OP(ATOMICFETCHCLR, AtomicFetchCLR);
REGISTER_OP(ATOMICFETCHOR, AtomicFetchOr);
REGISTER_OP(ATOMICFETCHXOR, AtomicFetchXor);
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
@@ -295,6 +295,7 @@ private:
DEF_OP(AtomicFetchAdd);
DEF_OP(AtomicFetchSub);
DEF_OP(AtomicFetchAnd);
DEF_OP(AtomicFetchCLR);
DEF_OP(AtomicFetchOr);
DEF_OP(AtomicFetchXor);
DEF_OP(AtomicFetchNeg);
@@ -399,6 +399,92 @@ DEF_OP(AtomicFetchAnd) {
}
}
DEF_OP(AtomicFetchCLR) {
auto Op = IROp->C<IR::IROp_AtomicFetchCLR>();
// TMP1 = rax
Xbyak::Reg MemReg = GetSrc<RA_64>(Op->Addr.ID());
switch (IROp->Size) {
case 1: {
mov(TMP1.cvt8(), byte [MemReg]);
Label Loop;
L(Loop);
mov(TMP2.cvt8(), TMP1.cvt8());
mov(TMP3.cvt8(), TMP1.cvt8());
mov(TMP4.cvt8(), GetSrc<RA_8>(Op->Value.ID()));
not_(TMP4.cvt8());
and_(TMP2.cvt8(), TMP4.cvt8());
// Updates RAX with the value from memory
lock(); cmpxchg(byte [MemReg], TMP2.cvt8());
jne(Loop);
// Result is the previous value from memory, which is currently in TMP3
movzx(GetDst<RA_64>(Node), TMP3.cvt8());
break;
}
case 2: {
mov(TMP1.cvt16(), word [MemReg]);
Label Loop;
L(Loop);
mov(TMP2.cvt16(), TMP1.cvt16());
mov(TMP3.cvt16(), TMP1.cvt16());
mov(TMP4.cvt16(), GetSrc<RA_16>(Op->Value.ID()));
not_(TMP4.cvt16());
and_(TMP2.cvt16(), TMP4.cvt16());
// Updates RAX with the value from memory
lock(); cmpxchg(word [MemReg], TMP2.cvt16());
jne(Loop);
// Result is the previous value from memory, which is currently in TMP3
movzx(GetDst<RA_64>(Node), TMP3.cvt16());
break;
}
case 4: {
mov(TMP1.cvt32(), dword [MemReg]);
Label Loop;
L(Loop);
mov(TMP2.cvt32(), TMP1.cvt32());
mov(TMP3.cvt32(), TMP1.cvt32());
mov(TMP4.cvt32(), GetSrc<RA_32>(Op->Value.ID()));
not_(TMP4.cvt32());
and_(TMP2.cvt32(), TMP4.cvt32());
// Updates RAX with the value from memory
lock(); cmpxchg(dword [MemReg], TMP2.cvt32());
jne(Loop);
// Result is the previous value from memory, which is currently in TMP3
mov(GetDst<RA_32>(Node), TMP3.cvt32());
break;
}
case 8: {
mov(TMP1.cvt64(), qword [MemReg]);
Label Loop;
L(Loop);
mov(TMP2.cvt64(), TMP1.cvt64());
mov(TMP3.cvt64(), TMP1.cvt64());
mov(TMP4.cvt64(), GetSrc<RA_64>(Op->Value.ID()));
not_(TMP4.cvt64());
and_(TMP2.cvt64(), TMP4.cvt64());
// Updates RAX with the value from memory
lock(); cmpxchg(qword [MemReg], TMP2.cvt64());
jne(Loop);
// Result is the previous value from memory, which is currently in TMP3
mov(GetDst<RA_64>(Node), TMP3.cvt64());
break;
}
default: LOGMAN_MSG_A_FMT("Unhandled AtomicFetchAnd size: {}", IROp->Size);
}
}
DEF_OP(AtomicFetchOr) {
auto Op = IROp->C<IR::IROp_AtomicFetchOr>();
@@ -656,6 +742,8 @@ void X86JITCore::RegisterAtomicHandlers() {
REGISTER_OP(ATOMICFETCHADD, AtomicFetchAdd);
REGISTER_OP(ATOMICFETCHSUB, AtomicFetchSub);
REGISTER_OP(ATOMICFETCHAND, AtomicFetchAnd);
REGISTER_OP(ATOMICFETCHCLR, AtomicFetchCLR);
REGISTER_OP(ATOMICFETCHOR, AtomicFetchOr);
REGISTER_OP(ATOMICFETCHXOR, AtomicFetchXor);
REGISTER_OP(ATOMICFETCHNEG, AtomicFetchNeg);
@@ -308,6 +308,7 @@ private:
DEF_OP(AtomicFetchAdd);
DEF_OP(AtomicFetchSub);
DEF_OP(AtomicFetchAnd);
DEF_OP(AtomicFetchCLR);
DEF_OP(AtomicFetchOr);
DEF_OP(AtomicFetchXor);
DEF_OP(AtomicFetchNeg);
@@ -3084,10 +3084,8 @@ void OpDispatchBuilder::BTROp(OpcodeArgs) {
if (DestIsLockedMem(Op)) {
HandledLock = true;
BitMask = _Not(OpSize::i64Bit, BitMask);
// XXX: Technically this can optimize to an AArch64 ldclralb
// We don't current support this IR op though
Result = _AtomicFetchAnd(OpSize::i8Bit, BitMask, MemoryLocation);
Result = _AtomicFetchCLR(OpSize::i8Bit, BitMask, MemoryLocation);
// Now shift in to the correct bit location
Result = _Lshr(IR::SizeToOpSize(std::max<uint8_t>(4u, GetOpSize(Result))), Result, BitSelect);
} else {
+13
View File
@@ -690,6 +690,19 @@
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = AtomicFetchCLR OpSize:#Size, GPR:$Value, GPR:$Addr": {
"HasSideEffects": true,
"Desc": ["Atomic integer fetch and binary clear",
"Atomically fetches %Addr and binary clears %value to the memory location",
"Dest is the value prior to operating on the value in memory",
"Matches ARM ldclral semantics",
"eg: Dest[Addr] &= ~Value"
],
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i8Bit || Size == FEXCore::IR::OpSize::i16Bit || Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = AtomicFetchOr OpSize:#Size, GPR:$Value, GPR:$Addr": {
"HasSideEffects": true,
"Desc": ["Atomic integer fetch and binary or",
+159
View File
@@ -2264,6 +2264,59 @@
"str w20, [x28, #728]"
]
},
"lock bts [rax], bx": {
"ExpectedInstructionCount": 11,
"Optimal": "No",
"Comment": "0x0f 0xb3",
"ExpectedArm64ASM": [
"uxth w20, w7",
"ubfx w21, w20, #0, #3",
"sbfx x20, x20, #3, #13",
"add x20, x4, x20",
"mov w22, #0x1",
"lsl x22, x22, x21",
"ldsetalb w22, w20, [x20]",
"lsr w20, w20, w21",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock bts [rax], ebx": {
"ExpectedInstructionCount": 11,
"Optimal": "No",
"Comment": "0x0f 0xb3",
"ExpectedArm64ASM": [
"mov w20, w7",
"ubfx w21, w20, #0, #3",
"sbfx x20, x20, #3, #29",
"add x20, x4, x20",
"mov w22, #0x1",
"lsl x22, x22, x21",
"ldsetalb w22, w20, [x20]",
"lsr w20, w20, w21",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock bts [rax], rbx": {
"ExpectedInstructionCount": 10,
"Optimal": "No",
"Comment": "0x0f 0xb3",
"ExpectedArm64ASM": [
"ubfx x20, x7, #0, #3",
"asr x21, x7, #3",
"add x21, x4, x21",
"mov w22, #0x1",
"lsl x22, x22, x20",
"ldsetalb w22, w21, [x21]",
"lsr w20, w21, w20",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"imul ax, bx": {
"ExpectedInstructionCount": 13,
"Optimal": "No",
@@ -2657,6 +2710,59 @@
"bfxil x4, x20, #0, #16"
]
},
"lock btr [rax], bx": {
"ExpectedInstructionCount": 11,
"Optimal": "No",
"Comment": "0x0f 0xb3",
"ExpectedArm64ASM": [
"uxth w20, w7",
"ubfx w21, w20, #0, #3",
"sbfx x20, x20, #3, #13",
"add x20, x4, x20",
"mov w22, #0x1",
"lsl x22, x22, x21",
"ldclralb w22, w20, [x20]",
"lsr w20, w20, w21",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btr [rax], ebx": {
"ExpectedInstructionCount": 11,
"Optimal": "No",
"Comment": "0x0f 0xb3",
"ExpectedArm64ASM": [
"mov w20, w7",
"ubfx w21, w20, #0, #3",
"sbfx x20, x20, #3, #29",
"add x20, x4, x20",
"mov w22, #0x1",
"lsl x22, x22, x21",
"ldclralb w22, w20, [x20]",
"lsr w20, w20, w21",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btr [rax], rbx": {
"ExpectedInstructionCount": 10,
"Optimal": "No",
"Comment": "0x0f 0xb3",
"ExpectedArm64ASM": [
"ubfx x20, x7, #0, #3",
"asr x21, x7, #3",
"add x21, x4, x21",
"mov w22, #0x1",
"lsl x22, x22, x20",
"ldclralb w22, w21, [x21]",
"lsr w20, w21, w20",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"movzx ax, byte [rax]": {
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
@@ -2835,6 +2941,59 @@
"str w20, [x28, #728]"
]
},
"lock btc [rax], bx": {
"ExpectedInstructionCount": 11,
"Optimal": "No",
"Comment": "0x0f 0xb3",
"ExpectedArm64ASM": [
"uxth w20, w7",
"ubfx w21, w20, #0, #3",
"sbfx x20, x20, #3, #13",
"add x20, x4, x20",
"mov w22, #0x1",
"lsl x22, x22, x21",
"ldeoralb w22, w20, [x20]",
"lsr w20, w20, w21",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btc [rax], ebx": {
"ExpectedInstructionCount": 11,
"Optimal": "No",
"Comment": "0x0f 0xb3",
"ExpectedArm64ASM": [
"mov w20, w7",
"ubfx w21, w20, #0, #3",
"sbfx x20, x20, #3, #29",
"add x20, x4, x20",
"mov w22, #0x1",
"lsl x22, x22, x21",
"ldeoralb w22, w20, [x20]",
"lsr w20, w20, w21",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btc [rax], rbx": {
"ExpectedInstructionCount": 10,
"Optimal": "No",
"Comment": "0x0f 0xb3",
"ExpectedArm64ASM": [
"ubfx x20, x7, #0, #3",
"asr x21, x7, #3",
"add x21, x4, x21",
"mov w22, #0x1",
"lsl x22, x22, x20",
"ldeoralb w22, w21, [x21]",
"lsr w20, w21, w20",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"bsf ax, bx": {
"ExpectedInstructionCount": 14,
"Optimal": "No",
@@ -84,6 +84,75 @@
"str w20, [x28, #728]"
]
},
"bt word [rax], 0": {
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"bt dword [rax], 0": {
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"bt qword [rax], 0": {
"ExpectedInstructionCount": 4,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"bt word [rax], 15": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4, #1]",
"lsr w20, w20, #7",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"bt dword [rax], 31": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4, #3]",
"lsr w20, w20, #7",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"bt qword [rax], 63": {
"ExpectedInstructionCount": 5,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4, #7]",
"lsr w20, w20, #7",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"bts ax, 0": {
"ExpectedInstructionCount": 8,
"Optimal": "No",
@@ -162,6 +231,168 @@
"str w20, [x28, #728]"
]
},
"bts word [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4]",
"orr x21, x20, #0x1",
"strb w21, [x4]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"bts dword [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4]",
"orr x21, x20, #0x1",
"strb w21, [x4]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"bts qword [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4]",
"orr x21, x20, #0x1",
"strb w21, [x4]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"bts word [rax], 15": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4, #1]",
"lsr w21, w20, #7",
"orr x20, x20, #0x80",
"strb w20, [x4, #1]",
"ubfx w20, w21, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"bts dword [rax], 31": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4, #3]",
"lsr w21, w20, #7",
"orr x20, x20, #0x80",
"strb w20, [x4, #3]",
"ubfx w20, w21, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"bts qword [rax], 63": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4, #7]",
"lsr w21, w20, #7",
"orr x20, x20, #0x80",
"strb w20, [x4, #7]",
"ubfx w20, w21, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock bts word [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x0 (0)",
"mov w21, #0x1",
"ldsetalb w21, w20, [x20]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock bts dword [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x0 (0)",
"mov w21, #0x1",
"ldsetalb w21, w20, [x20]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock bts qword [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x0 (0)",
"mov w21, #0x1",
"ldsetalb w21, w20, [x20]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock bts word [rax], 15": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x1 (1)",
"mov w21, #0x80",
"ldsetalb w21, w20, [x20]",
"lsr w20, w20, #7",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock bts dword [rax], 31": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x3 (3)",
"mov w21, #0x80",
"ldsetalb w21, w20, [x20]",
"lsr w20, w20, #7",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock bts qword [rax], 63": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x7 (7)",
"mov w21, #0x80",
"ldsetalb w21, w20, [x20]",
"lsr w20, w20, #7",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"btr ax, 0": {
"ExpectedInstructionCount": 8,
"Optimal": "No",
@@ -240,6 +471,168 @@
"str w20, [x28, #728]"
]
},
"btr word [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4]",
"and x21, x20, #0xfffffffffffffffe",
"strb w21, [x4]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"btr dword [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4]",
"and x21, x20, #0xfffffffffffffffe",
"strb w21, [x4]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"btr qword [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4]",
"and x21, x20, #0xfffffffffffffffe",
"strb w21, [x4]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"btr word [rax], 15": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4, #1]",
"lsr w21, w20, #7",
"and x20, x20, #0xffffffffffffff7f",
"strb w20, [x4, #1]",
"ubfx w20, w21, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"btr dword [rax], 31": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4, #3]",
"lsr w21, w20, #7",
"and x20, x20, #0xffffffffffffff7f",
"strb w20, [x4, #3]",
"ubfx w20, w21, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"btr qword [rax], 63": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4, #7]",
"lsr w21, w20, #7",
"and x20, x20, #0xffffffffffffff7f",
"strb w20, [x4, #7]",
"ubfx w20, w21, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btr word [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x0 (0)",
"mov w21, #0x1",
"ldclralb w21, w20, [x20]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btr dword [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x0 (0)",
"mov w21, #0x1",
"ldclralb w21, w20, [x20]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btr qword [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x0 (0)",
"mov w21, #0x1",
"ldclralb w21, w20, [x20]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btr word [rax], 15": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x1 (1)",
"mov w21, #0x80",
"ldclralb w21, w20, [x20]",
"lsr w20, w20, #7",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btr dword [rax], 31": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x3 (3)",
"mov w21, #0x80",
"ldclralb w21, w20, [x20]",
"lsr w20, w20, #7",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btr qword [rax], 63": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x7 (7)",
"mov w21, #0x80",
"ldclralb w21, w20, [x20]",
"lsr w20, w20, #7",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"btc ax, 0": {
"ExpectedInstructionCount": 8,
"Optimal": "No",
@@ -318,6 +711,168 @@
"str w20, [x28, #728]"
]
},
"btc word [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4]",
"eor x21, x20, #0x1",
"strb w21, [x4]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"btc dword [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4]",
"eor x21, x20, #0x1",
"strb w21, [x4]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"btc qword [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4]",
"eor x21, x20, #0x1",
"strb w21, [x4]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"btc word [rax], 15": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4, #1]",
"lsr w21, w20, #7",
"eor x20, x20, #0x80",
"strb w20, [x4, #1]",
"ubfx w20, w21, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"btc dword [rax], 31": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4, #3]",
"lsr w21, w20, #7",
"eor x20, x20, #0x80",
"strb w20, [x4, #3]",
"ubfx w20, w21, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"btc qword [rax], 63": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"ldrb w20, [x4, #7]",
"lsr w21, w20, #7",
"eor x20, x20, #0x80",
"strb w20, [x4, #7]",
"ubfx w20, w21, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btc word [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x0 (0)",
"mov w21, #0x1",
"ldeoralb w21, w20, [x20]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btc dword [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x0 (0)",
"mov w21, #0x1",
"ldeoralb w21, w20, [x20]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btc qword [rax], 0": {
"ExpectedInstructionCount": 6,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x0 (0)",
"mov w21, #0x1",
"ldeoralb w21, w20, [x20]",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btc word [rax], 15": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x1 (1)",
"mov w21, #0x80",
"ldeoralb w21, w20, [x20]",
"lsr w20, w20, #7",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btc dword [rax], 31": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x3 (3)",
"mov w21, #0x80",
"ldeoralb w21, w20, [x20]",
"lsr w20, w20, #7",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"lock btc qword [rax], 63": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "GROUP8 0x0F 0xBA /6",
"ExpectedArm64ASM": [
"add x20, x4, #0x7 (7)",
"mov w21, #0x80",
"ldeoralb w21, w20, [x20]",
"lsr w20, w20, #7",
"ubfx w20, w20, #0, #1",
"lsl x20, x20, #29",
"str w20, [x28, #728]"
]
},
"cmpxchg8b [rbp]": {
"ExpectedInstructionCount": 25,
"Optimal": "No",