Merge pull request #3059 from alyssarosenzweig/flag/defer-af-xor

Defer second XOR for AF
This commit is contained in:
Ryan Houdek authored and GitHub committed 2023-09-05 11:38:47 -07:00
commit a8bc6bbb2e
11 files changed
+300 -380

No files matched your search

+7 -4
View File
@@ -252,12 +252,13 @@ namespace FEXCore::Context {
// PF calculation is deferred, calculate it now.
// Popcount the 8-bit flag and then extract the lower bit.
uint32_t PF = std::popcount(Frame->State.flags[X86State::RFLAG_PF_LOC]) & 1;
uint32_t PFByte = Frame->State.flags[X86State::RFLAG_PF_LOC];
uint32_t PF = std::popcount(PFByte) & 1;
EFLAGS |= PF << X86State::RFLAG_PF_LOC;
// AF calculation is deferred, calculate it now.
// extract bit 4.
uint32_t AF = (Frame->State.flags[X86State::RFLAG_AF_LOC] & (1 << 4)) ? 1 : 0;
// XOR with PF byte and extract bit 4.
uint32_t AF = ((Frame->State.flags[X86State::RFLAG_AF_LOC] ^ PFByte) & (1 << 4)) ? 1 : 0;
EFLAGS |= AF << X86State::RFLAG_AF_LOC;
return EFLAGS;
@@ -274,7 +275,9 @@ namespace FEXCore::Context {
// Intentionally do nothing.
break;
case X86State::RFLAG_AF_LOC:
// AF stored in bit 4 in our internal representation
// AF stored in bit 4 in our internal representation. It is also
// XORed with byte 4 of the PF byte, but we write that as zero here so
// we don't need any special handling for that.
Frame->State.flags[i] = (EFLAGS & (1U << i)) ? (1 << 4) : 0;
break;
case X86State::RFLAG_PF_LOC:
@@ -3485,6 +3485,7 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(OpSize::i64Bit, AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(OpSize::i64Bit, AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
CalculatePFUncheckedABI(AL);
FixupAF();
}
void OpDispatchBuilder::DASOp(OpcodeArgs) {
@@ -3557,6 +3558,7 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(OpSize::i64Bit, AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(OpSize::i64Bit, AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
CalculatePFUncheckedABI(AL);
FixupAF();
}
void OpDispatchBuilder::AAAOp(OpcodeArgs) {
@@ -3653,6 +3655,7 @@ void OpDispatchBuilder::AAMOp(OpcodeArgs) {
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(OpSize::i64Bit, AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(OpSize::i64Bit, AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
CalculatePFUncheckedABI(AL);
_InvalidateFlags(1u << X86State::RFLAG_AF_LOC);
}
void OpDispatchBuilder::AADOp(OpcodeArgs) {
@@ -3670,6 +3673,7 @@ void OpDispatchBuilder::AADOp(OpcodeArgs) {
SetRFLAG<FEXCore::X86State::RFLAG_SF_LOC>(_Select(FEXCore::IR::COND_UGE, _And(OpSize::i64Bit, AL, _Constant(0x80)), _Constant(0), _Constant(1), _Constant(0)));
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(_Select(FEXCore::IR::COND_EQ, _And(OpSize::i64Bit, AL, _Constant(0xFF)), _Constant(0), _Constant(1), _Constant(0)));
CalculatePFUncheckedABI(AL);
_InvalidateFlags(1u << X86State::RFLAG_AF_LOC);
}
void OpDispatchBuilder::XLATOp(OpcodeArgs) {
@@ -1378,6 +1378,7 @@ private:
* @{ */
OrderedNode *LoadPF();
OrderedNode *LoadAF();
void FixupAF();
void CalculatePFUncheckedABI(OrderedNode *Res, OrderedNode *condition = nullptr);
void CalculatePF(OrderedNode *Res, OrderedNode *condition = nullptr);
void CalculateAF(OpSize OpSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2);
@@ -110,6 +110,9 @@ void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) {
if (FlagOffset == FEXCore::X86State::RFLAG_AF_LOC) {
// AF is in bit 4 architecturally, and we need to store it to bit 4 of our
// AF register, with garbage in the other bits. The extract is deferred.
// We also defer a XOR with the result bit, which is implemented as XOR
// with PF[4]. But the _Bfe below reliably zeros bit 4 of the PF byte, so
// that will be a no-op and we get the right result.
//
// So we write out the whole flags byte to AF without an extract.
static_assert(FEXCore::X86State::RFLAG_AF_LOC == 4);
@@ -152,18 +155,16 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
}
// Note that the Bfi only considers the bottom bit of the flag, the rest of
// the byte is allowed to be garbage. AF is intentionally not using LoadAF.
OrderedNode *Flag = FlagOffset == FEXCore::X86State::RFLAG_PF_LOC ?
LoadPF() :
GetRFLAG(FlagOffset);
// the byte is allowed to be garbage.
OrderedNode *Flag;
if (FlagOffset == FEXCore::X86State::RFLAG_PF_LOC)
Flag = LoadPF();
else if (FlagOffset == FEXCore::X86State::RFLAG_AF_LOC)
Flag = LoadAF();
else
Flag = GetRFLAG(FlagOffset);
// AF is special, since the flag value is at bit 4 of our AF register, and
// we need to put it at bit 4 too. So instead of the usual Bfe+Orlshl we
// instead do And+Or. This is the same instruction count, but fewer cycles.
if (FlagOffset == FEXCore::X86State::RFLAG_AF_LOC) {
auto MaskedAF = _And(OpSize::i32Bit, Flag, _Constant(1u << 4));
Original = _Or(OpSize::i32Bit, Original, MaskedAF);
} else if (CTX->BackendFeatures.SupportsShiftedBitwise) {
if (CTX->BackendFeatures.SupportsShiftedBitwise) {
Original = _Orlshl(OpSize::i64Bit, Original, Flag, FlagOffset);
} else {
Original = _Bfi(OpSize::i32Bit, 1, FlagOffset, Original, Flag);
@@ -211,11 +212,29 @@ OrderedNode *OpDispatchBuilder::LoadPF() {
}
OrderedNode *OpDispatchBuilder::LoadAF() {
// Read the stored byte. This is the XOR result.
// Read the stored byte. This is the XOR of the arguments.
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
// What's left is to extract the flag from the XOR. This is the deferred part.
return _Bfe(OpSize::i32Bit, 1, 4, AFByte);
// Read the result ^ 1, stored as the PF byte for deferred PF calculation.
// This is the same as result as far as the extracted bit 4 is concerned.
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC);
// What's left is to XOR and extract. This is the deferred part.
return _Bfe(OpSize::i32Bit, 1, 4, _Xor(OpSize::i32Bit, AFByte, PFByte));
}
void OpDispatchBuilder::FixupAF() {
// The caller has set a desired value of AF in AF[4], regardless of the value
// of PF. We need to fixup AF[4] so that we get the right value when we XOR in
// PF[4] later. The easiest solution is to XOR by PF[4], since:
//
// (AF[4] ^ PF[4]) ^ PF[4] = AF[4]
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC);
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
OrderedNode *XorRes = _Xor(OpSize::i32Bit, AFByte, PFByte);
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(XorRes);
}
void OpDispatchBuilder::CalculatePFUncheckedABI(OrderedNode *Res, OrderedNode *condition) {
@@ -244,14 +263,18 @@ void OpDispatchBuilder::CalculatePF(OrderedNode *Res, OrderedNode *condition) {
if (!CTX->Config.ABINoPF) {
CalculatePFUncheckedABI(Res, condition);
} else {
_InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC);
// Even if we are skipping PF calculation as a speed hack, we still need to
// zero PF[4] for correct AF. I suspect ABINoPF can be removed now that
// we defer the expensive parts of PF calculation anyway.
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(_Constant(0));
}
}
void OpDispatchBuilder::CalculateAF(OpSize OpSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
// We store the XOR of the arguments. The AF flag will be available as a bit
// to be extracted from this result. The extraction is deferred until read.
OrderedNode *XorRes = _Xor(OpSize, _Xor(OpSize, Src1, Src2), Res);
// We store the XOR of the arguments. At read time, we XOR with the
// appropriate bit of the result (available as the PF flag) and extract the
// appropriate bit.
OrderedNode *XorRes = _Xor(OpSize, Src1, Src2);
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(XorRes);
}
@@ -986,6 +1009,9 @@ void OpDispatchBuilder::CalculateFlags_FCMP(uint8_t SrcSize, OrderedNode *Res, O
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(HostFlag_ZF);
SetRFLAG<FEXCore::X86State::RFLAG_PF_LOC>(HostFlag_Unordered);
// Zero AF. Note that we set the PF byte to 0/1 above, so PF[4] is 0 so the
// XOR with PF will have no effect, so setting the AF byte to zero will indeed
// zero AF as intended.
uint32_t FlagsMaskToZero =
(1U << X86State::RFLAG_AF_LOC) |
(1U << X86State::RFLAG_SF_LOC) |
+3
View File
@@ -30,6 +30,7 @@ test rax, rdx
mov rax, 0
lahf
mov rcx, rax
and rcx, 0xffffffffffffefff
mov rax, 0x5152535455565758
mov rdx, 0x71727374
@@ -50,6 +51,7 @@ test eax, edx
mov rax, 0
lahf
mov rbx, rax
and rbx, 0xffffffffffffefff
mov rax, 0x4142434445464748
mov rdx, 0x7172
@@ -69,5 +71,6 @@ test ax, dx
mov rax, 0
lahf
and rax, 0xffffffffffffefff
hlt
+3
View File
@@ -29,6 +29,7 @@ test rax, 0x71727374
mov rax, 0
lahf
mov rcx, rax
and rcx, 0xffffffffffffefff
mov rax, 0x5152535455565758
test eax, 0x71727374
@@ -48,6 +49,7 @@ test eax, 0x71727374
mov rax, 0
lahf
mov rbx, rax
and rbx, 0xffffffffffffefff
mov rax, 0x4142434445464748
test ax, 0x7172
@@ -66,5 +68,6 @@ test ax, 0x7172
mov rax, 0
lahf
and rax, 0xffffffffffffefff
hlt
+3
View File
@@ -37,6 +37,7 @@ test qword [rdx + 8 * 2], 0x71727374
mov rax, 0
lahf
mov rcx, rax
and rcx, 0xffffffffffffefff
test dword [rdx + 8 * 1], 0x71727374
; test = 0x55565758 & 0x71727374 = 0x51525350
@@ -55,6 +56,7 @@ test dword [rdx + 8 * 1], 0x71727374
mov rax, 0
lahf
mov rbx, rax
and rbx, 0xffffffffffffefff
test word [rdx + 8 * 0], 0x7172
; test = 0x4748 & 0x7172 = 0x4140
@@ -72,6 +74,7 @@ test word [rdx + 8 * 0], 0x7172
mov rax, 0
lahf
and rax, 0xffffffffffffefff
hlt
File diff suppressed because it is too large. Load diff
+43 -82
View File
@@ -13,7 +13,7 @@
],
"Instructions": {
"add al, 1": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": "GROUP1 0x80 /0",
"ExpectedArm64ASM": [
@@ -22,7 +22,6 @@
"bfxil x4, x21, #0, #8",
"uxtb w21, w21",
"eor w22, w20, #0x1",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x22, x21, #0x1",
"strb w22, [x28, #706]",
@@ -61,7 +60,7 @@
]
},
"adc al, 1": {
"ExpectedInstructionCount": 31,
"ExpectedInstructionCount": 30,
"Optimal": "No",
"Comment": "GROUP1 0x80 /2",
"ExpectedArm64ASM": [
@@ -74,7 +73,6 @@
"bfxil x4, x20, #0, #8",
"uxtb w20, w20",
"eor w23, w22, #0x1",
"eor w23, w23, w20",
"strb w23, [x28, #708]",
"eor x23, x20, #0x1",
"strb w23, [x28, #706]",
@@ -99,7 +97,7 @@
]
},
"sbb al, 1": {
"ExpectedInstructionCount": 31,
"ExpectedInstructionCount": 30,
"Optimal": "No",
"Comment": "GROUP1 0x80 /3",
"ExpectedArm64ASM": [
@@ -112,7 +110,6 @@
"bfxil x4, x20, #0, #8",
"uxtb w20, w20",
"eor w23, w22, #0x1",
"eor w23, w23, w20",
"strb w23, [x28, #708]",
"eor x23, x20, #0x1",
"strb w23, [x28, #706]",
@@ -155,7 +152,7 @@
]
},
"sub al, 1": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": "GROUP1 0x80 /5",
"ExpectedArm64ASM": [
@@ -164,7 +161,6 @@
"bfxil x4, x21, #0, #8",
"uxtb w21, w21",
"eor w22, w20, #0x1",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x22, x21, #0x1",
"strb w22, [x28, #706]",
@@ -203,7 +199,7 @@
]
},
"cmp al, 1": {
"ExpectedInstructionCount": 22,
"ExpectedInstructionCount": 21,
"Optimal": "No",
"Comment": "GROUP1 0x80 /7",
"ExpectedArm64ASM": [
@@ -211,7 +207,6 @@
"sub w21, w20, #0x1 (1)",
"uxtb w21, w21",
"eor w22, w20, #0x1",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x22, x21, #0x1",
"strb w22, [x28, #706]",
@@ -232,7 +227,7 @@
]
},
"add ax, 256": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
@@ -241,7 +236,6 @@
"bfxil x4, x21, #0, #16",
"uxth w21, w21",
"eor w22, w20, #0x100",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x22, x21, #0x1",
"strb w22, [x28, #706]",
@@ -262,14 +256,13 @@
]
},
"add eax, 256": {
"ExpectedInstructionCount": 10,
"ExpectedInstructionCount": 9,
"Optimal": "No",
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"lsr w20, w4, #0",
"add w4, w20, #0x100 (256)",
"eor w21, w20, #0x100",
"eor w21, w21, w4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
@@ -279,14 +272,13 @@
]
},
"add rax, 256": {
"ExpectedInstructionCount": 10,
"ExpectedInstructionCount": 9,
"Optimal": "Yes",
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov x20, x4",
"add x4, x20, #0x100 (256)",
"eor x21, x20, #0x100",
"eor x21, x21, x4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
@@ -323,7 +315,7 @@
]
},
"adc eax, 256": {
"ExpectedInstructionCount": 26,
"ExpectedInstructionCount": 25,
"Optimal": "No",
"Comment": "GROUP1 0x81 /2",
"ExpectedArm64ASM": [
@@ -334,7 +326,6 @@
"lsr w22, w4, #0",
"add w4, w22, w20",
"eor w20, w22, #0x100",
"eor w20, w20, w4",
"strb w20, [x28, #708]",
"eor x20, x4, #0x1",
"strb w20, [x28, #706]",
@@ -356,7 +347,7 @@
]
},
"adc rax, 256": {
"ExpectedInstructionCount": 26,
"ExpectedInstructionCount": 25,
"Optimal": "No",
"Comment": "GROUP1 0x81 /2",
"ExpectedArm64ASM": [
@@ -367,7 +358,6 @@
"mov x22, x4",
"add x4, x22, x20",
"eor x20, x22, #0x100",
"eor x20, x20, x4",
"strb w20, [x28, #708]",
"eor x20, x4, #0x1",
"strb w20, [x28, #706]",
@@ -389,7 +379,7 @@
]
},
"sbb eax, 256": {
"ExpectedInstructionCount": 26,
"ExpectedInstructionCount": 25,
"Optimal": "No",
"Comment": "GROUP1 0x81 /3",
"ExpectedArm64ASM": [
@@ -400,7 +390,6 @@
"lsr w22, w4, #0",
"sub w4, w22, w20",
"eor w20, w22, #0x100",
"eor w20, w20, w4",
"strb w20, [x28, #708]",
"eor x20, x4, #0x1",
"strb w20, [x28, #706]",
@@ -422,7 +411,7 @@
]
},
"sbb rax, 256": {
"ExpectedInstructionCount": 26,
"ExpectedInstructionCount": 25,
"Optimal": "No",
"Comment": "GROUP1 0x81 /3",
"ExpectedArm64ASM": [
@@ -433,7 +422,6 @@
"mov x22, x4",
"sub x4, x22, x20",
"eor x20, x22, #0x100",
"eor x20, x20, x4",
"strb w20, [x28, #708]",
"eor x20, x4, #0x1",
"strb w20, [x28, #706]",
@@ -482,14 +470,13 @@
]
},
"sub eax, 256": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 10,
"Optimal": "No",
"Comment": "GROUP1 0x81 /5",
"ExpectedArm64ASM": [
"lsr w20, w4, #0",
"sub w4, w20, #0x100 (256)",
"eor w21, w20, #0x100",
"eor w21, w21, w4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
@@ -500,14 +487,13 @@
]
},
"sub rax, 256": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 10,
"Optimal": "Yes",
"Comment": "GROUP1 0x81 /5",
"ExpectedArm64ASM": [
"mov x20, x4",
"sub x4, x20, #0x100 (256)",
"eor x21, x20, #0x100",
"eor x21, x21, x4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
@@ -545,14 +531,13 @@
]
},
"cmp eax, 256": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 10,
"Optimal": "No",
"Comment": "GROUP1 0x81 /7",
"ExpectedArm64ASM": [
"lsr w20, w4, #0",
"sub w21, w20, #0x100 (256)",
"eor w22, w20, #0x100",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x21, x21, #0x1",
"strb w21, [x28, #706]",
@@ -563,13 +548,12 @@
]
},
"cmp rax, 256": {
"ExpectedInstructionCount": 10,
"ExpectedInstructionCount": 9,
"Optimal": "Yes",
"Comment": "GROUP1 0x81 /7",
"ExpectedArm64ASM": [
"sub x20, x4, #0x100 (256)",
"eor x21, x4, #0x100",
"eor x21, x21, x20",
"strb w21, [x28, #708]",
"eor x20, x20, #0x1",
"strb w20, [x28, #706]",
@@ -580,7 +564,7 @@
]
},
"add ax, 1": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": "GROUP1 0x83 /0",
"ExpectedArm64ASM": [
@@ -589,7 +573,6 @@
"bfxil x4, x21, #0, #16",
"uxth w21, w21",
"eor w22, w20, #0x1",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x22, x21, #0x1",
"strb w22, [x28, #706]",
@@ -610,14 +593,13 @@
]
},
"add eax, 1": {
"ExpectedInstructionCount": 10,
"ExpectedInstructionCount": 9,
"Optimal": "No",
"Comment": "GROUP1 0x83 /0",
"ExpectedArm64ASM": [
"lsr w20, w4, #0",
"add w4, w20, #0x1 (1)",
"eor w21, w20, #0x1",
"eor w21, w21, w4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
@@ -627,14 +609,13 @@
]
},
"add rax, 1": {
"ExpectedInstructionCount": 10,
"ExpectedInstructionCount": 9,
"Optimal": "Yes",
"Comment": "GROUP1 0x83 /0",
"ExpectedArm64ASM": [
"mov x20, x4",
"add x4, x20, #0x1 (1)",
"eor x21, x20, #0x1",
"eor x21, x21, x4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
@@ -671,7 +652,7 @@
]
},
"adc eax, 1": {
"ExpectedInstructionCount": 26,
"ExpectedInstructionCount": 25,
"Optimal": "No",
"Comment": "GROUP1 0x83 /2",
"ExpectedArm64ASM": [
@@ -682,7 +663,6 @@
"lsr w22, w4, #0",
"add w4, w22, w20",
"eor w20, w22, #0x1",
"eor w20, w20, w4",
"strb w20, [x28, #708]",
"eor x20, x4, #0x1",
"strb w20, [x28, #706]",
@@ -704,7 +684,7 @@
]
},
"adc rax, 1": {
"ExpectedInstructionCount": 26,
"ExpectedInstructionCount": 25,
"Optimal": "No",
"Comment": "GROUP1 0x83 /2",
"ExpectedArm64ASM": [
@@ -715,7 +695,6 @@
"mov x22, x4",
"add x4, x22, x20",
"eor x20, x22, #0x1",
"eor x20, x20, x4",
"strb w20, [x28, #708]",
"eor x20, x4, #0x1",
"strb w20, [x28, #706]",
@@ -737,7 +716,7 @@
]
},
"sbb eax, 1": {
"ExpectedInstructionCount": 26,
"ExpectedInstructionCount": 25,
"Optimal": "No",
"Comment": "GROUP1 0x83 /3",
"ExpectedArm64ASM": [
@@ -748,7 +727,6 @@
"lsr w22, w4, #0",
"sub w4, w22, w20",
"eor w20, w22, #0x1",
"eor w20, w20, w4",
"strb w20, [x28, #708]",
"eor x20, x4, #0x1",
"strb w20, [x28, #706]",
@@ -770,7 +748,7 @@
]
},
"sbb rax, 1": {
"ExpectedInstructionCount": 26,
"ExpectedInstructionCount": 25,
"Optimal": "No",
"Comment": "GROUP1 0x83 /3",
"ExpectedArm64ASM": [
@@ -781,7 +759,6 @@
"mov x22, x4",
"sub x4, x22, x20",
"eor x20, x22, #0x1",
"eor x20, x20, x4",
"strb w20, [x28, #708]",
"eor x20, x4, #0x1",
"strb w20, [x28, #706]",
@@ -830,14 +807,13 @@
]
},
"sub eax, 1": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 10,
"Optimal": "No",
"Comment": "GROUP1 0x83 /5",
"ExpectedArm64ASM": [
"lsr w20, w4, #0",
"sub w4, w20, #0x1 (1)",
"eor w21, w20, #0x1",
"eor w21, w21, w4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
@@ -848,14 +824,13 @@
]
},
"sub rax, 1": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 10,
"Optimal": "No",
"Comment": "GROUP1 0x83 /5",
"ExpectedArm64ASM": [
"mov x20, x4",
"sub x4, x20, #0x1 (1)",
"eor x21, x20, #0x1",
"eor x21, x21, x4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
@@ -893,14 +868,13 @@
]
},
"cmp eax, 1": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 10,
"Optimal": "No",
"Comment": "GROUP1 0x83 /7",
"ExpectedArm64ASM": [
"lsr w20, w4, #0",
"sub w21, w20, #0x1 (1)",
"eor w22, w20, #0x1",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x21, x21, #0x1",
"strb w21, [x28, #706]",
@@ -911,13 +885,12 @@
]
},
"cmp rax, 1": {
"ExpectedInstructionCount": 10,
"ExpectedInstructionCount": 9,
"Optimal": "No",
"Comment": "GROUP1 0x83 /7",
"ExpectedArm64ASM": [
"sub x20, x4, #0x1 (1)",
"eor x21, x4, #0x1",
"eor x21, x21, x20",
"strb w21, [x28, #708]",
"eor x20, x20, #0x1",
"strb w20, [x28, #706]",
@@ -2962,7 +2935,7 @@
]
},
"neg bl": {
"ExpectedInstructionCount": 21,
"ExpectedInstructionCount": 20,
"Optimal": "No",
"Comment": "GROUP2 0xf6 /3",
"ExpectedArm64ASM": [
@@ -2971,8 +2944,7 @@
"neg w22, w21",
"bfxil x7, x22, #0, #8",
"uxtb w22, w22",
"eor w23, w21, w22",
"strb w23, [x28, #708]",
"strb w21, [x28, #708]",
"eor x23, x22, #0x1",
"strb w23, [x28, #706]",
"lsl w23, w22, #24",
@@ -3137,7 +3109,7 @@
]
},
"neg bx": {
"ExpectedInstructionCount": 21,
"ExpectedInstructionCount": 20,
"Optimal": "No",
"Comment": "GROUP2 0xf7 /2",
"ExpectedArm64ASM": [
@@ -3146,8 +3118,7 @@
"neg w22, w21",
"bfxil x7, x22, #0, #16",
"uxth w22, w22",
"eor w23, w21, w22",
"strb w23, [x28, #708]",
"strb w21, [x28, #708]",
"eor x23, x22, #0x1",
"strb w23, [x28, #706]",
"lsl w23, w22, #16",
@@ -3165,14 +3136,13 @@
]
},
"neg ebx": {
"ExpectedInstructionCount": 10,
"ExpectedInstructionCount": 9,
"Optimal": "No",
"Comment": "GROUP2 0xf7 /2",
"ExpectedArm64ASM": [
"lsr w20, w7, #0",
"neg w7, w20",
"eor w21, w20, w7",
"strb w21, [x28, #708]",
"strb w20, [x28, #708]",
"eor x21, x7, #0x1",
"strb w21, [x28, #706]",
"cmp wzr, w20",
@@ -3182,14 +3152,13 @@
]
},
"neg rbx": {
"ExpectedInstructionCount": 10,
"ExpectedInstructionCount": 9,
"Optimal": "No",
"Comment": "GROUP2 0xf7 /2",
"ExpectedArm64ASM": [
"mov x20, x7",
"neg x7, x20",
"eor x21, x20, x7",
"strb w21, [x28, #708]",
"strb w20, [x28, #708]",
"eor x21, x7, #0x1",
"strb w21, [x28, #706]",
"cmp xzr, x20",
@@ -3452,7 +3421,7 @@
]
},
"inc al": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": "GROUP3 0xfe /0",
"ExpectedArm64ASM": [
@@ -3461,7 +3430,6 @@
"bfxil x4, x21, #0, #8",
"uxtb w21, w21",
"eor w22, w20, #0x1",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x22, x21, #0x1",
"strb w22, [x28, #706]",
@@ -3482,7 +3450,7 @@
]
},
"dec al": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": "GROUP3 0xfe /1",
"ExpectedArm64ASM": [
@@ -3491,7 +3459,6 @@
"bfxil x4, x21, #0, #8",
"uxtb w21, w21",
"eor w22, w20, #0x1",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x22, x21, #0x1",
"strb w22, [x28, #706]",
@@ -3512,7 +3479,7 @@
]
},
"inc ax": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": "GROUP4 0xfe /0",
"ExpectedArm64ASM": [
@@ -3521,7 +3488,6 @@
"bfxil x4, x21, #0, #16",
"uxth w21, w21",
"eor w22, w20, #0x1",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x22, x21, #0x1",
"strb w22, [x28, #706]",
@@ -3542,14 +3508,13 @@
]
},
"inc eax": {
"ExpectedInstructionCount": 13,
"ExpectedInstructionCount": 12,
"Optimal": "No",
"Comment": "GROUP4 0xfe /0",
"ExpectedArm64ASM": [
"lsr w20, w4, #0",
"add w4, w20, #0x1 (1)",
"eor w21, w20, #0x1",
"eor w21, w21, w4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
@@ -3562,14 +3527,13 @@
]
},
"inc rax": {
"ExpectedInstructionCount": 13,
"ExpectedInstructionCount": 12,
"Optimal": "Yes",
"Comment": "GROUP4 0xfe /0",
"ExpectedArm64ASM": [
"mov x20, x4",
"add x4, x20, #0x1 (1)",
"eor x21, x20, #0x1",
"eor x21, x21, x4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
@@ -3582,7 +3546,7 @@
]
},
"dec ax": {
"ExpectedInstructionCount": 23,
"ExpectedInstructionCount": 22,
"Optimal": "No",
"Comment": "GROUP4 0xfe /1",
"ExpectedArm64ASM": [
@@ -3591,7 +3555,6 @@
"bfxil x4, x21, #0, #16",
"uxth w21, w21",
"eor w22, w20, #0x1",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x22, x21, #0x1",
"strb w22, [x28, #706]",
@@ -3612,14 +3575,13 @@
]
},
"dec eax": {
"ExpectedInstructionCount": 14,
"ExpectedInstructionCount": 13,
"Optimal": "No",
"Comment": "GROUP4 0xfe /1",
"ExpectedArm64ASM": [
"lsr w20, w4, #0",
"sub w4, w20, #0x1 (1)",
"eor w21, w20, #0x1",
"eor w21, w21, w4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
@@ -3633,14 +3595,13 @@
]
},
"dec rax": {
"ExpectedInstructionCount": 14,
"ExpectedInstructionCount": 13,
"Optimal": "Yes",
"Comment": "GROUP4 0xfe /1",
"ExpectedArm64ASM": [
"mov x20, x4",
"sub x4, x20, #0x1 (1)",
"eor x21, x20, #0x1",
"eor x21, x21, x4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
+22 -12
View File
@@ -94,13 +94,15 @@
]
},
"daa": {
"ExpectedInstructionCount": 55,
"ExpectedInstructionCount": 60,
"Optimal": "No",
"Comment": "0x27",
"ExpectedArm64ASM": [
"ldr w20, [x28, #728]",
"ubfx w21, w20, #29, #1",
"ldrb w22, [x28, #708]",
"ldrb w23, [x28, #706]",
"eor w22, w22, w23",
"ubfx w22, w22, #4, #1",
"uxtb w23, w4",
"and w20, w20, #0xdfffffff",
@@ -152,17 +154,22 @@
"bfi w21, w22, #30, #1",
"eor x20, x20, #0x1",
"strb w20, [x28, #706]",
"ldrb w22, [x28, #708]",
"eor w20, w22, w20",
"strb w20, [x28, #708]",
"str w21, [x28, #728]"
]
},
"das": {
"ExpectedInstructionCount": 55,
"ExpectedInstructionCount": 60,
"Optimal": "No",
"Comment": "0x2f",
"ExpectedArm64ASM": [
"ldr w20, [x28, #728]",
"ubfx w21, w20, #29, #1",
"ldrb w22, [x28, #708]",
"ldrb w23, [x28, #706]",
"eor w22, w22, w23",
"ubfx w22, w22, #4, #1",
"uxtb w23, w4",
"and w20, w20, #0xdfffffff",
@@ -214,15 +221,20 @@
"bfi w21, w22, #30, #1",
"eor x20, x20, #0x1",
"strb w20, [x28, #706]",
"ldrb w22, [x28, #708]",
"eor w20, w22, w20",
"strb w20, [x28, #708]",
"str w21, [x28, #728]"
]
},
"aaa": {
"ExpectedInstructionCount": 29,
"ExpectedInstructionCount": 31,
"Optimal": "No",
"Comment": "0x37",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #708]",
"ldrb w21, [x28, #706]",
"eor w20, w20, w21",
"ubfx w20, w20, #4, #1",
"uxtb w21, w4",
"uxth w22, w4",
@@ -254,11 +266,13 @@
]
},
"aas": {
"ExpectedInstructionCount": 30,
"ExpectedInstructionCount": 32,
"Optimal": "No",
"Comment": "0x3f",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #708]",
"ldrb w21, [x28, #706]",
"eor w20, w20, w21",
"ubfx w20, w20, #4, #1",
"uxtb w21, w4",
"uxth w22, w4",
@@ -291,7 +305,7 @@
]
},
"inc ax": {
"ExpectedInstructionCount": 24,
"ExpectedInstructionCount": 23,
"Optimal": "No",
"Comment": "0x40",
"ExpectedArm64ASM": [
@@ -301,7 +315,6 @@
"bfxil x4, x4, #0, #32",
"uxth w21, w21",
"eor w22, w20, #0x1",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x22, x21, #0x1",
"strb w22, [x28, #706]",
@@ -322,7 +335,7 @@
]
},
"inc eax": {
"ExpectedInstructionCount": 14,
"ExpectedInstructionCount": 13,
"Optimal": "No",
"Comment": "0x40",
"ExpectedArm64ASM": [
@@ -330,7 +343,6 @@
"add w4, w20, #0x1 (1)",
"bfxil x4, x4, #0, #32",
"eor w21, w20, #0x1",
"eor w21, w21, w4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
@@ -343,7 +355,7 @@
]
},
"dec ax": {
"ExpectedInstructionCount": 24,
"ExpectedInstructionCount": 23,
"Optimal": "No",
"Comment": "0x48",
"ExpectedArm64ASM": [
@@ -353,7 +365,6 @@
"bfxil x4, x4, #0, #32",
"uxth w21, w21",
"eor w22, w20, #0x1",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x22, x21, #0x1",
"strb w22, [x28, #706]",
@@ -374,7 +385,7 @@
]
},
"dec eax": {
"ExpectedInstructionCount": 15,
"ExpectedInstructionCount": 14,
"Optimal": "No",
"Comment": "0x48",
"ExpectedArm64ASM": [
@@ -382,7 +393,6 @@
"sub w4, w20, #0x1 (1)",
"bfxil x4, x4, #0, #32",
"eor w21, w20, #0x1",
"eor w21, w21, w4",
"strb w21, [x28, #708]",
"eor x21, x4, #0x1",
"strb w21, [x28, #706]",
+16 -32
View File
@@ -2494,7 +2494,7 @@
]
},
"cmpxchg al, bl": {
"ExpectedInstructionCount": 30,
"ExpectedInstructionCount": 29,
"Optimal": "No",
"Comment": "0x0f 0xb0",
"ExpectedArm64ASM": [
@@ -2510,7 +2510,6 @@
"sub x20, x22, x23",
"uxtb x20, w20",
"eor w21, w22, w23",
"eor w21, w21, w20",
"strb w21, [x28, #708]",
"eor x21, x20, #0x1",
"strb w21, [x28, #706]",
@@ -2531,7 +2530,7 @@
]
},
"cmpxchg [rax], bl": {
"ExpectedInstructionCount": 27,
"ExpectedInstructionCount": 26,
"Optimal": "No",
"Comment": "0x0f 0xb0",
"ExpectedArm64ASM": [
@@ -2544,7 +2543,6 @@
"sub w22, w21, w20",
"uxtb w22, w22",
"eor w23, w21, w20",
"eor w23, w23, w22",
"strb w23, [x28, #708]",
"eor x23, x22, #0x1",
"strb w23, [x28, #706]",
@@ -2565,7 +2563,7 @@
]
},
"cmpxchg ax, bx": {
"ExpectedInstructionCount": 30,
"ExpectedInstructionCount": 29,
"Optimal": "No",
"Comment": "0x0f 0xb1",
"ExpectedArm64ASM": [
@@ -2581,7 +2579,6 @@
"sub x20, x22, x23",
"uxth x20, w20",
"eor w21, w22, w23",
"eor w21, w21, w20",
"strb w21, [x28, #708]",
"eor x21, x20, #0x1",
"strb w21, [x28, #706]",
@@ -2602,7 +2599,7 @@
]
},
"cmpxchg [rax], bx": {
"ExpectedInstructionCount": 27,
"ExpectedInstructionCount": 26,
"Optimal": "No",
"Comment": "0x0f 0xb1",
"ExpectedArm64ASM": [
@@ -2615,7 +2612,6 @@
"sub w22, w21, w20",
"uxth w22, w22",
"eor w23, w21, w20",
"eor w23, w23, w22",
"strb w23, [x28, #708]",
"eor x23, x22, #0x1",
"strb w23, [x28, #706]",
@@ -2636,7 +2632,7 @@
]
},
"cmpxchg eax, ebx": {
"ExpectedInstructionCount": 22,
"ExpectedInstructionCount": 21,
"Optimal": "No",
"Comment": "0x0f 0xb1",
"ExpectedArm64ASM": [
@@ -2654,7 +2650,6 @@
"mov x4, x20",
"sub x20, x24, x25",
"eor w21, w24, w25",
"eor w21, w21, w20",
"strb w21, [x28, #708]",
"eor x20, x20, #0x1",
"strb w20, [x28, #706]",
@@ -2665,7 +2660,7 @@
]
},
"cmpxchg [rax], ebx": {
"ExpectedInstructionCount": 18,
"ExpectedInstructionCount": 17,
"Optimal": "No",
"Comment": "0x0f 0xb1",
"ExpectedArm64ASM": [
@@ -2679,7 +2674,6 @@
"csel x4, x21, x20, eq",
"sub w21, w22, w20",
"eor w23, w22, w20",
"eor w23, w23, w21",
"strb w23, [x28, #708]",
"eor x21, x21, #0x1",
"strb w21, [x28, #706]",
@@ -2690,7 +2684,7 @@
]
},
"cmpxchg rax, rbx": {
"ExpectedInstructionCount": 18,
"ExpectedInstructionCount": 17,
"Optimal": "No",
"Comment": "0x0f 0xb1",
"ExpectedArm64ASM": [
@@ -2704,7 +2698,6 @@
"mov x4, x20",
"sub x20, x21, x22",
"eor x23, x21, x22",
"eor x23, x23, x20",
"strb w23, [x28, #708]",
"eor x20, x20, #0x1",
"strb w20, [x28, #706]",
@@ -2715,7 +2708,7 @@
]
},
"cmpxchg [rax], rbx": {
"ExpectedInstructionCount": 14,
"ExpectedInstructionCount": 13,
"Optimal": "No",
"Comment": "0x0f 0xb1",
"ExpectedArm64ASM": [
@@ -2725,7 +2718,6 @@
"mov x4, x1",
"sub x21, x20, x4",
"eor x22, x20, x4",
"eor x22, x22, x21",
"strb w22, [x28, #708]",
"eor x21, x21, #0x1",
"strb w21, [x28, #706]",
@@ -3275,7 +3267,7 @@
]
},
"xadd al, bl": {
"ExpectedInstructionCount": 25,
"ExpectedInstructionCount": 24,
"Optimal": "No",
"Comment": "0x0f 0xc0",
"ExpectedArm64ASM": [
@@ -3286,7 +3278,6 @@
"bfxil x4, x22, #0, #8",
"uxtb w22, w22",
"eor w23, w20, w21",
"eor w23, w23, w22",
"strb w23, [x28, #708]",
"eor x23, x22, #0x1",
"strb w23, [x28, #706]",
@@ -3307,7 +3298,7 @@
]
},
"xadd [rax], bl": {
"ExpectedInstructionCount": 24,
"ExpectedInstructionCount": 23,
"Optimal": "No",
"Comment": "0x0f 0xc0",
"ExpectedArm64ASM": [
@@ -3317,7 +3308,6 @@
"add w22, w21, w20",
"uxtb w22, w22",
"eor w23, w21, w20",
"eor w23, w23, w22",
"strb w23, [x28, #708]",
"eor x23, x22, #0x1",
"strb w23, [x28, #706]",
@@ -3338,7 +3328,7 @@
]
},
"xadd ax, bx": {
"ExpectedInstructionCount": 25,
"ExpectedInstructionCount": 24,
"Optimal": "No",
"Comment": "0x0f 0xc1",
"ExpectedArm64ASM": [
@@ -3349,7 +3339,6 @@
"bfxil x4, x22, #0, #16",
"uxth w22, w22",
"eor w23, w20, w21",
"eor w23, w23, w22",
"strb w23, [x28, #708]",
"eor x23, x22, #0x1",
"strb w23, [x28, #706]",
@@ -3370,7 +3359,7 @@
]
},
"xadd [rax], bx": {
"ExpectedInstructionCount": 24,
"ExpectedInstructionCount": 23,
"Optimal": "No",
"Comment": "0x0f 0xc1",
"ExpectedArm64ASM": [
@@ -3380,7 +3369,6 @@
"add w22, w21, w20",
"uxth w22, w22",
"eor w23, w21, w20",
"eor w23, w23, w22",
"strb w23, [x28, #708]",
"eor x23, x22, #0x1",
"strb w23, [x28, #706]",
@@ -3401,7 +3389,7 @@
]
},
"xadd eax, ebx": {
"ExpectedInstructionCount": 12,
"ExpectedInstructionCount": 11,
"Optimal": "No",
"Comment": "0x0f 0xc1",
"ExpectedArm64ASM": [
@@ -3410,7 +3398,6 @@
"add w4, w20, w21",
"mov x7, x20",
"eor w22, w20, w21",
"eor w22, w22, w4",
"strb w22, [x28, #708]",
"eor x22, x4, #0x1",
"strb w22, [x28, #706]",
@@ -3420,7 +3407,7 @@
]
},
"xadd [rax], ebx": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 10,
"Optimal": "Yes",
"Comment": "0x0f 0xc1",
"ExpectedArm64ASM": [
@@ -3428,7 +3415,6 @@
"ldaddal w20, w7, [x4]",
"add w21, w7, w20",
"eor w22, w7, w20",
"eor w22, w22, w21",
"strb w22, [x28, #708]",
"eor x21, x21, #0x1",
"strb w21, [x28, #706]",
@@ -3438,7 +3424,7 @@
]
},
"xadd rax, rbx": {
"ExpectedInstructionCount": 12,
"ExpectedInstructionCount": 11,
"Optimal": "No",
"Comment": "0x0f 0xc1",
"ExpectedArm64ASM": [
@@ -3447,7 +3433,6 @@
"add x4, x20, x21",
"mov x7, x20",
"eor x22, x20, x21",
"eor x22, x22, x4",
"strb w22, [x28, #708]",
"eor x22, x4, #0x1",
"strb w22, [x28, #706]",
@@ -3457,7 +3442,7 @@
]
},
"xadd [rax], rbx": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 10,
"Optimal": "Yes",
"Comment": "0x0f 0xc1",
"ExpectedArm64ASM": [
@@ -3465,7 +3450,6 @@
"ldaddal x20, x7, [x4]",
"add x21, x7, x20",
"eor x22, x7, x20",
"eor x22, x22, x21",
"strb w22, [x28, #708]",
"eor x21, x21, #0x1",
"strb w21, [x28, #706]",