Merge pull request #3150 from alyssarosenzweig/opt/ornror

Optimize PF calculation in lahf
This commit is contained in:
Ryan Houdek authored and GitHub committed 2023-09-24 18:05:57 -07:00
commit 19a7b514e6
7 files changed
+61 -45

No files matched your search

@@ -438,6 +438,18 @@ DEF_OP(Orlshr) {
}
}
DEF_OP(Ornror) {
auto Op = IROp->C<IR::IROp_Ornror>();
const uint8_t OpSize = IROp->Size;
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
const auto Dst = GetReg(Node);
const auto Src1 = GetReg(Op->Src1.ID());
const auto Src2 = GetReg(Op->Src2.ID());
orn(EmitSize, Dst, Src1, Src2, ARMEmitter::ShiftType::ROR, Op->BitShift);
}
DEF_OP(And) {
auto Op = IROp->C<IR::IROp_And>();
const uint8_t OpSize = IROp->Size;
@@ -810,6 +810,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
REGISTER_OP(OR, Or);
REGISTER_OP(ORLSHL, Orlshl);
REGISTER_OP(ORLSHR, Orlshr);
REGISTER_OP(ORNROR, Ornror);
REGISTER_OP(AND, And);
REGISTER_OP(ANDN, Andn);
REGISTER_OP(XOR, Xor);
@@ -249,6 +249,7 @@ private:
DEF_OP(Or);
DEF_OP(Orlshl);
DEF_OP(Orlshr);
DEF_OP(Ornror);
DEF_OP(And);
DEF_OP(Andn);
DEF_OP(Xor);
@@ -1352,7 +1352,6 @@ private:
/**
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
* @{ */
OrderedNode *LoadPF();
OrderedNode *LoadPFRaw();
OrderedNode *LoadAF();
void FixupAF();
@@ -141,7 +141,6 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
CalculateDeferredFlags();
OrderedNode *Original = _Constant(0);
bool Nonzero = false;
// SF/ZF and N/Z are together on both arm64 and x86_64, so we special case that.
bool GetNZ = (FlagsMask & (1 << FEXCore::X86State::RFLAG_SF_LOC)) &&
@@ -151,7 +150,6 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_CF_LOC)) {
static_assert(FEXCore::X86State::RFLAG_CF_LOC == 0);
Original = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
Nonzero = true;
}
for (size_t i = 0; i < FlagOffsets.size(); ++i) {
@@ -162,7 +160,8 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
if ((GetNZ && (FlagOffset == FEXCore::X86State::RFLAG_SF_LOC ||
FlagOffset == FEXCore::X86State::RFLAG_ZF_LOC)) ||
FlagOffset == FEXCore::X86State::RFLAG_CF_LOC) {
FlagOffset == FEXCore::X86State::RFLAG_CF_LOC ||
FlagOffset == FEXCore::X86State::RFLAG_PF_LOC) {
// Already handled
continue;
}
@@ -170,19 +169,26 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
// Note that the Bfi only considers the bottom bit of the flag, the rest of
// the byte is allowed to be garbage.
OrderedNode *Flag;
if (FlagOffset == FEXCore::X86State::RFLAG_PF_LOC)
Flag = LoadPF();
else if (FlagOffset == FEXCore::X86State::RFLAG_AF_LOC)
if (FlagOffset == FEXCore::X86State::RFLAG_AF_LOC)
Flag = LoadAF();
else
Flag = GetRFLAG(FlagOffset);
if (Nonzero)
Original = _Orlshl(OpSize::i64Bit, Original, Flag, FlagOffset);
else
Original = _Lshl(OpSize::i64Bit, Flag, _Constant(FlagOffset));
Original = _Orlshl(OpSize::i64Bit, Original, Flag, FlagOffset);
}
Nonzero = true;
// Raw PF value needs to have its bottom bit masked out and inverted. The
// naive sequence is and/eor/orlshl. But we can do the inversion implicitly
// instead.
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_PF_LOC)) {
// Set every bit except the bottommost.
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(), _Constant(~1ull));
// Rotate the bottom bit to the appropriate location for PF, so we get
// something like 111P1111. Then invert that to get 000p0000. Then OR that
// into the flags. This is 1 A64 instruction :-)
auto RightRotation = 64 - FEXCore::X86State::RFLAG_PF_LOC;
Original = _Ornror(OpSize::i64Bit, Original, OnesInvPF, RightRotation);
}
// OR in the SF/ZF flags at the end, allowing the lshr to fold with the OR
@@ -248,14 +254,6 @@ OrderedNode *OpDispatchBuilder::LoadPFRaw() {
return _VExtractToGPR(8, 1, Count, 0);
}
OrderedNode *OpDispatchBuilder::LoadPF() {
// Mask off the bottom bit only.
OrderedNode *Bit = _And(OpSize::i64Bit, LoadPFRaw(), _Constant(1));
// Invert
return _Xor(OpSize::i32Bit, Bit, _Constant(1));
}
OrderedNode *OpDispatchBuilder::LoadAF() {
// Read the stored byte. This is the XOR of the arguments.
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
@@ -283,7 +281,7 @@ void OpDispatchBuilder::FixupAF() {
void OpDispatchBuilder::CalculatePF(OrderedNode *Res, OrderedNode *condition) {
// For shifts, we can only update for nonzero shift. If zero, we nop out the flag write by
// writing the existing value. Note we call GetRFLAG directly, rather than LoadPF, because
// writing the existing value. Note we call GetRFLAG directly, rather than LoadPFRaw, because
// we need the existing /encoded/ value rather than the decoded PF value. In particular,
// this does not calculate a popcount.
if (condition) {
+8
View File
@@ -949,6 +949,14 @@
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Ornror OpSize:#Size, GPR:$Src1, GPR:$Src2, u8:$BitShift": {
"Desc": ["Integer binary or with NOT on second source and rotation right"
],
"DestSize": "Size",
"EmitValidation": [
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
]
},
"GPR = Xor OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Integer binary exclusive or"
],
+21 -24
View File
@@ -3619,19 +3619,12 @@
"ExpectedArm64ASM": []
},
"pushf": {
"ExpectedInstructionCount": 42,
"ExpectedInstructionCount": 41,
"Optimal": "No",
"Comment": "0x9c",
"ExpectedArm64ASM": [
"ldr w20, [x28, #728]",
"ubfx w21, w20, #29, #1",
"ldrb w22, [x28, #706]",
"fmov s2, w22",
"cnt v2.16b, v2.16b",
"umov w22, v2.b[0]",
"and x22, x22, #0x1",
"eor w22, w22, #0x1",
"orr x21, x21, x22, lsl #2",
"ldrb w22, [x28, #708]",
"ldrb w23, [x28, #706]",
"eor w22, w22, w23",
@@ -3661,6 +3654,12 @@
"orr x21, x21, x22, lsl #20",
"ldrb w22, [x28, #725]",
"orr x21, x21, x22, lsl #21",
"ldrb w22, [x28, #706]",
"fmov s2, w22",
"cnt v2.16b, v2.16b",
"umov w22, v2.b[0]",
"orr x22, x22, #0xfffffffffffffffe",
"orn x21, x21, x22, ror #62",
"and x20, x20, #0xc0000000",
"orr x20, x21, x20, lsr #24",
"orr x20, x20, #0x2",
@@ -3668,19 +3667,12 @@
]
},
"pushfq": {
"ExpectedInstructionCount": 42,
"ExpectedInstructionCount": 41,
"Optimal": "No",
"Comment": "0x9c",
"ExpectedArm64ASM": [
"ldr w20, [x28, #728]",
"ubfx w21, w20, #29, #1",
"ldrb w22, [x28, #706]",
"fmov s2, w22",
"cnt v2.16b, v2.16b",
"umov w22, v2.b[0]",
"and x22, x22, #0x1",
"eor w22, w22, #0x1",
"orr x21, x21, x22, lsl #2",
"ldrb w22, [x28, #708]",
"ldrb w23, [x28, #706]",
"eor w22, w22, w23",
@@ -3710,6 +3702,12 @@
"orr x21, x21, x22, lsl #20",
"ldrb w22, [x28, #725]",
"orr x21, x21, x22, lsl #21",
"ldrb w22, [x28, #706]",
"fmov s2, w22",
"cnt v2.16b, v2.16b",
"umov w22, v2.b[0]",
"orr x22, x22, #0xfffffffffffffffe",
"orn x21, x21, x22, ror #62",
"and x20, x20, #0xc0000000",
"orr x20, x21, x20, lsr #24",
"orr x20, x20, #0x2",
@@ -3790,24 +3788,23 @@
]
},
"lahf": {
"ExpectedInstructionCount": 18,
"ExpectedInstructionCount": 17,
"Optimal": "Yes",
"Comment": "0x9f",
"ExpectedArm64ASM": [
"ldr w20, [x28, #728]",
"ubfx w21, w20, #29, #1",
"ldrb w22, [x28, #706]",
"fmov s2, w22",
"cnt v2.16b, v2.16b",
"umov w22, v2.b[0]",
"and x22, x22, #0x1",
"eor w22, w22, #0x1",
"orr x21, x21, x22, lsl #2",
"ldrb w22, [x28, #708]",
"ldrb w23, [x28, #706]",
"eor w22, w22, w23",
"ubfx w22, w22, #4, #1",
"orr x21, x21, x22, lsl #4",
"ldrb w22, [x28, #706]",
"fmov s2, w22",
"cnt v2.16b, v2.16b",
"umov w22, v2.b[0]",
"orr x22, x22, #0xfffffffffffffffe",
"orn x21, x21, x22, ror #62",
"and x20, x20, #0xc0000000",
"orr x20, x21, x20, lsr #24",
"orr x20, x20, #0x2",