mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 09:00:17 +02:00
Merge pull request #3150 from alyssarosenzweig/opt/ornror
Optimize PF calculation in lahf
This commit is contained in:
7 files changed
+61
-45
No files matched your search
@@ -438,6 +438,18 @@ DEF_OP(Orlshr) {
|
||||
}
|
||||
}
|
||||
|
||||
DEF_OP(Ornror) {
|
||||
auto Op = IROp->C<IR::IROp_Ornror>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src1 = GetReg(Op->Src1.ID());
|
||||
|
||||
const auto Src2 = GetReg(Op->Src2.ID());
|
||||
orn(EmitSize, Dst, Src1, Src2, ARMEmitter::ShiftType::ROR, Op->BitShift);
|
||||
}
|
||||
|
||||
DEF_OP(And) {
|
||||
auto Op = IROp->C<IR::IROp_And>();
|
||||
const uint8_t OpSize = IROp->Size;
|
||||
|
||||
@@ -810,6 +810,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry,
|
||||
REGISTER_OP(OR, Or);
|
||||
REGISTER_OP(ORLSHL, Orlshl);
|
||||
REGISTER_OP(ORLSHR, Orlshr);
|
||||
REGISTER_OP(ORNROR, Ornror);
|
||||
REGISTER_OP(AND, And);
|
||||
REGISTER_OP(ANDN, Andn);
|
||||
REGISTER_OP(XOR, Xor);
|
||||
|
||||
@@ -249,6 +249,7 @@ private:
|
||||
DEF_OP(Or);
|
||||
DEF_OP(Orlshl);
|
||||
DEF_OP(Orlshr);
|
||||
DEF_OP(Ornror);
|
||||
DEF_OP(And);
|
||||
DEF_OP(Andn);
|
||||
DEF_OP(Xor);
|
||||
|
||||
@@ -1352,7 +1352,6 @@ private:
|
||||
/**
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
OrderedNode *LoadPF();
|
||||
OrderedNode *LoadPFRaw();
|
||||
OrderedNode *LoadAF();
|
||||
void FixupAF();
|
||||
|
||||
@@ -141,7 +141,6 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
CalculateDeferredFlags();
|
||||
|
||||
OrderedNode *Original = _Constant(0);
|
||||
bool Nonzero = false;
|
||||
|
||||
// SF/ZF and N/Z are together on both arm64 and x86_64, so we special case that.
|
||||
bool GetNZ = (FlagsMask & (1 << FEXCore::X86State::RFLAG_SF_LOC)) &&
|
||||
@@ -151,7 +150,6 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_CF_LOC)) {
|
||||
static_assert(FEXCore::X86State::RFLAG_CF_LOC == 0);
|
||||
Original = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
Nonzero = true;
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < FlagOffsets.size(); ++i) {
|
||||
@@ -162,7 +160,8 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
|
||||
if ((GetNZ && (FlagOffset == FEXCore::X86State::RFLAG_SF_LOC ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_ZF_LOC)) ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_CF_LOC) {
|
||||
FlagOffset == FEXCore::X86State::RFLAG_CF_LOC ||
|
||||
FlagOffset == FEXCore::X86State::RFLAG_PF_LOC) {
|
||||
// Already handled
|
||||
continue;
|
||||
}
|
||||
@@ -170,19 +169,26 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
|
||||
// Note that the Bfi only considers the bottom bit of the flag, the rest of
|
||||
// the byte is allowed to be garbage.
|
||||
OrderedNode *Flag;
|
||||
if (FlagOffset == FEXCore::X86State::RFLAG_PF_LOC)
|
||||
Flag = LoadPF();
|
||||
else if (FlagOffset == FEXCore::X86State::RFLAG_AF_LOC)
|
||||
if (FlagOffset == FEXCore::X86State::RFLAG_AF_LOC)
|
||||
Flag = LoadAF();
|
||||
else
|
||||
Flag = GetRFLAG(FlagOffset);
|
||||
|
||||
if (Nonzero)
|
||||
Original = _Orlshl(OpSize::i64Bit, Original, Flag, FlagOffset);
|
||||
else
|
||||
Original = _Lshl(OpSize::i64Bit, Flag, _Constant(FlagOffset));
|
||||
Original = _Orlshl(OpSize::i64Bit, Original, Flag, FlagOffset);
|
||||
}
|
||||
|
||||
Nonzero = true;
|
||||
// Raw PF value needs to have its bottom bit masked out and inverted. The
|
||||
// naive sequence is and/eor/orlshl. But we can do the inversion implicitly
|
||||
// instead.
|
||||
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_PF_LOC)) {
|
||||
// Set every bit except the bottommost.
|
||||
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(), _Constant(~1ull));
|
||||
|
||||
// Rotate the bottom bit to the appropriate location for PF, so we get
|
||||
// something like 111P1111. Then invert that to get 000p0000. Then OR that
|
||||
// into the flags. This is 1 A64 instruction :-)
|
||||
auto RightRotation = 64 - FEXCore::X86State::RFLAG_PF_LOC;
|
||||
Original = _Ornror(OpSize::i64Bit, Original, OnesInvPF, RightRotation);
|
||||
}
|
||||
|
||||
// OR in the SF/ZF flags at the end, allowing the lshr to fold with the OR
|
||||
@@ -248,14 +254,6 @@ OrderedNode *OpDispatchBuilder::LoadPFRaw() {
|
||||
return _VExtractToGPR(8, 1, Count, 0);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadPF() {
|
||||
// Mask off the bottom bit only.
|
||||
OrderedNode *Bit = _And(OpSize::i64Bit, LoadPFRaw(), _Constant(1));
|
||||
|
||||
// Invert
|
||||
return _Xor(OpSize::i32Bit, Bit, _Constant(1));
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadAF() {
|
||||
// Read the stored byte. This is the XOR of the arguments.
|
||||
auto AFByte = GetRFLAG(FEXCore::X86State::RFLAG_AF_LOC);
|
||||
@@ -283,7 +281,7 @@ void OpDispatchBuilder::FixupAF() {
|
||||
|
||||
void OpDispatchBuilder::CalculatePF(OrderedNode *Res, OrderedNode *condition) {
|
||||
// For shifts, we can only update for nonzero shift. If zero, we nop out the flag write by
|
||||
// writing the existing value. Note we call GetRFLAG directly, rather than LoadPF, because
|
||||
// writing the existing value. Note we call GetRFLAG directly, rather than LoadPFRaw, because
|
||||
// we need the existing /encoded/ value rather than the decoded PF value. In particular,
|
||||
// this does not calculate a popcount.
|
||||
if (condition) {
|
||||
|
||||
@@ -949,6 +949,14 @@
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Ornror OpSize:#Size, GPR:$Src1, GPR:$Src2, u8:$BitShift": {
|
||||
"Desc": ["Integer binary or with NOT on second source and rotation right"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Xor OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer binary exclusive or"
|
||||
],
|
||||
|
||||
@@ -3619,19 +3619,12 @@
|
||||
"ExpectedArm64ASM": []
|
||||
},
|
||||
"pushf": {
|
||||
"ExpectedInstructionCount": 42,
|
||||
"ExpectedInstructionCount": 41,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x9c",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr w20, [x28, #728]",
|
||||
"ubfx w21, w20, #29, #1",
|
||||
"ldrb w22, [x28, #706]",
|
||||
"fmov s2, w22",
|
||||
"cnt v2.16b, v2.16b",
|
||||
"umov w22, v2.b[0]",
|
||||
"and x22, x22, #0x1",
|
||||
"eor w22, w22, #0x1",
|
||||
"orr x21, x21, x22, lsl #2",
|
||||
"ldrb w22, [x28, #708]",
|
||||
"ldrb w23, [x28, #706]",
|
||||
"eor w22, w22, w23",
|
||||
@@ -3661,6 +3654,12 @@
|
||||
"orr x21, x21, x22, lsl #20",
|
||||
"ldrb w22, [x28, #725]",
|
||||
"orr x21, x21, x22, lsl #21",
|
||||
"ldrb w22, [x28, #706]",
|
||||
"fmov s2, w22",
|
||||
"cnt v2.16b, v2.16b",
|
||||
"umov w22, v2.b[0]",
|
||||
"orr x22, x22, #0xfffffffffffffffe",
|
||||
"orn x21, x21, x22, ror #62",
|
||||
"and x20, x20, #0xc0000000",
|
||||
"orr x20, x21, x20, lsr #24",
|
||||
"orr x20, x20, #0x2",
|
||||
@@ -3668,19 +3667,12 @@
|
||||
]
|
||||
},
|
||||
"pushfq": {
|
||||
"ExpectedInstructionCount": 42,
|
||||
"ExpectedInstructionCount": 41,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x9c",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr w20, [x28, #728]",
|
||||
"ubfx w21, w20, #29, #1",
|
||||
"ldrb w22, [x28, #706]",
|
||||
"fmov s2, w22",
|
||||
"cnt v2.16b, v2.16b",
|
||||
"umov w22, v2.b[0]",
|
||||
"and x22, x22, #0x1",
|
||||
"eor w22, w22, #0x1",
|
||||
"orr x21, x21, x22, lsl #2",
|
||||
"ldrb w22, [x28, #708]",
|
||||
"ldrb w23, [x28, #706]",
|
||||
"eor w22, w22, w23",
|
||||
@@ -3710,6 +3702,12 @@
|
||||
"orr x21, x21, x22, lsl #20",
|
||||
"ldrb w22, [x28, #725]",
|
||||
"orr x21, x21, x22, lsl #21",
|
||||
"ldrb w22, [x28, #706]",
|
||||
"fmov s2, w22",
|
||||
"cnt v2.16b, v2.16b",
|
||||
"umov w22, v2.b[0]",
|
||||
"orr x22, x22, #0xfffffffffffffffe",
|
||||
"orn x21, x21, x22, ror #62",
|
||||
"and x20, x20, #0xc0000000",
|
||||
"orr x20, x21, x20, lsr #24",
|
||||
"orr x20, x20, #0x2",
|
||||
@@ -3790,24 +3788,23 @@
|
||||
]
|
||||
},
|
||||
"lahf": {
|
||||
"ExpectedInstructionCount": 18,
|
||||
"ExpectedInstructionCount": 17,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x9f",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr w20, [x28, #728]",
|
||||
"ubfx w21, w20, #29, #1",
|
||||
"ldrb w22, [x28, #706]",
|
||||
"fmov s2, w22",
|
||||
"cnt v2.16b, v2.16b",
|
||||
"umov w22, v2.b[0]",
|
||||
"and x22, x22, #0x1",
|
||||
"eor w22, w22, #0x1",
|
||||
"orr x21, x21, x22, lsl #2",
|
||||
"ldrb w22, [x28, #708]",
|
||||
"ldrb w23, [x28, #706]",
|
||||
"eor w22, w22, w23",
|
||||
"ubfx w22, w22, #4, #1",
|
||||
"orr x21, x21, x22, lsl #4",
|
||||
"ldrb w22, [x28, #706]",
|
||||
"fmov s2, w22",
|
||||
"cnt v2.16b, v2.16b",
|
||||
"umov w22, v2.b[0]",
|
||||
"orr x22, x22, #0xfffffffffffffffe",
|
||||
"orn x21, x21, x22, ror #62",
|
||||
"and x20, x20, #0xc0000000",
|
||||
"orr x20, x21, x20, lsr #24",
|
||||
"orr x20, x20, #0x2",
|
||||
|
||||
Reference in new issue
Block a user