mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 09:00:17 +02:00
Merge pull request #3138 from alyssarosenzweig/opt/train
Requiem for the x86 jit
This commit is contained in:
14 files changed
+865
-1322
No files matched your search
@@ -106,12 +106,18 @@ DEF_OP(AddNZCV) {
|
||||
DEF_OP(TestNZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestNZ>();
|
||||
const uint8_t OpSize = Op->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src1.ID());
|
||||
auto Src = GetReg(Op->Src1.ID());
|
||||
|
||||
// Shift the sign bit into place, clearing out the garbage in upper bits.
|
||||
// setf+rmif would avoid the scratch register, but higher latency on M1.
|
||||
if (OpSize < 4) {
|
||||
lsl(EmitSize, Dst, Src, 32 - (OpSize * 8));
|
||||
Src = Dst;
|
||||
}
|
||||
|
||||
tst(EmitSize, Src, Src);
|
||||
|
||||
// TODO: Optimize this out
|
||||
@@ -1223,7 +1229,9 @@ DEF_OP(Sbfe) {
|
||||
|
||||
ARMEmitter::Condition MapSelectCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_ANDZ:
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_ANDNZ:
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
@@ -1257,15 +1265,28 @@ DEF_OP(Select) {
|
||||
|
||||
uint64_t Const;
|
||||
auto cc = MapSelectCC(Op->Cond);
|
||||
bool tests = Op->Cond == FEXCore::IR::COND_ANDZ ||
|
||||
Op->Cond == FEXCore::IR::COND_ANDNZ;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(!tests || IsGPR(Op->Cmp1.ID()), "Only GPRs can be tested");
|
||||
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
const auto Src1 = GetReg(Op->Cmp1.ID());
|
||||
|
||||
if (IsInlineConstant(Op->Cmp2, &Const))
|
||||
cmp(CompareEmitSize, Src1, Const);
|
||||
else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
if (tests) {
|
||||
if (IsInlineConstant(Op->Cmp2, &Const))
|
||||
tst(CompareEmitSize, Src1, Const);
|
||||
else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
tst(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
} else {
|
||||
if (IsInlineConstant(Op->Cmp2, &Const))
|
||||
cmp(CompareEmitSize, Src1, Const);
|
||||
else {
|
||||
const auto Src2 = GetReg(Op->Cmp2.ID());
|
||||
cmp(CompareEmitSize, Src1, Src2);
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (IsGPRPair(Op->Cmp1.ID())) {
|
||||
|
||||
@@ -96,7 +96,9 @@ DEF_OP(Jump) {
|
||||
|
||||
static ARMEmitter::Condition MapBranchCC(IR::CondClassType Cond) {
|
||||
switch (Cond.Val) {
|
||||
case FEXCore::IR::COND_ANDZ:
|
||||
case FEXCore::IR::COND_EQ: return ARMEmitter::Condition::CC_EQ;
|
||||
case FEXCore::IR::COND_ANDNZ:
|
||||
case FEXCore::IR::COND_NEQ: return ARMEmitter::Condition::CC_NE;
|
||||
case FEXCore::IR::COND_SGE: return ARMEmitter::Condition::CC_GE;
|
||||
case FEXCore::IR::COND_SLT: return ARMEmitter::Condition::CC_LT;
|
||||
@@ -129,6 +131,8 @@ DEF_OP(CondJump) {
|
||||
|
||||
uint64_t Const;
|
||||
const bool isConst = IsInlineConstant(Op->Cmp2, &Const);
|
||||
bool tests = Op->Cond == FEXCore::IR::COND_ANDZ ||
|
||||
Op->Cond == FEXCore::IR::COND_ANDNZ;
|
||||
|
||||
const auto Size = Op->CompareSize == 4 ? ARMEmitter::Size::i32Bit : ARMEmitter::Size::i64Bit;
|
||||
const auto SubSize = ARMEmitter::ToVectorSizePair(Op->CompareSize == 4 ? ARMEmitter::SubRegSize::i32Bit : ARMEmitter::SubRegSize::i64Bit);
|
||||
@@ -141,10 +145,18 @@ DEF_OP(CondJump) {
|
||||
cbnz(Size, GetReg(Op->Cmp1.ID()), TrueTargetLabel);
|
||||
} else {
|
||||
if (IsGPR(Op->Cmp1.ID())) {
|
||||
if (isConst) {
|
||||
cmp(Size, GetReg(Op->Cmp1.ID()), Const);
|
||||
if (tests) {
|
||||
if (isConst) {
|
||||
tst(Size, GetReg(Op->Cmp1.ID()), Const);
|
||||
} else {
|
||||
tst(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
|
||||
}
|
||||
} else {
|
||||
cmp(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
|
||||
if (isConst) {
|
||||
cmp(Size, GetReg(Op->Cmp1.ID()), Const);
|
||||
} else {
|
||||
cmp(Size, GetReg(Op->Cmp1.ID()), GetReg(Op->Cmp2.ID()));
|
||||
}
|
||||
}
|
||||
} else if (IsFPR(Op->Cmp1.ID())) {
|
||||
fcmp(SubSize.Scalar, GetVReg(Op->Cmp1.ID()), GetVReg(Op->Cmp2.ID()));
|
||||
|
||||
@@ -831,6 +831,17 @@ void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) {
|
||||
_ExitFunction(JMPPCOffset); // If we get here then leave the function now
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::SelectMask(OrderedNode *Cmp, uint64_t Mask, bool TrueIsNonzero, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue) {
|
||||
return _Select(ResultSize, OpSize::i32Bit,
|
||||
TrueIsNonzero ? CondClassType{COND_ANDNZ} : CondClassType{COND_ANDZ},
|
||||
Cmp, _Constant(Mask),
|
||||
TrueValue, FalseValue);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::SelectNZCV(unsigned BitOffset, bool TrueIsNonzero, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue) {
|
||||
return SelectMask(GetNZCV(), 1u << IndexNZCV(BitOffset), TrueIsNonzero, ResultSize, TrueValue, FalseValue);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::SelectCC(uint8_t OP, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue) {
|
||||
OrderedNode *SrcCond = nullptr;
|
||||
|
||||
@@ -839,77 +850,56 @@ OrderedNode *OpDispatchBuilder::SelectCC(uint8_t OP, IR::OpSize ResultSize, Orde
|
||||
|
||||
switch (OP) {
|
||||
case 0x0: { // JO - Jump if OF == 1
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ},
|
||||
Flag, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = SelectNZCV(FEXCore::X86State::RFLAG_OF_LOC, true, ResultSize, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0x1:{ // JNO - Jump if OF == 0
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ},
|
||||
Flag, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = SelectNZCV(FEXCore::X86State::RFLAG_OF_LOC, false, ResultSize, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0x2: { // JC - Jump if CF == 1
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ},
|
||||
Flag, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = SelectNZCV(FEXCore::X86State::RFLAG_CF_LOC, true, ResultSize, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0x3: { // JNC - Jump if CF == 0
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ},
|
||||
Flag, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = SelectNZCV(FEXCore::X86State::RFLAG_CF_LOC, false, ResultSize, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0x4: { // JE - Jump if ZF == 1
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC);
|
||||
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ},
|
||||
Flag, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = SelectNZCV(FEXCore::X86State::RFLAG_ZF_LOC, true, ResultSize, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0x5: { // JNE - Jump if ZF == 0
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC);
|
||||
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ},
|
||||
Flag, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = SelectNZCV(FEXCore::X86State::RFLAG_ZF_LOC, false, ResultSize, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0x6: { // JNA - Jump if CF == 1 || ZC == 1
|
||||
auto Flag1 = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC);
|
||||
auto Flag2 = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
auto Check = _Or(OpSize::i32Bit, Flag1, Flag2);
|
||||
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ},
|
||||
Check, OneConst, TrueValue, FalseValue);
|
||||
SrcCond = SelectMask(GetNZCV(), (1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_ZF_LOC)),
|
||||
true, ResultSize, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0x7: { // JA - Jump if CF == 0 && ZF == 0
|
||||
auto Flag1 = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC);
|
||||
auto Flag2 = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
|
||||
auto Check = _Or(OpSize::i32Bit, Flag1, Flag2);
|
||||
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ},
|
||||
Check, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = SelectMask(GetNZCV(), (1u << IndexNZCV(FEXCore::X86State::RFLAG_CF_LOC)) |
|
||||
(1u << IndexNZCV(FEXCore::X86State::RFLAG_ZF_LOC)),
|
||||
false, ResultSize, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0x8: { // JS - Jump if SF == 1
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_SF_LOC);
|
||||
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ},
|
||||
Flag, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = SelectNZCV(FEXCore::X86State::RFLAG_SF_LOC, true, ResultSize, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0x9: { // JNS - Jump if SF == 0
|
||||
auto Flag = GetRFLAG(FEXCore::X86State::RFLAG_SF_LOC);
|
||||
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ},
|
||||
Flag, ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = SelectNZCV(FEXCore::X86State::RFLAG_SF_LOC, false, ResultSize, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0xA: { // JP - Jump if PF == 1
|
||||
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ},
|
||||
LoadPFInverted(), ZeroConst, TrueValue, FalseValue);
|
||||
// Raw value contains inverted PF in bottom bit
|
||||
SrcCond = SelectMask(LoadPFRaw(), 0x1, false, ResultSize, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0xB: { // JNP - Jump if PF == 0
|
||||
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ},
|
||||
LoadPFInverted(), ZeroConst, TrueValue, FalseValue);
|
||||
SrcCond = SelectMask(LoadPFRaw(), 0x1, true, ResultSize, TrueValue, FalseValue);
|
||||
break;
|
||||
}
|
||||
case 0xC: { // SF <> OF
|
||||
@@ -927,13 +917,11 @@ OrderedNode *OpDispatchBuilder::SelectCC(uint8_t OP, IR::OpSize ResultSize, Orde
|
||||
break;
|
||||
}
|
||||
case 0xE: {// ZF = 1 || SF <> OF
|
||||
auto Flag1 = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC);
|
||||
auto Select1 = SelectNZCV(FEXCore::X86State::RFLAG_ZF_LOC, true, OpSize::i32Bit,
|
||||
OneConst, ZeroConst);
|
||||
|
||||
auto Flag2 = GetRFLAG(FEXCore::X86State::RFLAG_SF_LOC);
|
||||
auto Flag3 = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
|
||||
auto Select1 = _Select(OpSize::i32Bit, OpSize::i32Bit, CondClassType{COND_EQ},
|
||||
Flag1, OneConst, OneConst, ZeroConst);
|
||||
|
||||
auto Select2 = _Select(OpSize::i32Bit, OpSize::i32Bit, CondClassType{COND_NEQ},
|
||||
Flag2, Flag3, OneConst, ZeroConst);
|
||||
|
||||
@@ -943,13 +931,11 @@ OrderedNode *OpDispatchBuilder::SelectCC(uint8_t OP, IR::OpSize ResultSize, Orde
|
||||
break;
|
||||
}
|
||||
case 0xF: {// ZF = 0 && SF = OF
|
||||
auto Flag1 = GetRFLAG(FEXCore::X86State::RFLAG_ZF_LOC);
|
||||
auto Select1 = SelectNZCV(FEXCore::X86State::RFLAG_ZF_LOC, false, OpSize::i32Bit,
|
||||
OneConst, ZeroConst);
|
||||
|
||||
auto Flag2 = GetRFLAG(FEXCore::X86State::RFLAG_SF_LOC);
|
||||
auto Flag3 = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
|
||||
|
||||
auto Select1 = _Select(OpSize::i32Bit, OpSize::i32Bit, CondClassType{COND_EQ},
|
||||
Flag1, ZeroConst, OneConst, ZeroConst);
|
||||
|
||||
auto Select2 = _Select(OpSize::i32Bit, OpSize::i32Bit, CondClassType{COND_EQ},
|
||||
Flag2, Flag3, OneConst, ZeroConst);
|
||||
|
||||
|
||||
@@ -1137,45 +1137,10 @@ private:
|
||||
SetNZCV(_And(OpSize::i32Bit, OldNZCV, _Constant(PossiblySetNZCVBits)));
|
||||
}
|
||||
|
||||
void SetN_ZeroZCV(unsigned SrcSize, OrderedNode *Res) {
|
||||
static_assert(IndexNZCV(FEXCore::X86State::RFLAG_SF_LOC) == 31);
|
||||
|
||||
unsigned NBit = 31;
|
||||
unsigned SignBit = (SrcSize * 8) - 1;
|
||||
|
||||
OrderedNode *Shifted;
|
||||
|
||||
// Shift the sign bit into the N bit
|
||||
if (SignBit > NBit)
|
||||
Shifted = _Ashr(OpSize::i64Bit, Res, _Constant(SignBit - NBit));
|
||||
else if (SignBit < NBit)
|
||||
Shifted = _Lshl(OpSize::i32Bit, Res, _Constant(NBit - SignBit));
|
||||
else
|
||||
Shifted = Res;
|
||||
|
||||
// Mask off just the N bit, which now equals the sign bit
|
||||
CachedNZCV = _And(OpSize::i32Bit, Shifted, _Constant(1u << NBit));
|
||||
PossiblySetNZCVBits = (1u << NBit);
|
||||
NZCVDirty = true;
|
||||
}
|
||||
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, OrderedNode *Res) {
|
||||
// The TestNZ opcode does this operation natively for 32-bit or 64-bit.
|
||||
// Otherwise we can implement the functionality ourselves with some bit math.
|
||||
if (SrcSize >= 4) {
|
||||
CachedNZCV = _TestNZ(SrcSize, Res);
|
||||
PossiblySetNZCVBits = (1u << 31) | (1u << 30);
|
||||
NZCVDirty = true;
|
||||
} else {
|
||||
// N
|
||||
SetN_ZeroZCV(SrcSize, Res);
|
||||
|
||||
// Z
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto SelectOp = _Select(FEXCore::IR::COND_EQ, Res, Zero, One, Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(SelectOp);
|
||||
}
|
||||
CachedNZCV = _TestNZ(SrcSize, Res);
|
||||
PossiblySetNZCVBits = (1u << 31) | (1u << 30);
|
||||
NZCVDirty = true;
|
||||
}
|
||||
|
||||
OrderedNode *InsertNZCV(OrderedNode *NZCV, unsigned BitOffset, OrderedNode *Value) {
|
||||
@@ -1280,6 +1245,8 @@ private:
|
||||
CachedIndexedNamedVectorConstants.clear();
|
||||
}
|
||||
|
||||
OrderedNode *SelectMask(OrderedNode *Cmp, uint64_t Mask, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
OrderedNode *SelectNZCV(unsigned BitOffset, bool Invert, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
OrderedNode *SelectCC(uint8_t OP, IR::OpSize ResultSize, OrderedNode *TrueValue, OrderedNode *FalseValue);
|
||||
|
||||
/**
|
||||
@@ -1386,7 +1353,7 @@ private:
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
OrderedNode *LoadPF();
|
||||
OrderedNode *LoadPFInverted();
|
||||
OrderedNode *LoadPFRaw();
|
||||
OrderedNode *LoadAF();
|
||||
void FixupAF();
|
||||
void CalculatePF(OrderedNode *Res, OrderedNode *condition = nullptr);
|
||||
|
||||
@@ -209,7 +209,7 @@ void OpDispatchBuilder::CalculateOF_Add(uint8_t SrcSize, OrderedNode *Res, Order
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_OF_LOC>(AndOp1);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadPFInverted() {
|
||||
OrderedNode *OpDispatchBuilder::LoadPFRaw() {
|
||||
// Read the stored byte. This is the original 8-bit result, it needs parity calculated.
|
||||
auto PFByte = GetRFLAG(FEXCore::X86State::RFLAG_PF_LOC);
|
||||
|
||||
@@ -219,14 +219,15 @@ OrderedNode *OpDispatchBuilder::LoadPFInverted() {
|
||||
|
||||
// Calculate the popcount.
|
||||
auto Count = _VPopcount(1, 1, InputFPR);
|
||||
auto Parity = _VExtractToGPR(8, 1, Count, 0);
|
||||
|
||||
// Mask off the bottom bit only.
|
||||
return _And(OpSize::i64Bit, Parity, _Constant(1));
|
||||
return _VExtractToGPR(8, 1, Count, 0);
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadPF() {
|
||||
return _Xor(OpSize::i32Bit, LoadPFInverted(), _Constant(1));
|
||||
// Mask off the bottom bit only.
|
||||
OrderedNode *Bit = _And(OpSize::i64Bit, LoadPFRaw(), _Constant(1));
|
||||
|
||||
// Invert
|
||||
return _Xor(OpSize::i32Bit, Bit, _Constant(1));
|
||||
}
|
||||
|
||||
OrderedNode *OpDispatchBuilder::LoadAF() {
|
||||
@@ -269,6 +270,14 @@ void OpDispatchBuilder::CalculatePF(OrderedNode *Res, OrderedNode *condition) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CalculateAF(OpSize OpSize, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) {
|
||||
// We only care about bit 4 in the subsequent XOR. If we'll XOR with 0,
|
||||
// there's no sense XOR'ing at all. This affects INC.
|
||||
uint64_t Const;
|
||||
if (IsValueConstant(WrapNode(Src2), &Const) && (Const & (1u << 4)) == 0) {
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_LOC>(Src1);
|
||||
return;
|
||||
}
|
||||
|
||||
// We store the XOR of the arguments. At read time, we XOR with the
|
||||
// appropriate bit of the result (available as the PF flag) and extract the
|
||||
// appropriate bit.
|
||||
|
||||
@@ -67,6 +67,8 @@
|
||||
"constexpr uint8_t COND_SLT = 11",
|
||||
"constexpr uint8_t COND_SGT = 12",
|
||||
"constexpr uint8_t COND_SLE = 13",
|
||||
"constexpr uint8_t COND_ANDZ = 14 /* (a & b) == 0 */",
|
||||
"constexpr uint8_t COND_ANDNZ = 15 /* (a & b) != 0 */",
|
||||
|
||||
"constexpr uint8_t COND_FLU = 16 /* float less or unordred */",
|
||||
"constexpr uint8_t COND_FGE = 17 /* float greater or equal */",
|
||||
|
||||
@@ -59,8 +59,8 @@ static void PrintArg(fextl::stringstream *out, [[maybe_unused]] IRListView const
|
||||
"SLT",
|
||||
"SGT",
|
||||
"SLE",
|
||||
"Invalid Cond",
|
||||
"Invalid Cond",
|
||||
"ANDZ",
|
||||
"ANDNZ",
|
||||
"FLU",
|
||||
"FGE",
|
||||
"FLEU",
|
||||
|
||||
@@ -210,8 +210,8 @@ class IRParser: public FEXCore::IR::IREmitter {
|
||||
"SLT",
|
||||
"SGT",
|
||||
"SLE",
|
||||
"Invalid Cond",
|
||||
"Invalid Cond",
|
||||
"ANDZ",
|
||||
"ANDNZ",
|
||||
"FLU",
|
||||
"FGE",
|
||||
"FLEU",
|
||||
|
||||
@@ -1087,9 +1087,12 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
|
||||
{
|
||||
auto Op = IROp->C<IR::IROp_Select>();
|
||||
|
||||
bool Bitwise = Op->Cond == COND_ANDZ ||
|
||||
Op->Cond == COND_ANDNZ;
|
||||
|
||||
uint64_t Constant1{};
|
||||
if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1)) {
|
||||
if (IsImmAddSub(Constant1)) {
|
||||
if (Bitwise ? IsImmLogical(Constant1, IROp->Size * 8) : IsImmAddSub(Constant1)) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1));
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -286,7 +286,7 @@
|
||||
]
|
||||
},
|
||||
"inc ax": {
|
||||
"ExpectedInstructionCount": 21,
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x40",
|
||||
"ExpectedArm64ASM": [
|
||||
@@ -294,16 +294,13 @@
|
||||
"add w21, w20, #0x1 (1)",
|
||||
"bfxil w4, w21, #0, #16",
|
||||
"uxth w21, w21",
|
||||
"eor w22, w20, #0x1",
|
||||
"strb w22, [x28, #708]",
|
||||
"strb w20, [x28, #708]",
|
||||
"strb w21, [x28, #706]",
|
||||
"ldr w22, [x28, #728]",
|
||||
"ubfx w22, w22, #29, #1",
|
||||
"lsl w23, w21, #16",
|
||||
"and w23, w23, #0x80000000",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x24, eq",
|
||||
"orr w23, w23, w24, lsl #30",
|
||||
"tst w23, w23",
|
||||
"mrs x23, nzcv",
|
||||
"eor w24, w20, #0x1",
|
||||
"eor w20, w21, w20",
|
||||
"bic w20, w20, w24",
|
||||
@@ -314,14 +311,13 @@
|
||||
]
|
||||
},
|
||||
"inc eax": {
|
||||
"ExpectedInstructionCount": 11,
|
||||
"ExpectedInstructionCount": 10,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x40",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, w4",
|
||||
"add w4, w20, #0x1 (1)",
|
||||
"eor w21, w20, #0x1",
|
||||
"strb w21, [x28, #708]",
|
||||
"strb w20, [x28, #708]",
|
||||
"strb w4, [x28, #706]",
|
||||
"ldr w21, [x28, #728]",
|
||||
"ubfx w21, w21, #29, #1",
|
||||
@@ -332,7 +328,7 @@
|
||||
]
|
||||
},
|
||||
"dec ax": {
|
||||
"ExpectedInstructionCount": 21,
|
||||
"ExpectedInstructionCount": 18,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x48",
|
||||
"ExpectedArm64ASM": [
|
||||
@@ -340,16 +336,13 @@
|
||||
"sub w21, w20, #0x1 (1)",
|
||||
"bfxil w4, w21, #0, #16",
|
||||
"uxth w21, w21",
|
||||
"eor w22, w20, #0x1",
|
||||
"strb w22, [x28, #708]",
|
||||
"strb w20, [x28, #708]",
|
||||
"strb w21, [x28, #706]",
|
||||
"ldr w22, [x28, #728]",
|
||||
"ubfx w22, w22, #29, #1",
|
||||
"lsl w23, w21, #16",
|
||||
"and w23, w23, #0x80000000",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x24, eq",
|
||||
"orr w23, w23, w24, lsl #30",
|
||||
"tst w23, w23",
|
||||
"mrs x23, nzcv",
|
||||
"eor w24, w20, #0x1",
|
||||
"eor w20, w21, w20",
|
||||
"and w20, w24, w20",
|
||||
@@ -360,14 +353,13 @@
|
||||
]
|
||||
},
|
||||
"dec eax": {
|
||||
"ExpectedInstructionCount": 11,
|
||||
"ExpectedInstructionCount": 10,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x48",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, w4",
|
||||
"sub w4, w20, #0x1 (1)",
|
||||
"eor w21, w20, #0x1",
|
||||
"strb w21, [x28, #708]",
|
||||
"strb w20, [x28, #708]",
|
||||
"strb w4, [x28, #706]",
|
||||
"ldr w21, [x28, #728]",
|
||||
"ubfx w21, w21, #29, #1",
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
Reference in new issue
Block a user