mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 17:00:19 +02:00
IR: Support <32-bit TestNZ
Originally this was going to use setf8/setf16, but it looks like the approach of shift-and-test turns out to be faster. As a bonus this is a nice delete-the-code win :-) Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
This commit is contained in:
1 parent
b1231c24ef
commit
223a6562ff
2 files changed
+12
-41
No files matched your search
@@ -106,12 +106,18 @@ DEF_OP(AddNZCV) {
|
||||
DEF_OP(TestNZ) {
|
||||
auto Op = IROp->C<IR::IROp_TestNZ>();
|
||||
const uint8_t OpSize = Op->Size;
|
||||
|
||||
LOGMAN_THROW_AA_FMT(OpSize == 4 || OpSize == 8, "Unsupported {} size: {}", __func__, OpSize);
|
||||
const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit;
|
||||
|
||||
const auto Dst = GetReg(Node);
|
||||
const auto Src = GetReg(Op->Src1.ID());
|
||||
auto Src = GetReg(Op->Src1.ID());
|
||||
|
||||
// Shift the sign bit into place, clearing out the garbage in upper bits.
|
||||
// setf+rmif would avoid the scratch register, but higher latency on M1.
|
||||
if (OpSize < 4) {
|
||||
lsl(EmitSize, Dst, Src, 32 - (OpSize * 8));
|
||||
Src = Dst;
|
||||
}
|
||||
|
||||
tst(EmitSize, Src, Src);
|
||||
|
||||
// TODO: Optimize this out
|
||||
|
||||
@@ -1137,45 +1137,10 @@ private:
|
||||
SetNZCV(_And(OpSize::i32Bit, OldNZCV, _Constant(PossiblySetNZCVBits)));
|
||||
}
|
||||
|
||||
void SetN_ZeroZCV(unsigned SrcSize, OrderedNode *Res) {
|
||||
static_assert(IndexNZCV(FEXCore::X86State::RFLAG_SF_LOC) == 31);
|
||||
|
||||
unsigned NBit = 31;
|
||||
unsigned SignBit = (SrcSize * 8) - 1;
|
||||
|
||||
OrderedNode *Shifted;
|
||||
|
||||
// Shift the sign bit into the N bit
|
||||
if (SignBit > NBit)
|
||||
Shifted = _Ashr(OpSize::i64Bit, Res, _Constant(SignBit - NBit));
|
||||
else if (SignBit < NBit)
|
||||
Shifted = _Lshl(OpSize::i32Bit, Res, _Constant(NBit - SignBit));
|
||||
else
|
||||
Shifted = Res;
|
||||
|
||||
// Mask off just the N bit, which now equals the sign bit
|
||||
CachedNZCV = _And(OpSize::i32Bit, Shifted, _Constant(1u << NBit));
|
||||
PossiblySetNZCVBits = (1u << NBit);
|
||||
NZCVDirty = true;
|
||||
}
|
||||
|
||||
void SetNZ_ZeroCV(unsigned SrcSize, OrderedNode *Res) {
|
||||
// The TestNZ opcode does this operation natively for 32-bit or 64-bit.
|
||||
// Otherwise we can implement the functionality ourselves with some bit math.
|
||||
if (SrcSize >= 4) {
|
||||
CachedNZCV = _TestNZ(SrcSize, Res);
|
||||
PossiblySetNZCVBits = (1u << 31) | (1u << 30);
|
||||
NZCVDirty = true;
|
||||
} else {
|
||||
// N
|
||||
SetN_ZeroZCV(SrcSize, Res);
|
||||
|
||||
// Z
|
||||
auto Zero = _Constant(0);
|
||||
auto One = _Constant(1);
|
||||
auto SelectOp = _Select(FEXCore::IR::COND_EQ, Res, Zero, One, Zero);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_LOC>(SelectOp);
|
||||
}
|
||||
CachedNZCV = _TestNZ(SrcSize, Res);
|
||||
PossiblySetNZCVBits = (1u << 31) | (1u << 30);
|
||||
NZCVDirty = true;
|
||||
}
|
||||
|
||||
OrderedNode *InsertNZCV(OrderedNode *NZCV, unsigned BitOffset, OrderedNode *Value) {
|
||||
|
||||
Reference in new issue
Block a user