FEXCore: Switch constant emission to default to NoPad

Most constants don't need to be padded for relocations. So now that
these have all been audited, switch to defaulting to NoPad to reduce
verbosity.

The number of constant that need to be explicitly padded are now marked
and with all the prior changes, this allows bisecting if something has
gone wrong.
This commit is contained in:
Ryan Houdek committed 2025-12-29 11:45:51 -08:00
1 parent fd2ee4e990
commit 217bbf423b
22 files changed
+264 -276

No files matched your search

+6 -6
View File
@@ -11,7 +11,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
Ref Tmp = A.Base;
if (A.Offset) {
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Offset) : IREmit->Constant(A.Offset, IR::ConstPad::NoPad);
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Offset) : IREmit->Constant(A.Offset);
}
if (A.Index) {
@@ -21,7 +21,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
if (Tmp) {
Tmp = IREmit->_AddShift(GPRSize, Tmp, A.Index, ShiftType::LSL, Log2);
} else {
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->Constant(Log2, IR::ConstPad::NoPad));
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->Constant(Log2));
}
} else {
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Index) : A.Index;
@@ -40,7 +40,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
} else if (A.Offset) {
uint64_t X = A.Offset;
X &= (1ull << Bits) - 1;
Tmp = IREmit->Constant(X, IR::ConstPad::NoPad);
Tmp = IREmit->Constant(X);
}
}
@@ -48,7 +48,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Segment) : A.Segment;
}
return Tmp ?: IREmit->Constant(0, IR::ConstPad::NoPad);
return Tmp ?: IREmit->Constant(0);
}
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
@@ -102,7 +102,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
return {
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
.Index = IREmit->Constant(A.Offset, ConstPad::NoPad),
.Index = IREmit->Constant(A.Offset),
.IndexType = MemOffsetType::SXTX,
.IndexScale = 1,
};
@@ -135,7 +135,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
return {
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
.Index = IREmit->Constant(A.Offset, ConstPad::NoPad),
.Index = IREmit->Constant(A.Offset),
.IndexType = MemOffsetType::SXTX,
.IndexScale = 1,
};
@@ -117,7 +117,7 @@ public:
// Choose to pad or not depending on if code-caching is enabled.
AUTOPAD,
};
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, PadType Pad, int MaxBytes = 0);
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, PadType Pad = PadType::NOPAD, int MaxBytes = 0);
protected:
FEXCore::Context::ContextImpl* EmitterCTX;
+3 -5
View File
@@ -970,15 +970,13 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
// Thunk entry-points don't get cached, don't need to be padded.
if (GPRSize == IR::OpSize::i64Bit) {
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint, IR::ConstPad::NoPad), GPRSize);
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint), GPRSize);
R->Reg = IR::PhysicalRegister(IR::RegClass::GPRFixed, X86State::REG_R11).Raw;
} else {
emit->_StoreContextFPR(GPRSize,
emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint, IR::ConstPad::NoPad)),
emit->_StoreContextFPR(GPRSize, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
offsetof(Core::CPUState, mm[0][0]));
}
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint, IR::ConstPad::NoPad), IR::BranchHint::None,
emit->Invalid(), emit->Invalid());
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint), IR::BranchHint::None, emit->Invalid(), emit->Invalid());
},
ThunkHandler, (void*)GuestThunkEntrypoint);
@@ -188,7 +188,7 @@ void Dispatcher::EmitDispatcher() {
if (std::popcount(VirtualMemorySize) == 1) {
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
}
@@ -261,7 +261,7 @@ void Dispatcher::EmitDispatcher() {
#ifdef _M_ARM_64EC
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
#endif
@@ -429,7 +429,7 @@ void Dispatcher::EmitDispatcher() {
PopCalleeSavedRegisters();
ret();
} else {
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
}
}
@@ -488,7 +488,7 @@ void Dispatcher::EmitDispatcher() {
// Now push the callback return trampoline to the guest stack
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->SignalDelegation->GetThunkCallbackRET(), CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->SignalDelegation->GetThunkCallbackRET());
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, CTX->Config.Is64BitMode ? 16 : 12);
+1 -1
View File
@@ -1297,7 +1297,7 @@ DEF_OP(MaskGenerateFromBitWidth) {
auto Op = IROp->C<IR::IROp_MaskGenerateFromBitWidth>();
auto BitWidth = GetReg(Op->BitWidth);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, -1, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, -1);
cmp(ARMEmitter::Size::i64Bit, BitWidth, 0);
lslv(ARMEmitter::Size::i64Bit, TMP2, TMP1, BitWidth);
csinv(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1, TMP2, ARMEmitter::Condition::CC_EQ);
@@ -271,7 +271,7 @@ DEF_OP(Syscall) {
// Still without overwriting registers that matter
// 16bit LoadConstant to be a single instruction
// This gives the signal handler a value to check to see if we are in a syscall at all
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GPRSpillMask & 0xFFFF, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GPRSpillMask & 0xFFFF);
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
@@ -362,29 +362,29 @@ DEF_OP(ValidateCode) {
EmitCheck(8, [&]() {
ldr(TMP1, Base, Offset);
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, *(const uint64_t*)(OldCode + Offset), CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, *(const uint64_t*)(OldCode + Offset));
});
EmitCheck(4, [&]() {
ldr(TMP1.W(), Base, Offset);
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint32_t*)(OldCode + Offset), CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint32_t*)(OldCode + Offset));
});
EmitCheck(2, [&]() {
ldrh(TMP1.W(), Base, Offset);
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint16_t*)(OldCode + Offset), CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint16_t*)(OldCode + Offset));
});
EmitCheck(1, [&]() {
ldrb(TMP1.W(), Base, Offset);
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint8_t*)(OldCode + Offset), CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint8_t*)(OldCode + Offset));
});
ARMEmitter::ForwardLabel End;
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0);
b_OrRestart(&End);
BindOrRestart(&Fail);
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1);
BindOrRestart(&End);
}
@@ -135,7 +135,7 @@ DEF_OP(VAESKeyGenAssist) {
if (Op->RCON) {
tbl(Dst.Q(), Dst.Q(), Swizzle.Q());
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, static_cast<uint64_t>(Op->RCON) << 32, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, static_cast<uint64_t>(Op->RCON) << 32);
dup(ARMEmitter::SubRegSize::i64Bit, VTMP2.Q(), TMP1);
eor(Dst.Q(), Dst.Q(), VTMP2.Q());
} else {
+4 -4
View File
@@ -758,14 +758,14 @@ void Arm64JITCore::EmitTFCheck() {
uint64_t Constant {};
memcpy(&Constant, &State, sizeof(State));
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
br(TMP1);
(void)Bind(&l_TFBlocked);
// If TF was blocked for this instruction, unblock it for the next.
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
(void)Bind(&l_TFUnset);
}
@@ -805,7 +805,7 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
if (ARMEmitter::IsImmAddSub(TotalSpillSlotsSize)) {
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
}
}
@@ -1163,7 +1163,7 @@ void Arm64JITCore::ResetStack() {
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
} else {
// Too big to fit in a 12bit immediate
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
}
}
+17 -17
View File
@@ -378,7 +378,7 @@ DEF_OP(SpillRegister) {
switch (OpSize) {
case IR::OpSize::i8Bit: {
if (SlotOffset > LSByteMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
strb(Src, ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
strb(Src, ARMEmitter::Reg::rsp, SlotOffset);
@@ -387,7 +387,7 @@ DEF_OP(SpillRegister) {
}
case IR::OpSize::i16Bit: {
if (SlotOffset > LSHalfMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
strh(Src, ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
strh(Src, ARMEmitter::Reg::rsp, SlotOffset);
@@ -396,7 +396,7 @@ DEF_OP(SpillRegister) {
}
case IR::OpSize::i32Bit: {
if (SlotOffset > LSWordMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
str(Src.W(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
str(Src.W(), ARMEmitter::Reg::rsp, SlotOffset);
@@ -405,7 +405,7 @@ DEF_OP(SpillRegister) {
}
case IR::OpSize::i64Bit: {
if (SlotOffset > LSDWordMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
str(Src.X(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
str(Src.X(), ARMEmitter::Reg::rsp, SlotOffset);
@@ -420,7 +420,7 @@ DEF_OP(SpillRegister) {
switch (OpSize) {
case IR::OpSize::i32Bit: {
if (SlotOffset > LSWordMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
str(Src.S(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
str(Src.S(), ARMEmitter::Reg::rsp, SlotOffset);
@@ -429,7 +429,7 @@ DEF_OP(SpillRegister) {
}
case IR::OpSize::i64Bit: {
if (SlotOffset > LSDWordMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
str(Src.D(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
str(Src.D(), ARMEmitter::Reg::rsp, SlotOffset);
@@ -438,7 +438,7 @@ DEF_OP(SpillRegister) {
}
case IR::OpSize::i128Bit: {
if (SlotOffset > LSQWordMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
str(Src.Q(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
str(Src.Q(), ARMEmitter::Reg::rsp, SlotOffset);
@@ -467,7 +467,7 @@ DEF_OP(FillRegister) {
switch (OpSize) {
case IR::OpSize::i8Bit: {
if (SlotOffset > LSByteMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
ldrb(Dst, ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
ldrb(Dst, ARMEmitter::Reg::rsp, SlotOffset);
@@ -476,7 +476,7 @@ DEF_OP(FillRegister) {
}
case IR::OpSize::i16Bit: {
if (SlotOffset > LSHalfMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
ldrh(Dst, ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
ldrh(Dst, ARMEmitter::Reg::rsp, SlotOffset);
@@ -485,7 +485,7 @@ DEF_OP(FillRegister) {
}
case IR::OpSize::i32Bit: {
if (SlotOffset > LSWordMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
ldr(Dst.W(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
ldr(Dst.W(), ARMEmitter::Reg::rsp, SlotOffset);
@@ -494,7 +494,7 @@ DEF_OP(FillRegister) {
}
case IR::OpSize::i64Bit: {
if (SlotOffset > LSDWordMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
ldr(Dst.X(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
ldr(Dst.X(), ARMEmitter::Reg::rsp, SlotOffset);
@@ -509,7 +509,7 @@ DEF_OP(FillRegister) {
switch (OpSize) {
case IR::OpSize::i32Bit: {
if (SlotOffset > LSWordMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
ldr(Dst.S(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
ldr(Dst.S(), ARMEmitter::Reg::rsp, SlotOffset);
@@ -518,7 +518,7 @@ DEF_OP(FillRegister) {
}
case IR::OpSize::i64Bit: {
if (SlotOffset > LSDWordMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
ldr(Dst.D(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
ldr(Dst.D(), ARMEmitter::Reg::rsp, SlotOffset);
@@ -527,7 +527,7 @@ DEF_OP(FillRegister) {
}
case IR::OpSize::i128Bit: {
if (SlotOffset > LSQWordMaxUnsignedOffset) {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
ldr(Dst.Q(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
} else {
ldr(Dst.Q(), ARMEmitter::Reg::rsp, SlotOffset);
@@ -609,7 +609,7 @@ ARMEmitter::Register Arm64JITCore::ApplyMemOperand(IR::OpSize AccessSize, ARMEmi
if (Const == 0) {
return Base;
}
LoadConstant(ARMEmitter::Size::i64Bit, Tmp, Const, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, Tmp, Const);
add(ARMEmitter::Size::i64Bit, Tmp, Base, Tmp, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
} else {
auto RegOffset = GetReg(Offset);
@@ -1213,7 +1213,7 @@ DEF_OP(VLoadVectorGatherMasked) {
AddrReg = GetReg(Op->AddrBase);
} else {
///< OpcodeDispatcher didn't provide a Base address while SVE requires one.
LoadConstant(ARMEmitter::Size::i64Bit, AddrReg, 0, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, AddrReg, 0);
}
MemDst = ARMEmitter::SVEMemOperand(AddrReg.X(), VectorIndexLow.Z(), ModType, SVEScale);
}
@@ -1299,7 +1299,7 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
AddrReg = *BaseAddr;
} else {
///< OpcodeDispatcher didn't provide a Base address while SVE requires one.
LoadConstant(ARMEmitter::Size::i64Bit, AddrReg, 0, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, AddrReg, 0);
}
MemDst = ARMEmitter::SVEMemOperand(AddrReg.X(), VectorIndex.Z(), ModType, SVEScale);
}
@@ -73,7 +73,7 @@ DEF_OP(Break) {
uint64_t Constant {};
memcpy(&Constant, &State, sizeof(State));
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Constant, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Constant);
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
switch (Op->Reason.Signal) {
@@ -234,7 +234,7 @@ DEF_OP(ProcessorID) {
// 16bit LoadConstant to be a single instruction
// We must always spill at least one register (x8) so this value always has a bit set
// This gives the signal handler a value to check to see if we are in a syscall at all
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SpillMask & 0xFFFF, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SpillMask & 0xFFFF);
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
// Allocate some temporary space for storing the uint32_t CPU and Node IDs
@@ -247,7 +247,7 @@ DEF_OP(ProcessorID) {
#else
constexpr auto GetCPUSyscallNum = SYS_getcpu;
#endif
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, GetCPUSyscallNum, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, GetCPUSyscallNum);
// CPU pointer in x0
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::rsp, 0);
@@ -307,7 +307,7 @@ DEF_OP(MonoBackpatcherWrite) {
#ifdef _M_ARM_64EC
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
#endif
@@ -939,7 +939,7 @@ DEF_OP(VectorImm) {
LOGMAN_THROW_A_FMT(Op->ShiftAmount == 0, "SVE VectorImm doesn't support a shift");
if (ElementSize > IR::OpSize::i8Bit && (Op->Immediate & 0x80)) {
// SVE dup uses sign extension where VectorImm wants zext
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Immediate, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Immediate);
dup(SubRegSize, Dst.Z(), TMP1);
} else {
dup_imm(SubRegSize, Dst.Z(), static_cast<int8_t>(Op->Immediate));
@@ -947,7 +947,7 @@ DEF_OP(VectorImm) {
} else {
if (ElementSize == IR::OpSize::i64Bit) {
// movi with 64bit element size doesn't do what we want here
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, static_cast<uint64_t>(Op->Immediate) << Op->ShiftAmount, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, static_cast<uint64_t>(Op->Immediate) << Op->ShiftAmount);
dup(SubRegSize, Dst.Q(), TMP1.R());
} else {
movi(SubRegSize, Dst.Q(), Op->Immediate, Op->ShiftAmount);
@@ -2521,7 +2521,7 @@ DEF_OP(VUShl) {
movi(SubRegSize, VTMP1.Q(), MaxShift);
umin(SubRegSize, VTMP1.Q(), VTMP1.Q(), ShiftVector.Q());
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift);
dup(SubRegSize, VTMP1.Q(), TMP1.R());
// UMIN is silly on Adv.SIMD and doesn't have a variant that handles 64-bit elements
@@ -2577,7 +2577,7 @@ DEF_OP(VUShr) {
movi(SubRegSize, VTMP1.Q(), MaxShift);
umin(SubRegSize, VTMP1.Q(), VTMP1.Q(), ShiftVector.Q());
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift);
dup(SubRegSize, VTMP1.Q(), TMP1.R());
// UMIN is silly on Adv.SIMD and doesn't have a variant that handles 64-bit elements
@@ -2636,7 +2636,7 @@ DEF_OP(VSShr) {
movi(SubRegSize, VTMP1.Q(), MaxShift);
umin(SubRegSize, VTMP1.Q(), VTMP1.Q(), ShiftVector.Q());
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift, CPU::Arm64Emitter::PadType::NOPAD);
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift);
dup(SubRegSize, VTMP1.Q(), TMP1.R());
// UMIN is silly on Adv.SIMD and doesn't have a variant that handles 64-bit elements
@@ -475,8 +475,7 @@ void OpDispatchBuilder::POPSegmentOp(OpcodeArgs, uint32_t SegmentReg) {
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
// Unset the 'active' bit in the packed TF, skipping the single step exception after this instruction
SetRFLAG<FEXCore::X86State::RFLAG_TF_RAW_LOC>(
_And(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::RFLAG_TF_RAW_LOC), Constant(1, ConstPad::NoPad)));
SetRFLAG<FEXCore::X86State::RFLAG_TF_RAW_LOC>(_And(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::RFLAG_TF_RAW_LOC), Constant(1)));
_StoreContextGPR(DstSize, NewSegment, offsetof(FEXCore::Core::CPUState, ss_idx));
break;
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
@@ -1169,7 +1168,7 @@ void OpDispatchBuilder::SAHFOp(OpcodeArgs) {
Ref Src = LoadGPRRegister(X86State::REG_RAX, OpSize::i8Bit, 8);
// Clear bits that aren't supposed to be set
Src = _Andn(OpSize::i64Bit, Src, Constant(0b101000, ConstPad::NoPad));
Src = _Andn(OpSize::i64Bit, Src, Constant(0b101000));
// Set the bit that is always set here
Src = _Or(OpSize::i64Bit, Src, _InlineConstant(0b10));
@@ -1194,17 +1193,17 @@ void OpDispatchBuilder::FLAGControlOp(OpcodeArgs) {
CarryInvert();
break;
case 0xF8: // CLC
SetCFInverted(Constant(1, ConstPad::NoPad));
SetCFInverted(Constant(1));
break;
case 0xF9: // STC
SetCFInverted(Constant(0, ConstPad::NoPad));
SetCFInverted(Constant(0));
break;
case 0xFC: // CLD
// Transformed
StoreDF(Constant(1, ConstPad::NoPad));
StoreDF(Constant(1));
break;
case 0xFD: // STD
StoreDF(Constant(-1, ConstPad::NoPad));
StoreDF(Constant(-1));
break;
}
}
@@ -1300,7 +1299,7 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs, bool ToSeg) {
case FEXCore::X86State::REG_RBP: // GS
case FEXCore::X86State::REG_R13: // GS
if (Is64BitMode) {
Segment = Constant(0, ConstPad::NoPad);
Segment = Constant(0);
} else {
Segment = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, gs_idx));
}
@@ -1308,7 +1307,7 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs, bool ToSeg) {
case FEXCore::X86State::REG_RSP: // FS
case FEXCore::X86State::REG_R12: // FS
if (Is64BitMode) {
Segment = Constant(0, ConstPad::NoPad);
Segment = Constant(0);
} else {
Segment = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, fs_idx));
}
@@ -1406,7 +1405,7 @@ void OpDispatchBuilder::SHLImmediateOp(OpcodeArgs, bool SHL1Bit) {
uint64_t Shift = GetConstantShift(Op, SHL1Bit);
const auto Size = GetSrcBitSize(Op);
Ref Result = _Lshl(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, Constant(Shift, ConstPad::NoPad));
Ref Result = _Lshl(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, Constant(Shift));
CalculateFlags_ShiftLeftImmediate(OpSizeFromSrc(Op), Result, Dest, Shift);
CalculateDeferredFlags();
@@ -1427,7 +1426,7 @@ void OpDispatchBuilder::SHRImmediateOp(OpcodeArgs, bool SHR1Bit) {
auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32});
uint64_t Shift = GetConstantShift(Op, SHR1Bit);
auto ALUOp = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, Constant(Shift, ConstPad::NoPad));
auto ALUOp = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, Constant(Shift));
CalculateFlags_ShiftRightImmediate(OpSizeFromSrc(Op), ALUOp, Dest, Shift);
CalculateDeferredFlags();
@@ -1456,7 +1455,7 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) {
// a64 masks the bottom bits, so if we're using a native 32/64-bit shift, we
// can negate to do the subtract (it's congruent), which saves a constant.
auto ShiftRight = Size >= 32 ? _Neg(OpSize::i64Bit, Shift) : Sub(OpSize::i64Bit, Constant(Size, ConstPad::NoPad), Shift);
auto ShiftRight = Size >= 32 ? _Neg(OpSize::i64Bit, Shift) : Sub(OpSize::i64Bit, Constant(Size), Shift);
auto Tmp1 = _Lshl(OpSize::i64Bit, Dest, Shift);
auto Tmp2 = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Src, ShiftRight);
@@ -1472,7 +1471,7 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) {
//
// TODO: This whole function wants to be wrapped in the if. Maybe b/w pass is
// a good idea after all.
Res = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Shift, Constant(0, ConstPad::NoPad), Dest, Res);
Res = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Shift, Constant(0), Dest, Res);
HandleShift(Op, Res, Dest, ShiftType::LSL, Shift);
}
@@ -1487,11 +1486,11 @@ void OpDispatchBuilder::SHLDImmediateOp(OpcodeArgs) {
if (Shift != 0) {
Ref Res {};
if (Size < 32) {
Ref ShiftLeft = Constant(Shift, ConstPad::NoPad);
Ref ShiftLeft = Constant(Shift);
auto ShiftRight = Size - Shift;
auto Tmp1 = _Lshl(OpSize::i64Bit, Dest, ShiftLeft);
Ref Tmp2 = ShiftRight ? _Lshr(OpSize::i32Bit, Src, Constant(ShiftRight, ConstPad::NoPad)) : Src;
Ref Tmp2 = ShiftRight ? _Lshr(OpSize::i32Bit, Src, Constant(ShiftRight)) : Src;
Res = _Or(OpSize::i64Bit, Tmp1, Tmp2);
} else {
@@ -1527,7 +1526,7 @@ void OpDispatchBuilder::SHRDOp(OpcodeArgs) {
Shift = _And(OpSize::i64Bit, Shift, _InlineConstant(0x1F));
}
auto ShiftLeft = Sub(OpSize::i64Bit, Constant(Size, ConstPad::NoPad), Shift);
auto ShiftLeft = Sub(OpSize::i64Bit, Constant(Size), Shift);
auto Tmp1 = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, Shift);
auto Tmp2 = _Lshl(OpSize::i64Bit, Src, ShiftLeft);
@@ -1537,7 +1536,7 @@ void OpDispatchBuilder::SHRDOp(OpcodeArgs) {
// If shift count was zero then output doesn't change
// Needs to be checked for the 32bit operand case
// where shift = 0 and the source register still gets Zext
Res = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Shift, Constant(0, ConstPad::NoPad), Dest, Res);
Res = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Shift, Constant(0), Dest, Res);
HandleShift(Op, Res, Dest, ShiftType::LSR, Shift);
}
@@ -1552,8 +1551,8 @@ void OpDispatchBuilder::SHRDImmediateOp(OpcodeArgs) {
if (Shift != 0) {
Ref Res {};
if (Size < 32) {
Ref ShiftRight = Constant(Shift, ConstPad::NoPad);
auto ShiftLeft = Constant(Size - Shift, ConstPad::NoPad);
Ref ShiftRight = Constant(Shift);
auto ShiftLeft = Constant(Size - Shift);
auto Tmp1 = _Lshr(OpSize::i32Bit, Dest, ShiftRight);
auto Tmp2 = _Lshl(OpSize::i64Bit, Src, ShiftLeft);
@@ -1588,7 +1587,7 @@ void OpDispatchBuilder::ASHROp(OpcodeArgs, bool Immediate, bool SHR1Bit) {
if (Immediate) {
uint64_t Shift = GetConstantShift(Op, SHR1Bit);
Ref Result = _Ashr(OpSize, Dest, Constant(Shift, ConstPad::NoPad));
Ref Result = _Ashr(OpSize, Dest, Constant(Shift));
CalculateFlags_SignShiftRightImmediate(OpSizeFromSrc(Op), Result, Dest, Shift);
CalculateDeferredFlags();
@@ -1682,21 +1681,21 @@ void OpDispatchBuilder::BEXTRBMIOp(OpcodeArgs) {
const auto Size = OpSizeFromSrc(Op);
const auto SrcSize = IR::OpSizeAsBits(Size);
const auto MaxSrcBit = SrcSize - 1;
auto MaxSrcBitOp = Constant(MaxSrcBit, ConstPad::NoPad);
auto MaxSrcBitOp = Constant(MaxSrcBit);
// Shift the operand down to the starting bit
auto Start = _Bfe(OpSizeFromSrc(Op), 8, 0, Src2);
auto Shifted = _Lshr(Size, Src1, Start);
// Shifts larger than operand size need to be set to zero.
auto SanitizedShifted = _Select(Size, Size, CondClass::ULE, Start, MaxSrcBitOp, Shifted, Constant(0, ConstPad::NoPad));
auto SanitizedShifted = _Select(Size, Size, CondClass::ULE, Start, MaxSrcBitOp, Shifted, Constant(0));
// Now handle the length specifier.
auto Length = _Bfe(Size, 8, 8, Src2);
// Now build up the mask
// (1 << Length) - 1 = ~(~0 << Length)
auto AllOnes = Constant(~0ull, ConstPad::NoPad);
auto AllOnes = Constant(~0ull);
auto InvertedMask = _Lshl(Size, AllOnes, Length);
// Now put it all together and make the result.
@@ -1811,7 +1810,7 @@ void OpDispatchBuilder::BZHI(OpcodeArgs) {
// Clear the high bits specified by the index. A64 only considers bottom bits
// of the shift, so we don't need to mask bottom 8-bits ourselves.
// Out-of-bounds results ignored after.
auto Mask = _Lshl(Size, Constant(-1, ConstPad::NoPad), Index);
auto Mask = _Lshl(Size, Constant(-1), Index);
auto MaskResult = _Andn(Size, Src, Mask);
// If the index is above OperandSize, we don't clear anything. BZHI only
@@ -1820,7 +1819,7 @@ void OpDispatchBuilder::BZHI(OpcodeArgs) {
//
// Because we're clobbering flags internally we ignore all carry invert
// shenanigans and use the raw versions here.
_TestNZ(OpSize::i64Bit, Index, Constant(0xFF & ~(OperandSize - 1), ConstPad::NoPad));
_TestNZ(OpSize::i64Bit, Index, Constant(0xFF & ~(OperandSize - 1)));
auto Result = _NZCVSelect(Size, CondClass::NEQ, Src, MaskResult);
StoreResultGPR(Op, Result);
@@ -1914,7 +1913,7 @@ void OpDispatchBuilder::ADXOp(OpcodeArgs) {
// Handles ADCX and ADOX
const bool IsADCX = Op->OP == 0x1F6;
auto Zero = Constant(0, ConstPad::NoPad);
auto Zero = Constant(0);
// Before we go trashing NZCV, save the current NZCV state.
Ref OldNZCV = GetNZCV();
@@ -2334,7 +2333,7 @@ void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) {
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
Ref Tmp = Constant(0, ConstPad::NoPad);
Ref Tmp = Constant(0);
for (size_t i = 0; i < (32 + Size + 1); i += (Size + 1)) {
// Insert incoming value
@@ -2713,9 +2712,9 @@ void OpDispatchBuilder::NOTOp(OpcodeArgs) {
Ref MaskConst {};
if (Size == OpSize::i64Bit) {
MaskConst = Constant(~0ULL, ConstPad::NoPad);
MaskConst = Constant(~0ULL);
} else {
MaskConst = Constant((1ULL << SizeBits) - 1, ConstPad::NoPad);
MaskConst = Constant((1ULL << SizeBits) - 1);
}
if (DestIsLockedMem(Op)) {
@@ -2734,7 +2733,7 @@ void OpDispatchBuilder::NOTOp(OpcodeArgs) {
auto Dest = Op->Dest;
if (Dest.Data.GPR.HighBits) {
LOGMAN_THROW_A_FMT(Size == OpSize::i8Bit, "Only 8-bit GPRs get high bits");
MaskConst = Constant(0xFF00, ConstPad::NoPad);
MaskConst = Constant(0xFF00);
Dest.Data.GPR.HighBits = false;
}
@@ -2796,8 +2795,8 @@ void OpDispatchBuilder::PopcountOp(OpcodeArgs) {
}
Ref OpDispatchBuilder::CalculateAFForDecimal(Ref A) {
auto Nibble = _And(OpSize::i64Bit, A, Constant(0xF, ConstPad::NoPad));
auto Greater = Select01(OpSize::i64Bit, CondClass::UGT, Nibble, Constant(9, ConstPad::NoPad));
auto Nibble = _And(OpSize::i64Bit, A, Constant(0xF));
auto Greater = Select01(OpSize::i64Bit, CondClass::UGT, Nibble, Constant(9));
return _Or(OpSize::i64Bit, LoadAF(), Greater);
}
@@ -2809,13 +2808,13 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
auto AF = CalculateAFForDecimal(AL);
// CF |= (AL > 0x99);
CFInv = _And(OpSize::i64Bit, CFInv, Select01(OpSize::i64Bit, CondClass::ULE, AL, Constant(0x99, ConstPad::NoPad)));
CFInv = _And(OpSize::i64Bit, CFInv, Select01(OpSize::i64Bit, CondClass::ULE, AL, Constant(0x99)));
// AL = AF ? (AL + 0x6) : AL;
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::NEQ, AF, Constant(0, ConstPad::NoPad), Add(OpSize::i64Bit, AL, 0x6), AL);
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::NEQ, AF, Constant(0), Add(OpSize::i64Bit, AL, 0x6), AL);
// AL = CF ? (AL + 0x60) : AL;
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, CFInv, Constant(0, ConstPad::NoPad), Add(OpSize::i64Bit, AL, 0x60), AL);
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, CFInv, Constant(0), Add(OpSize::i64Bit, AL, 0x60), AL);
// SF, ZF, PF set according to result. CF set per above. OF undefined.
StoreGPRRegister(X86State::REG_RAX, AL, OpSize::i8Bit);
@@ -2832,16 +2831,16 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
auto AF = CalculateAFForDecimal(AL);
// CF |= (AL > 0x99);
CF = _Or(OpSize::i64Bit, CF, Select01(OpSize::i64Bit, CondClass::UGT, AL, Constant(0x99, ConstPad::NoPad)));
CF = _Or(OpSize::i64Bit, CF, Select01(OpSize::i64Bit, CondClass::UGT, AL, Constant(0x99)));
// NewCF = CF | (AF && (Borrow from AL - 6))
auto NewCF = _Or(OpSize::i32Bit, CF, _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::ULT, AL, Constant(6, ConstPad::NoPad), AF, CF));
auto NewCF = _Or(OpSize::i32Bit, CF, _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::ULT, AL, Constant(6), AF, CF));
// AL = AF ? (AL - 0x6) : AL;
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::NEQ, AF, Constant(0, ConstPad::NoPad), Sub(OpSize::i64Bit, AL, 0x6), AL);
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::NEQ, AF, Constant(0), Sub(OpSize::i64Bit, AL, 0x6), AL);
// AL = CF ? (AL - 0x60) : AL;
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::NEQ, CF, Constant(0, ConstPad::NoPad), Sub(OpSize::i64Bit, AL, 0x60), AL);
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::NEQ, CF, Constant(0), Sub(OpSize::i64Bit, AL, 0x60), AL);
// SF, ZF, PF set according to result. CF set per above. OF undefined.
StoreGPRRegister(X86State::REG_RAX, AL, OpSize::i8Bit);
@@ -2864,7 +2863,7 @@ void OpDispatchBuilder::AAAOp(OpcodeArgs) {
A = NZCVSelect(OpSize::i32Bit, CondClass::UGE /* CF = 1 */, Add(OpSize::i32Bit, A, 0x106), A);
// AL = AL & 0x0F
A = _And(OpSize::i32Bit, A, Constant(0xFF0F, ConstPad::NoPad));
A = _And(OpSize::i32Bit, A, Constant(0xFF0F));
StoreGPRRegister(X86State::REG_RAX, A, OpSize::i16Bit);
}
@@ -2881,13 +2880,13 @@ void OpDispatchBuilder::AASOp(OpcodeArgs) {
A = NZCVSelect(OpSize::i32Bit, CondClass::UGE /* CF = 1 */, Sub(OpSize::i32Bit, A, 0x106), A);
// AL = AL & 0x0F
A = _And(OpSize::i32Bit, A, Constant(0xFF0F, ConstPad::NoPad));
A = _And(OpSize::i32Bit, A, Constant(0xFF0F));
StoreGPRRegister(X86State::REG_RAX, A, OpSize::i16Bit);
}
void OpDispatchBuilder::AAMOp(OpcodeArgs) {
auto AL = LoadGPRRegister(X86State::REG_RAX, OpSize::i8Bit);
auto Imm8 = Constant(Op->Src[0].Literal() & 0xFF, ConstPad::NoPad);
auto Imm8 = Constant(Op->Src[0].Literal() & 0xFF);
Ref Quotient = _AllocateGPR(true);
Ref Remainder = _AllocateGPR(true);
_UDiv(OpSize::i64Bit, AL, Invalid(), Imm8, Quotient, Remainder);
@@ -2901,10 +2900,10 @@ void OpDispatchBuilder::AAMOp(OpcodeArgs) {
void OpDispatchBuilder::AADOp(OpcodeArgs) {
auto A = LoadGPRRegister(X86State::REG_RAX);
auto AH = _Lshr(OpSize::i32Bit, A, Constant(8, ConstPad::NoPad));
auto Imm8 = Constant(Op->Src[0].Literal() & 0xFF, ConstPad::NoPad);
auto AH = _Lshr(OpSize::i32Bit, A, Constant(8));
auto Imm8 = Constant(Op->Src[0].Literal() & 0xFF);
auto NewAL = Add(OpSize::i64Bit, A, _Mul(OpSize::i64Bit, AH, Imm8));
auto Result = _And(OpSize::i64Bit, NewAL, Constant(0xFF, ConstPad::NoPad));
auto Result = _And(OpSize::i64Bit, NewAL, Constant(0xFF));
StoreGPRRegister(X86State::REG_RAX, Result, OpSize::i16Bit);
SetNZ_ZeroCV(OpSize::i8Bit, Result);
@@ -3001,9 +3000,8 @@ void OpDispatchBuilder::SGDTOp(OpcodeArgs) {
GDTStoreSize = OpSize::i32Bit;
}
_StoreMemGPRAutoTSO(OpSize::i16Bit, DestAddress, Constant(0, ConstPad::NoPad));
_StoreMemGPRAutoTSO(GDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit},
Constant(GDTAddress, ConstPad::NoPad));
_StoreMemGPRAutoTSO(OpSize::i16Bit, DestAddress, Constant(0));
_StoreMemGPRAutoTSO(GDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, Constant(GDTAddress));
}
void OpDispatchBuilder::SIDTOp(OpcodeArgs) {
@@ -3018,30 +3016,28 @@ void OpDispatchBuilder::SIDTOp(OpcodeArgs) {
IDTStoreSize = OpSize::i32Bit;
}
_StoreMemGPRAutoTSO(OpSize::i16Bit, DestAddress, Constant(0xfff, ConstPad::NoPad));
_StoreMemGPRAutoTSO(IDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit},
Constant(IDTAddress, ConstPad::NoPad));
_StoreMemGPRAutoTSO(OpSize::i16Bit, DestAddress, Constant(0xfff));
_StoreMemGPRAutoTSO(IDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, Constant(IDTAddress));
}
void OpDispatchBuilder::SMSWOp(OpcodeArgs) {
const bool IsMemDst = DestIsMem(Op);
IR::OpSize DstSize {OpSize::iInvalid};
Ref Const = Constant((1U << 31) | ///< PG - Paging
(0U << 30) | ///< CD - Cache Disable
(0U << 29) | ///< NW - Not Writethrough (Legacy, now ignored)
///< [28:19] - Reserved
(1U << 18) | ///< AM - Alignment Mask
///< 17 - Reserved
(1U << 16) | ///< WP - Write Protect
///< [15:6] - Reserved
(1U << 5) | ///< NE - Numeric Error
(1U << 4) | ///< ET - Extension Type (Legacy, now reserved and 1)
(0U << 3) | ///< TS - Task Switched
(0U << 2) | ///< EM - Emulation
(1U << 1) | ///< MP - Monitor Coprocessor
(1U << 0), ///< PE - Protection Enabled
ConstPad::NoPad);
Ref Const = Constant((1U << 31) | ///< PG - Paging
(0U << 30) | ///< CD - Cache Disable
(0U << 29) | ///< NW - Not Writethrough (Legacy, now ignored)
///< [28:19] - Reserved
(1U << 18) | ///< AM - Alignment Mask
///< 17 - Reserved
(1U << 16) | ///< WP - Write Protect
///< [15:6] - Reserved
(1U << 5) | ///< NE - Numeric Error
(1U << 4) | ///< ET - Extension Type (Legacy, now reserved and 1)
(0U << 3) | ///< TS - Task Switched
(0U << 2) | ///< EM - Emulation
(1U << 1) | ///< MP - Monitor Coprocessor
(1U << 0)); ///< PE - Protection Enabled
const auto OpAddr = X86Tables::DecodeFlags::GetOpAddr(Op->Flags, 0);
if (Is64BitMode) {
DstSize = OpAddr == X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST ? OpSize::i16Bit :
@@ -3072,8 +3068,8 @@ OpDispatchBuilder::CycleCounterPair OpDispatchBuilder::CycleCounter(bool SelfSyn
Ref CounterHigh {};
auto Counter = _CycleCounter(SelfSynchronizingLoads);
if (CTX->Config.TSCScale) {
CounterLow = _Lshl(OpSize::i32Bit, Counter, Constant(CTX->Config.TSCScale, ConstPad::NoPad));
CounterHigh = _Lshr(OpSize::i64Bit, Counter, Constant(32 - CTX->Config.TSCScale, ConstPad::NoPad));
CounterLow = _Lshl(OpSize::i32Bit, Counter, Constant(CTX->Config.TSCScale));
CounterHigh = _Lshr(OpSize::i64Bit, Counter, Constant(32 - CTX->Config.TSCScale));
} else {
CounterLow = _Bfe(OpSize::i64Bit, 32, 0, Counter);
CounterHigh = _Bfe(OpSize::i64Bit, 32, 32, Counter);
@@ -3101,7 +3097,7 @@ void OpDispatchBuilder::INCOp(OpcodeArgs) {
HandledLock = true;
Ref DestAddress = MakeSegmentAddress(Op, Op->Dest);
Dest = _AtomicFetchAdd(OpSizeFromSrc(Op), Constant(1, ConstPad::NoPad), DestAddress);
Dest = _AtomicFetchAdd(OpSizeFromSrc(Op), Constant(1), DestAddress);
} else {
Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32});
}
@@ -3112,7 +3108,7 @@ void OpDispatchBuilder::INCOp(OpcodeArgs) {
// Addition producing upper garbage
Result = Add(OpSize::i32Bit, Dest, 1);
CalculatePF(Result);
CalculateAF(Dest, Constant(1, ConstPad::NoPad));
CalculateAF(Dest, Constant(1));
// Correctly set NZ flags, preserving C
HandleNZCV_RMW();
@@ -3122,7 +3118,7 @@ void OpDispatchBuilder::INCOp(OpcodeArgs) {
// getting a negative. So compare the sign bits to calculate V.
_RmifNZCV(_Andn(OpSize::i32Bit, Result, Dest), Size - 1, 1);
} else {
Result = CalculateFlags_ADD(OpSizeFromSrc(Op), Dest, Constant(1, ConstPad::NoPad), false);
Result = CalculateFlags_ADD(OpSizeFromSrc(Op), Dest, Constant(1), false);
}
if (!IsLocked) {
@@ -3142,7 +3138,7 @@ void OpDispatchBuilder::DECOp(OpcodeArgs) {
Ref DestAddress = MakeSegmentAddress(Op, Op->Dest);
// Use Add instead of Sub to avoid a NEG
Dest = _AtomicFetchAdd(OpSizeFromSrc(Op), Constant(Size == 64 ? -1 : ((1ULL << Size) - 1), ConstPad::NoPad), DestAddress);
Dest = _AtomicFetchAdd(OpSizeFromSrc(Op), Constant(Size == 64 ? -1 : ((1ULL << Size) - 1)), DestAddress);
} else {
Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32});
}
@@ -3153,7 +3149,7 @@ void OpDispatchBuilder::DECOp(OpcodeArgs) {
// Subtraction producing upper garbage
Result = Sub(OpSize::i32Bit, Dest, 1);
CalculatePF(Result);
CalculateAF(Dest, Constant(1, ConstPad::NoPad));
CalculateAF(Dest, Constant(1));
// Correctly set NZ flags, preserving C
HandleNZCV_RMW();
@@ -3163,7 +3159,7 @@ void OpDispatchBuilder::DECOp(OpcodeArgs) {
// getting a positive. So compare the sign bits to calculate V.
_RmifNZCV(_Andn(OpSize::i32Bit, Dest, Result), Size - 1, 1);
} else {
Result = CalculateFlags_SUB(OpSizeFromSrc(Op), Dest, Constant(1, ConstPad::NoPad), false);
Result = CalculateFlags_SUB(OpSizeFromSrc(Op), Dest, Constant(1), false);
}
if (!IsLocked) {
@@ -3211,7 +3207,7 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) {
Ref Counter = LoadGPRRegister(X86State::REG_RCX);
auto Result = _MemSet(CTX->IsAtomicTSOEnabled(), Size, Segment ?: InvalidNode, Dest, Src, Counter, LoadDir(1));
StoreGPRRegister(X86State::REG_RCX, Constant(0, ConstPad::NoPad));
StoreGPRRegister(X86State::REG_RCX, Constant(0));
StoreGPRRegister(X86State::REG_RDI, Result);
}
}
@@ -3254,7 +3250,7 @@ void OpDispatchBuilder::MOVSOp(OpcodeArgs) {
Result_Src = Sub(OpSize::i64Bit, Result_Src, SrcSegment);
}
StoreGPRRegister(X86State::REG_RCX, Constant(0, ConstPad::NoPad));
StoreGPRRegister(X86State::REG_RCX, Constant(0));
StoreGPRRegister(X86State::REG_RDI, Result_Dst);
StoreGPRRegister(X86State::REG_RSI, Result_Src);
} else {
@@ -3610,7 +3606,7 @@ void OpDispatchBuilder::BSWAPOp(OpcodeArgs) {
const auto Size = OpSizeFromSrc(Op);
if (Size == OpSize::i16Bit) {
// BSWAP of 16bit is undef. ZEN+ causes the lower 16bits to get zero'd
Dest = Constant(0, ConstPad::NoPad);
Dest = Constant(0);
} else {
Dest = LoadSourceGPR_WithOpSize(Op, Op->Dest, GetGPROpSize(), Op->Flags);
Dest = _Rev(Size, Dest);
@@ -3632,7 +3628,7 @@ void OpDispatchBuilder::POPFOp(OpcodeArgs) {
// Bit 1 is always 1
// Bit 9 is always 1 because we always have interrupts enabled
Src = _Or(OpSize::i64Bit, Src, Constant(0x202, ConstPad::NoPad));
Src = _Or(OpSize::i64Bit, Src, Constant(0x202));
SetPackedRFLAG(false, Src);
@@ -3645,7 +3641,7 @@ void OpDispatchBuilder::NEGOp(OpcodeArgs) {
HandledLock = (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK) != 0;
const auto Size = OpSizeFromSrc(Op);
auto ZeroConst = Constant(0, ConstPad::NoPad);
auto ZeroConst = Constant(0);
if (DestIsLockedMem(Op)) {
Ref DestMem = MakeSegmentAddress(Op, Op->Dest);
@@ -4151,14 +4147,14 @@ void OpDispatchBuilder::UpdatePrefixFromSegment(Ref Segment, uint32_t SegmentReg
// In some cases the upper 16-bits of the 32-bit GPR contain garbage to ignore.
auto GDT = _Bfe(OpSize::i32Bit, 1, 2, Segment);
// Fun quirk, if we mask the selector then it is premultiplied by 8 which we need to do for accessing anyway.
auto SegmentOffset = _And(OpSize::i32Bit, Segment, _Constant(0xfff8, ConstPad::NoPad));
auto SegmentOffset = _And(OpSize::i32Bit, Segment, _Constant(0xfff8));
Ref SegmentBase = _LoadContextGPRIndexed(GDT, OpSize::i64Bit, offsetof(FEXCore::Core::CPUState, segment_arrays[0]), 8);
Ref NewSegment = _LoadMemGPR(OpSize::i64Bit, SegmentBase, SegmentOffset, OpSize::i8Bit, MemOffsetType::UXTW, 1);
CheckLegacySegmentWrite(NewSegment, SegmentReg);
// Extract the 32-bit base from the GDT segment.
auto Upper32 = _Lshr(OpSize::i64Bit, NewSegment, _Constant(32, ConstPad::NoPad));
auto Masked = _And(OpSize::i32Bit, Upper32, _Constant(0xFF00'0000, ConstPad::NoPad));
auto Upper32 = _Lshr(OpSize::i64Bit, NewSegment, _Constant(32));
auto Masked = _And(OpSize::i32Bit, Upper32, _Constant(0xFF00'0000));
Ref Merged = _Orlshr(OpSize::i32Bit, Masked, NewSegment, 16);
NewSegment = _Bfi(OpSize::i32Bit, 8, 16, Merged, Upper32);
@@ -4333,7 +4329,7 @@ Ref OpDispatchBuilder::LoadGPRRegister(uint32_t GPR, IR::OpSize Size, uint8_t Of
// Extract the subregister if requested.
const auto OpSize = std::max(OpSize::i32Bit, Size);
if (AllowUpperGarbage) {
Reg = _Lshr(OpSize, Reg, Constant(Offset, ConstPad::NoPad));
Reg = _Lshr(OpSize, Reg, Constant(Offset));
} else {
Reg = _Bfe(OpSize, IR::OpSizeAsBits(Size), Offset, Reg);
}
@@ -4440,7 +4436,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(RegClass Class, FEXCore::X86Table
// For X87 extended doubles, split before storing
_StoreMemFPR(OpSize::i64Bit, MemStoreDst, Src, Align);
auto Upper = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, Src, 1);
_StoreMemGPR(OpSize::i16Bit, Upper, MemStoreDst, Constant(8, ConstPad::NoPad), std::min(Align, OpSize::i64Bit), MemOffsetType::SXTX, 1);
_StoreMemGPR(OpSize::i16Bit, Upper, MemStoreDst, Constant(8), std::min(Align, OpSize::i64Bit), MemOffsetType::SXTX, 1);
}
} else {
_StoreMemAutoTSO(Class, OpSize, A, Src, Align == OpSize::iInvalid ? OpSize : Align);
@@ -4516,7 +4512,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
FlushRegisterCache();
// Move 0 into the register
StoreResultGPR(Op, Constant(0, ConstPad::NoPad));
StoreResultGPR(Op, Constant(0));
return;
}
@@ -4551,7 +4547,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
// adjusted constant here will inline into the arm64 and instruction, so if
// flags are not needed, we save an instruction overall.
if (ALUIROp == IR::IROps::OP_ANDWITHFLAGS) {
Src = Constant(Const | ~((1ull << IR::OpSizeAsBits(Size)) - 1), ConstPad::NoPad);
Src = Constant(Const | ~((1ull << IR::OpSizeAsBits(Size)) - 1));
ALUIROp = IR::IROps::OP_AND;
}
}
@@ -4606,7 +4602,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
void OpDispatchBuilder::LSLOp(OpcodeArgs) {
// Emulate by always returning failure, this deviates from both Linux and Windows but
// shouldn't be depended on by anything.
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(Constant(0, ConstPad::NoPad));
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(Constant(0));
}
void OpDispatchBuilder::INTOp(OpcodeArgs) {
@@ -244,7 +244,7 @@ public:
template<typename F>
void ForeachDirection(F&& Routine) {
// Otherwise, prepare to branch.
auto Zero = Constant(0, ConstPad::NoPad);
auto Zero = Constant(0);
// If the shift is zero, do not touch the flags.
auto ForwardBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
@@ -1172,7 +1172,7 @@ public:
auto Const = Header->C<IR::IROp_LoadNamedVectorConstant>();
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
Ref Zero = _Constant(0, ConstPad::NoPad);
Ref Zero = _Constant(0);
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, RegClass::GPR, Zero, Zero, Offset);
// XXX: This works around InlineConstant not having an associated
@@ -1263,7 +1263,7 @@ public:
StoreContextHelper(Size, Class, Value, Offset);
// If Partial and MMX register, then we need to store all 1s in bits 64-80
if (Partial && Index >= MM0Index && Index <= MM7Index) {
_StoreContextGPR(OpSize::i16Bit, Constant(0xFFFF, ConstPad::NoPad), Offset + 8);
_StoreContextGPR(OpSize::i16Bit, Constant(0xFFFF), Offset + 8);
}
}
}
@@ -1697,7 +1697,7 @@ private:
}
void ZeroNZCV() {
CachedNZCV = Constant(0, ConstPad::NoPad);
CachedNZCV = Constant(0);
NZCVDirty = true;
}
@@ -1712,7 +1712,7 @@ private:
if (SetPF) {
CalculatePF(SubWithFlags(SrcSize, Res, (uint64_t)0));
} else {
_SubNZCV(SrcSize, Res, Constant(0, ConstPad::NoPad));
_SubNZCV(SrcSize, Res, Constant(0));
}
CFInverted = true;
@@ -1783,7 +1783,7 @@ private:
} else {
// Invert as a GPR
unsigned Bit = IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), Constant(1u << Bit, ConstPad::NoPad)));
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), Constant(1u << Bit)));
CalculateDeferredFlags();
}
@@ -1821,7 +1821,7 @@ private:
}
HandleNZCVWrite();
_SubNZCV(OpSize::i32Bit, Constant(0, ConstPad::NoPad), Value);
_SubNZCV(OpSize::i32Bit, Constant(0), Value);
CFInverted = true;
}
@@ -1846,14 +1846,14 @@ private:
StoreRegister(Core::CPUState::AF_AS_GREG, false, Value);
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
// For DF, we need to transform 0/1 into 1/-1
StoreDF(_SubShift(OpSize::i64Bit, Constant(1, ConstPad::NoPad), Value, ShiftType::LSL, 1));
StoreDF(_SubShift(OpSize::i64Bit, Constant(1), Value, ShiftType::LSL, 1));
} else if (BitOffset == FEXCore::X86State::RFLAG_TF_RAW_LOC) {
auto PackedTF = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
// An exception should still be raised after an instruction that unsets TF, leave the unblocked bit set but unset
// the TF bit to cause such behaviour. The handling code at the start of the next block will then unset the
// unblocked bit before raising the exception.
auto NewPackedTF = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Value, Constant(0, ConstPad::NoPad),
_And(OpSize::i32Bit, PackedTF, Constant(~1, ConstPad::NoPad)), Constant(1, ConstPad::NoPad));
auto NewPackedTF =
_Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Value, Constant(0), _And(OpSize::i32Bit, PackedTF, Constant(~1)), Constant(1));
_StoreContextGPR(OpSize::i8Bit, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
} else {
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
@@ -1865,7 +1865,7 @@ private:
// bits. This allows us to defer the extract in the usual case. When it is
// read, bit 4 is extracted. In order to write a constant value of AF, that
// means we need to left-shift here to compensate.
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(K << 4, ConstPad::NoPad));
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(K << 4));
}
void ZeroPF_AF();
@@ -2089,7 +2089,7 @@ private:
auto Value = _Bfe(OpSize::i32Bit, 1, IndexNZCV(BitOffset), GetNZCV());
if (Invert) {
return _Xor(OpSize::i32Bit, Value, Constant(1, ConstPad::NoPad));
return _Xor(OpSize::i32Bit, Value, Constant(1));
} else {
return Value;
}
@@ -2104,7 +2104,7 @@ private:
return LoadGPR(Core::CPUState::AF_AS_GREG);
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
// Recover the sign bit, it is the logical DF value
return _Lshr(OpSize::i64Bit, LoadDF(), Constant(63, ConstPad::NoPad));
return _Lshr(OpSize::i64Bit, LoadDF(), Constant(63));
} else {
return _LoadContextGPR(OpSize::i8Bit, offsetof(Core::CPUState, flags[BitOffset]));
}
@@ -2176,7 +2176,7 @@ private:
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so
// PF[4] is 0 so the XOR with PF will have no effect, so setting the AF
// byte to zero will indeed zero AF as intended.
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0, ConstPad::NoPad));
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0));
}
// Convert NZCV from the Arm representation to an eXternal representation
@@ -2210,7 +2210,7 @@ private:
}
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(C);
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0, ConstPad::NoPad));
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(V);
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(Z);
}
@@ -2332,7 +2332,7 @@ private:
}
// Otherwise, prepare to branch.
auto Zero = Constant(0, ConstPad::NoPad);
auto Zero = Constant(0);
// If the shift is zero, do not touch the flags.
auto SetBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
@@ -2405,8 +2405,8 @@ private:
void ChgStateX87_MMX() override {
LOGMAN_THROW_A_FMT(MMXState == MMXState_X87, "Expected state to be x87");
_StackForceSlow();
SetX87Top(Constant(0, ConstPad::NoPad)); // top reset to zero
_StoreContextGPR(OpSize::i8Bit, Constant(0xFFFFUL, ConstPad::NoPad), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
SetX87Top(Constant(0)); // top reset to zero
_StoreContextGPR(OpSize::i8Bit, Constant(0xFFFFUL), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
MMXState = MMXState_MMX;
}
@@ -2639,11 +2639,11 @@ private:
}
ArithRef And(uint64_t K) {
return IsConstant ? ArithRef(E, C & K) : ArithRef(E, E->_And(OpSize::i64Bit, R, E->Constant(K, ConstPad::NoPad)));
return IsConstant ? ArithRef(E, C & K) : ArithRef(E, E->_And(OpSize::i64Bit, R, E->Constant(K)));
}
ArithRef Presub(uint64_t K) {
return IsConstant ? ArithRef(E, K - C) : ArithRef(E, E->Sub(OpSize::i64Bit, E->Constant(K, ConstPad::NoPad), R));
return IsConstant ? ArithRef(E, K - C) : ArithRef(E, E->Sub(OpSize::i64Bit, E->Constant(K), R));
}
ArithRef Lshl(uint64_t Shift) {
@@ -2652,7 +2652,7 @@ private:
} else if (IsConstant) {
return ArithRef(E, C << Shift);
} else {
return ArithRef(E, E->_Lshl(OpSize::i64Bit, R, E->Constant(Shift, ConstPad::NoPad)));
return ArithRef(E, E->_Lshl(OpSize::i64Bit, R, E->Constant(Shift)));
}
}
@@ -2692,7 +2692,7 @@ private:
}
if (IsConstant) {
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, E->Constant(C, ConstPad::NoPad));
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, E->Constant(C));
} else {
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, R);
}
@@ -2708,12 +2708,12 @@ private:
return ArithRef(E, Result);
} else {
return ArithRef(E, E->_Lshl(Size, E->Constant(1, ConstPad::NoPad), R));
return ArithRef(E, E->_Lshl(Size, E->Constant(1), R));
}
}
Ref Ref() {
return IsConstant ? E->Constant(C, ConstPad::NoPad) : R;
return IsConstant ? E->Constant(C) : R;
}
bool IsDefinitelyZero() const {
@@ -845,7 +845,7 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize) {
// Inserting the full lower 32-bits offset 31 so the sign bit ends up at offset 63.
GPR = _Bfi(OpSize::i64Bit, 32, 31, GPR, GPR);
// Shift right to only get the two sign bits we care about.
return _Lshr(OpSize::i64Bit, GPR, Constant(62, ConstPad::NoPad));
return _Lshr(OpSize::i64Bit, GPR, Constant(62));
};
auto Mask4Byte = [this](Ref Src) {
@@ -1838,7 +1838,7 @@ void OpDispatchBuilder::AVX128_VPERMD(OpcodeArgs) {
RefPair Result {};
Ref IndexMask = _VectorImm(OpSize::i128Bit, OpSize::i32Bit, 0b111);
Ref AddConst = Constant(0x03020100, ConstPad::NoPad);
Ref AddConst = Constant(0x03020100);
Ref Repeating3210 = _VDupFromGPR(OpSize::i128Bit, OpSize::i32Bit, AddConst);
Result.Low = DoPerm(Src, Indices.Low, IndexMask, Repeating3210);
@@ -2035,7 +2035,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpcodeArgs, Op
if (BaseAddr && VSIB.Displacement) {
BaseAddr = Add(OpSize::i64Bit, BaseAddr, VSIB.Displacement);
} else if (VSIB.Displacement) {
BaseAddr = Constant(VSIB.Displacement, ConstPad::NoPad);
BaseAddr = Constant(VSIB.Displacement);
} else if (!BaseAddr) {
BaseAddr = Invalid();
}
@@ -2133,7 +2133,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(OpcodeArgs,
if (BaseAddr && VSIB.Displacement) {
BaseAddr = Add(OpSize::i64Bit, BaseAddr, VSIB.Displacement);
} else if (VSIB.Displacement) {
BaseAddr = Constant(VSIB.Displacement, ConstPad::NoPad);
BaseAddr = Constant(VSIB.Displacement);
} else if (!BaseAddr) {
BaseAddr = Invalid();
}
@@ -28,7 +28,7 @@ constexpr std::array<uint32_t, 17> FlagOffsets = {
void OpDispatchBuilder::ZeroPF_AF() {
// PF is stored inverted, so invert it when we zero.
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(Constant(1, ConstPad::NoPad));
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(Constant(1));
SetAF(0);
}
@@ -247,7 +247,7 @@ void OpDispatchBuilder::CalculateAF(Ref Src1, Ref Src2) {
// We store the XOR of the arguments. At read time, we XOR with the
// appropriate bit of the result (available as the PF flag) and extract the
// appropriate bit. Again 64-bit to avoid masking.
Ref XorRes = Src1 == Src2 ? Constant(0, ConstPad::NoPad) : _Xor(OpSize::i64Bit, Src1, Src2);
Ref XorRes = Src1 == Src2 ? Constant(0) : _Xor(OpSize::i64Bit, Src1, Src2);
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
}
@@ -740,7 +740,7 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs, IR::OpSize ElementSize) {
// Inserting the full lower 32-bits offset 31 so the sign bit ends up at offset 63.
GPR = _Bfi(OpSize::i64Bit, 32, 31, GPR, GPR);
// Shift right to only get the two sign bits we care about.
GPR = _Lshr(OpSize::i64Bit, GPR, Constant(62, ConstPad::NoPad));
GPR = _Lshr(OpSize::i64Bit, GPR, Constant(62));
StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize());
} else if (Size == OpSize::i128Bit && ElementSize == OpSize::i32Bit) {
// Shift all the sign bits to the bottom of their respective elements.
@@ -755,7 +755,7 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs, IR::OpSize ElementSize) {
Ref GPR = _VExtractToGPR(Size, OpSize::i32Bit, Src, 0);
StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize());
} else {
Ref CurrentVal = Constant(0, ConstPad::NoPad);
Ref CurrentVal = Constant(0);
for (unsigned i = 0; i < NumElements; ++i) {
// Extract the top bit of the element
@@ -2121,7 +2121,7 @@ Ref OpDispatchBuilder::CVTFPR_To_GPRImpl(OpcodeArgs, Ref Src, IR::OpSize SrcElem
Ref Converted = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
bool Dst32 = GPRSize == OpSize::i32Bit;
Ref MaxI = Dst32 ? Constant(0x80000000, ConstPad::NoPad) : Constant(0x8000000000000000, ConstPad::NoPad);
Ref MaxI = Dst32 ? Constant(0x80000000) : Constant(0x8000000000000000);
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcElementSize, (SrcElementSize == OpSize::i32Bit) ?
(Dst32 ? NAMED_VECTOR_CVTMAX_F32_I32 : NAMED_VECTOR_CVTMAX_F32_I64) :
(Dst32 ? NAMED_VECTOR_CVTMAX_F64_I32 : NAMED_VECTOR_CVTMAX_F64_I64));
@@ -2552,7 +2552,7 @@ void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
// XSTATE_BV section of the header is 8 bytes in size, but we only really
// care about setting at most 3 bits in the first byte. We zero out the rest.
_StoreMemGPR(OpSize::i64Bit, RequestedFeatures, Base, Constant(512, ConstPad::NoPad), OpSize::i8Bit, MemOffsetType::SXTX, 1);
_StoreMemGPR(OpSize::i64Bit, RequestedFeatures, Base, Constant(512), OpSize::i8Bit, MemOffsetType::SXTX, 1);
}
}
@@ -2578,12 +2578,12 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
_StoreMemGPR(OpSize::i16Bit, MemBase, FCW, OpSize::i16Bit);
}
{ _StoreMemGPR(OpSize::i16Bit, ReconstructFSW_Helper(), MemBase, Constant(2, ConstPad::NoPad), OpSize::i16Bit, MemOffsetType::SXTX, 1); }
{ _StoreMemGPR(OpSize::i16Bit, ReconstructFSW_Helper(), MemBase, Constant(2), OpSize::i16Bit, MemOffsetType::SXTX, 1); }
{
// Abridged FTW
auto FTW = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
_StoreMemGPR(OpSize::i8Bit, FTW, MemBase, Constant(4, ConstPad::NoPad), OpSize::i8Bit, MemOffsetType::SXTX, 1);
_StoreMemGPR(OpSize::i8Bit, FTW, MemBase, Constant(4), OpSize::i8Bit, MemOffsetType::SXTX, 1);
}
// BYTE | 0 1 | 2 3 | 4 | 5 | 6 7 | 8 9 | a b | c d | e f |
@@ -2633,7 +2633,7 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
//
// x87 registers are stored rotated depending on the current TOP.
Ref Top = GetX87Top();
auto SevenConst = Constant(7, ConstPad::NoPad);
auto SevenConst = Constant(7);
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
@@ -2641,7 +2641,7 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
if (ReducedPrecisionMode) {
data = _F80CVTTo(data, OpSize::i64Bit);
}
_StoreMemFPR(OpSize::i128Bit, data, MemBase, Constant(16 * i + 32, ConstPad::NoPad), OpSize::i8Bit, MemOffsetType::SXTX, 1);
_StoreMemFPR(OpSize::i128Bit, data, MemBase, Constant(16 * i + 32), OpSize::i8Bit, MemOffsetType::SXTX, 1);
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
}
}
@@ -2656,7 +2656,7 @@ void OpDispatchBuilder::SaveSSEState(Ref MemBase) {
void OpDispatchBuilder::SaveMXCSRState(Ref MemBase) {
// Store MXCSR and the mask for all bits.
_StoreMemPairGPR(OpSize::i32Bit, GetMXCSR(), Constant(0xFFFF, ConstPad::NoPad), MemBase, 24);
_StoreMemPairGPR(OpSize::i32Bit, GetMXCSR(), Constant(0xFFFF), MemBase, 24);
}
void OpDispatchBuilder::SaveAVXState(Ref MemBase) {
@@ -2674,7 +2674,7 @@ Ref OpDispatchBuilder::GetMXCSR() {
Ref MXCSR = _LoadContextGPR(OpSize::i32Bit, offsetof(FEXCore::Core::CPUState, mxcsr));
// Mask out unsupported bits
// Keeps FZ, RC, exception masks, and DAZ
MXCSR = _And(OpSize::i32Bit, MXCSR, Constant(0xFFC0, ConstPad::NoPad));
MXCSR = _And(OpSize::i32Bit, MXCSR, Constant(0xFFC0));
return MXCSR;
}
@@ -2684,7 +2684,7 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
RestoreX87State(Mem);
RestoreSSEState(Mem);
Ref MXCSR = _LoadMemGPR(OpSize::i32Bit, Mem, Constant(24, ConstPad::NoPad), OpSize::i32Bit, MemOffsetType::SXTX, 1);
Ref MXCSR = _LoadMemGPR(OpSize::i32Bit, Mem, Constant(24), OpSize::i32Bit, MemOffsetType::SXTX, 1);
RestoreMXCSRState(MXCSR);
}
@@ -2701,7 +2701,7 @@ void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
// Note: we rematerialize Base/Mask in each block to avoid crossblock
// liveness.
Ref Base = XSaveBase(Op);
Ref Mask = _LoadMemGPR(OpSize::i64Bit, Base, Constant(512, ConstPad::NoPad), OpSize::i64Bit, MemOffsetType::SXTX, 1);
Ref Mask = _LoadMemGPR(OpSize::i64Bit, Base, Constant(512), OpSize::i64Bit, MemOffsetType::SXTX, 1);
Ref BitFlag = _Bfe(OpSize, FieldSize, BitIndex, Mask);
auto CondJump_ = CondJump(BitFlag, CondClass::NEQ);
@@ -2745,7 +2745,7 @@ void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
1,
[this, Op] {
Ref Base = XSaveBase(Op);
Ref MXCSR = _LoadMemGPR(OpSize::i32Bit, Base, Constant(24, ConstPad::NoPad), OpSize::i32Bit, MemOffsetType::SXTX, 1);
Ref MXCSR = _LoadMemGPR(OpSize::i32Bit, Base, Constant(24), OpSize::i32Bit, MemOffsetType::SXTX, 1);
RestoreMXCSRState(MXCSR);
},
[] { /* Intentionally do nothing*/ }, 2);
@@ -2759,13 +2759,13 @@ void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
{
auto NewFSW = _LoadMemGPR(OpSize::i16Bit, MemBase, Constant(2, ConstPad::NoPad), OpSize::i16Bit, MemOffsetType::SXTX, 1);
auto NewFSW = _LoadMemGPR(OpSize::i16Bit, MemBase, Constant(2), OpSize::i16Bit, MemOffsetType::SXTX, 1);
ReconstructX87StateFromFSW_Helper(NewFSW);
}
{
// Abridged FTW
auto NewFTW = _LoadMemGPR(OpSize::i8Bit, MemBase, Constant(4, ConstPad::NoPad), OpSize::i8Bit, MemOffsetType::SXTX, 1);
auto NewFTW = _LoadMemGPR(OpSize::i8Bit, MemBase, Constant(4), OpSize::i8Bit, MemOffsetType::SXTX, 1);
_StoreContextGPR(OpSize::i8Bit, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
}
@@ -2789,7 +2789,7 @@ void OpDispatchBuilder::RestoreSSEState(Ref MemBase) {
void OpDispatchBuilder::RestoreMXCSRState(Ref MXCSR) {
// Mask out unsupported bits
MXCSR = _And(OpSize::i32Bit, MXCSR, Constant(0xFFC0, ConstPad::NoPad));
MXCSR = _And(OpSize::i32Bit, MXCSR, Constant(0xFFC0));
_StoreContextGPR(OpSize::i32Bit, MXCSR, offsetof(FEXCore::Core::CPUState, mxcsr));
// We only support the rounding mode and FTZ bit being set
@@ -3988,7 +3988,7 @@ void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref
const auto ElementSizeInBits = IR::OpSizeAsBits(ElementSize);
const auto MaskConstant = uint64_t {1} << (ElementSizeInBits - 1);
Ref Mask = _VDupFromGPR(SrcSize, ElementSize, Constant(MaskConstant, ConstPad::NoPad));
Ref Mask = _VDupFromGPR(SrcSize, ElementSize, Constant(MaskConstant));
Ref AndTest = _VAnd(SrcSize, OpSize::i8Bit, Src2, Src1);
Ref AndNotTest = _VAndn(SrcSize, OpSize::i8Bit, Src2, Src1);
@@ -4589,7 +4589,7 @@ void OpDispatchBuilder::VPERMDOp(OpcodeArgs) {
// Get rid of any junk unrelated to the relevant selector index bits (bits [2:0])
Ref IndexMask = _VectorImm(DstSize, OpSize::i32Bit, 0b111);
Ref AddConst = Constant(0x03020100, ConstPad::NoPad);
Ref AddConst = Constant(0x03020100);
Ref Repeating3210 = _VDupFromGPR(DstSize, OpSize::i32Bit, AddConst);
Ref FinalIndices = VPERMDIndices(OpSizeFromDst(Op), Indices, IndexMask, Repeating3210);
@@ -4824,7 +4824,7 @@ Ref OpDispatchBuilder::VPERMILRegOpImpl(OpSize DstSize, IR::OpSize ElementSize,
Ref ShiftedIndices = _VShlI(DstSize, OpSize::i8Bit, IndexTrn3, IndexShift);
uint64_t VConstant = IsPD ? 0x0706050403020100 : 0x03020100;
Ref VectorConst = _VDupFromGPR(DstSize, ElementSize, Constant(VConstant, ConstPad::NoPad));
Ref VectorConst = _VDupFromGPR(DstSize, ElementSize, Constant(VConstant));
Ref FinalIndices {};
if (Is256Bit) {
@@ -4883,7 +4883,7 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
IntermediateResult = _VPCMPISTRX(Src1, Src2, Control);
}
Ref ZeroConst = Constant(0, ConstPad::NoPad);
Ref ZeroConst = Constant(0);
if (IsMask) {
// For the masked variant of the instructions, if control[6] is set, then we
@@ -4920,7 +4920,7 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
Ref ResultNoFlags = _Bfe(OpSize::i32Bit, 16, 0, IntermediateResult);
Ref IfZero = Constant(16 >> (Control & 1), ConstPad::NoPad);
Ref IfZero = Constant(16 >> (Control & 1));
Ref IfNotZero = UseMSBIndex ? _FindMSB(IR::OpSize::i32Bit, ResultNoFlags) : _FindLSB(IR::OpSize::i32Bit, ResultNoFlags);
Ref Result = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, ResultNoFlags, ZeroConst, IfZero, IfNotZero);
@@ -5110,7 +5110,7 @@ void OpDispatchBuilder::VPGATHER(OpcodeArgs) {
if (BaseAddr && VSIB.Displacement) {
BaseAddr = Add(OpSize::i64Bit, BaseAddr, VSIB.Displacement);
} else if (VSIB.Displacement) {
BaseAddr = Constant(VSIB.Displacement, ConstPad::NoPad);
BaseAddr = Constant(VSIB.Displacement);
} else if (!BaseAddr) {
BaseAddr = Invalid();
}
@@ -5156,7 +5156,7 @@ void OpDispatchBuilder::Extrq_imm(OpcodeArgs) {
}
const uint64_t Mask = ~0ULL >> (MaskWidth == 0 ? 0 : (64 - MaskWidth));
const Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask, ConstPad::NoPad));
const Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask));
Result = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Result, MaskVector);
StoreResultFPR(Op, Result);
@@ -5170,7 +5170,7 @@ void OpDispatchBuilder::Insertq_imm(OpcodeArgs) {
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
const uint64_t Mask = ~0ULL >> (MaskWidth == 0 ? 0 : (64 - MaskWidth));
Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask, ConstPad::NoPad));
Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask));
// Mask incoming source.
Src = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Src, MaskVector);
@@ -41,13 +41,13 @@ void OpDispatchBuilder::SetX87FTW(Ref FTW) {
// Invert FTW and clear the odd bits. Even bits are 1 if the pair
// is not equal to 11, and odd bits are 0.
FTW = _Andn(OpSize::i32Bit, Constant(0x55555555, ConstPad::NoPad), FTW);
FTW = _Andn(OpSize::i32Bit, Constant(0x55555555), FTW);
// All that's left is to compact away the odd bits. That is a Morton
// deinterleave operation, which has a standard solution. See
// https://stackoverflow.com/questions/3137266/how-to-de-interleave-bits-unmortonizing
FTW = _And(OpSize::i32Bit, _Orlshr(OpSize::i32Bit, FTW, FTW, 1), Constant(0x33333333, ConstPad::NoPad));
FTW = _And(OpSize::i32Bit, _Orlshr(OpSize::i32Bit, FTW, FTW, 2), Constant(0x0f0f0f0f, ConstPad::NoPad));
FTW = _And(OpSize::i32Bit, _Orlshr(OpSize::i32Bit, FTW, FTW, 1), Constant(0x33333333));
FTW = _And(OpSize::i32Bit, _Orlshr(OpSize::i32Bit, FTW, FTW, 2), Constant(0x0f0f0f0f));
FTW = _Orlshr(OpSize::i32Bit, FTW, FTW, 4);
// ...and that's it. StoreContext implicitly does the final masking.
@@ -107,16 +107,16 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
SaveNZCV();
// Extract sign and make integer absolute
auto zero = Constant(0, ConstPad::NoPad);
auto zero = Constant(0);
_SubNZCV(OpSize::i64Bit, Data, zero);
auto sign = _NZCVSelect(OpSize::i64Bit, CondClass::SLT, Constant(0x8000, ConstPad::NoPad), zero);
auto sign = _NZCVSelect(OpSize::i64Bit, CondClass::SLT, Constant(0x8000), zero);
auto absolute = _Neg(OpSize::i64Bit, Data, CondClass::MI);
// left justify the absolute integer
auto shift = Sub(OpSize::i64Bit, Constant(63, ConstPad::NoPad), _FindMSB(IR::OpSize::i64Bit, absolute));
auto shift = Sub(OpSize::i64Bit, Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
auto shifted = _Lshl(OpSize::i64Bit, absolute, shift);
auto adjusted_exponent = Sub(OpSize::i64Bit, Constant(0x3fff + 63, ConstPad::NoPad), shift);
auto adjusted_exponent = Sub(OpSize::i64Bit, Constant(0x3fff + 63), shift);
auto zeroed_exponent = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, absolute, zero, zero, adjusted_exponent);
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
@@ -159,11 +159,11 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
// Extract the 80-bit float value to check for special cases
// Get the upper 64 bits which contain sign and exponent and then the exponent from upper.
Ref Upper = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, Data, 1);
Ref Exponent = _And(OpSize::i64Bit, Upper, Constant(0x7fff, ConstPad::NoPad));
Ref Exponent = _And(OpSize::i64Bit, Upper, Constant(0x7fff));
// Check for NaN/Infinity: exponent = 0x7fff
SaveNZCV();
_TestNZ(OpSize::i64Bit, Exponent, Constant(0x7fff, ConstPad::NoPad));
_TestNZ(OpSize::i64Bit, Exponent, Constant(0x7fff));
Ref IsSpecial = _NZCVSelect01(CondClass::EQ);
// For overflow detection, check if exponent indicates a value >= 2^15
@@ -340,17 +340,17 @@ Ref OpDispatchBuilder::GetX87FTW_Helper() {
// https://graphics.stanford.edu/~seander/bithacks.html#InterleaveBMN
Ref X = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
X = _Orlshl(OpSize::i32Bit, X, X, 4);
X = _And(OpSize::i32Bit, X, Constant(0x0f0f0f0f, ConstPad::NoPad));
X = _And(OpSize::i32Bit, X, Constant(0x0f0f0f0f));
X = _Orlshl(OpSize::i32Bit, X, X, 2);
X = _And(OpSize::i32Bit, X, Constant(0x33333333, ConstPad::NoPad));
X = _And(OpSize::i32Bit, X, Constant(0x33333333));
X = _Orlshl(OpSize::i32Bit, X, X, 1);
X = _And(OpSize::i32Bit, X, Constant(0x55555555, ConstPad::NoPad));
X = _And(OpSize::i32Bit, X, Constant(0x55555555));
X = _Orlshl(OpSize::i32Bit, X, X, 1);
// The above sequence sets valid to 11 and empty to 00, so invert to finalize.
static_assert(static_cast<uint8_t>(FPState::X87Tag::Valid) == 0b00);
static_assert(static_cast<uint8_t>(FPState::X87Tag::Empty) == 0b11);
return _Xor(OpSize::i32Bit, X, Constant(0xffff, ConstPad::NoPad));
return _Xor(OpSize::i32Bit, X, Constant(0xffff));
}
void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
@@ -387,33 +387,33 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
_StoreMemGPR(Size, Mem, FCW, Size);
}
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1); }
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
auto ZeroConst = Constant(0, ConstPad::NoPad);
auto ZeroConst = Constant(0);
{
// FTW
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
}
{
// Instruction Offset
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
}
{
// Instruction CS selector (+ Opcode)
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
}
{
// Data pointer offset
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
}
{
// Data pointer selector
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
}
}
@@ -485,44 +485,43 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
_StoreMemGPR(Size, Mem, FCW, Size);
}
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1); }
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
auto ZeroConst = Constant(0, ConstPad::NoPad);
auto ZeroConst = Constant(0);
{
// FTW
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
}
{
// Instruction Offset
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
}
{
// Instruction CS selector (+ Opcode)
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
}
{
// Data pointer offset
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
}
{
// Data pointer selector
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
}
auto SevenConst = Constant(7, ConstPad::NoPad);
auto SevenConst = Constant(7);
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
for (int i = 0; i < 7; ++i) {
Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
if (ReducedPrecisionMode) {
data = _F80CVTTo(data, OpSize::i64Bit);
}
_StoreMemFPR(OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i), ConstPad::NoPad), OpSize::i8Bit,
MemOffsetType::SXTX, 1);
_StoreMemFPR(OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
}
@@ -534,11 +533,9 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
// ST7 broken in to two parts
// Lower 64bits [63:0]
// upper 16 bits [79:64]
_StoreMemFPR(OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10), ConstPad::NoPad), OpSize::i8Bit,
MemOffsetType::SXTX, 1);
_StoreMemFPR(OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
auto topBytes = _VDupElement(OpSize::i128Bit, OpSize::i16Bit, data, 4);
_StoreMemFPR(OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8, ConstPad::NoPad), OpSize::i8Bit,
MemOffsetType::SXTX, 1);
_StoreMemFPR(OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
// reset to default
FNINIT(Op);
@@ -555,28 +552,27 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
// ignore the rounding precision, we're always 64-bit in F64.
// extract rounding mode
Ref roundingMode = NewFCW;
auto roundShift = Constant(10, ConstPad::NoPad);
auto roundMask = Constant(3, ConstPad::NoPad);
auto roundShift = Constant(10);
auto roundMask = Constant(3);
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
roundingMode = _And(OpSize::i32Bit, roundingMode, roundMask);
_SetRoundingMode(roundingMode, false, roundingMode);
}
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 1, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1);
Ref Top = ReconstructX87StateFromFSW_Helper(NewFSW);
{
// FTW
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1));
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
}
auto SevenConst = Constant(7, ConstPad::NoPad);
auto low = Constant(~0ULL, ConstPad::NoPad);
auto high = Constant(0xFFFF, ConstPad::NoPad);
auto SevenConst = Constant(7);
auto low = Constant(~0ULL);
auto high = Constant(0xFFFF);
Ref Mask = _VLoadTwoGPRs(low, high);
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
for (int i = 0; i < 7; ++i) {
Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i), ConstPad::NoPad), OpSize::i8Bit,
MemOffsetType::SXTX, 1);
Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
// Mask off the top bits
Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask);
if (ReducedPrecisionMode) {
@@ -592,10 +588,8 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
// ST7 broken in to two parts
// Lower 64bits [63:0]
// upper 16 bits [79:64]
Ref Reg =
_LoadMemFPR(OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7), ConstPad::NoPad), OpSize::i8Bit, MemOffsetType::SXTX, 1);
Ref RegHigh = _LoadMemFPR(OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8, ConstPad::NoPad), OpSize::i8Bit,
MemOffsetType::SXTX, 1);
Ref Reg = _LoadMemFPR(OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
Ref RegHigh = _LoadMemFPR(OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
Reg = _VInsElement(OpSize::i128Bit, OpSize::i16Bit, 4, 0, Reg, RegHigh);
if (ReducedPrecisionMode) {
Reg = _F80CVT(OpSize::i64Bit, Reg); // Convert to double precision
@@ -624,13 +618,13 @@ void OpDispatchBuilder::FXCH(OpcodeArgs) {
if (Offset != 0) {
_F80StackXchange(Offset);
}
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0, ConstPad::NoPad));
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
}
void OpDispatchBuilder::X87FYL2X(OpcodeArgs, bool IsFYL2XP1) {
if (IsFYL2XP1) {
// create an add between top of stack and 1.
Ref One = ReducedPrecisionMode ? _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(0x3FF0000000000000, ConstPad::NoPad)) :
Ref One = ReducedPrecisionMode ? _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(0x3FF0000000000000)) :
LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
_F80AddValue(0, One);
}
@@ -671,7 +665,7 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
if (WhichFlags == FCOMIFlags::FLAGS_X87) {
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0, ConstPad::NoPad));
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
} else {
@@ -681,7 +675,7 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
// PF is stored inverted, so invert from the host flag.
// TODO: This could perhaps be optimized?
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, Constant(1, ConstPad::NoPad));
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, Constant(1));
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
}
@@ -706,7 +700,7 @@ void OpDispatchBuilder::FTST(OpcodeArgs) {
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0, ConstPad::NoPad));
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
@@ -717,7 +711,7 @@ void OpDispatchBuilder::FTST(OpcodeArgs) {
void OpDispatchBuilder::X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2) {
DeriveOp(Result, IROp, _F80SCALEStack());
if (ZeroC2) {
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(Constant(0, ConstPad::NoPad));
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(Constant(0));
}
}
@@ -743,7 +737,7 @@ void OpDispatchBuilder::X87ModifySTP(OpcodeArgs, bool Inc) {
Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
// Start with the top value
auto Top = T ? T : GetX87Top();
Ref FSW = _Lshl(OpSize::i64Bit, Top, Constant(11, ConstPad::NoPad));
Ref FSW = _Lshl(OpSize::i64Bit, Top, Constant(11));
// We must construct the FSW from our various bits
auto C0 = GetRFLAG(FEXCore::X86State::X87FLAG_C0_LOC);
@@ -775,20 +769,20 @@ void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
// Clear the exception flag bit
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Constant(0, ConstPad::NoPad));
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Constant(0));
}
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
_SyncStackToSlow(); // Invalidate x87 register caches
auto Zero = Constant(0, ConstPad::NoPad);
auto Zero = Constant(0);
if (ReducedPrecisionMode) {
_SetRoundingMode(Zero, false, Zero);
}
// Init FCW to 0x037F
auto NewFCW = Constant(0x037F, ConstPad::NoPad);
auto NewFCW = Constant(0x037F);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
// Set top to zero
@@ -867,7 +861,7 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
auto TopValid = _StackValidTag(0);
// In the case of top being invalid then C3:C2:C0 is 0b101
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1, ConstPad::NoPad));
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1));
auto C2 = TopValid;
auto C0 = C3; // Mirror C3 until something other than zero is supported
@@ -36,12 +36,12 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
_SetRoundingMode(roundingMode, false, roundingMode);
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size), ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MemOffsetType::SXTX, 1);
ReconstructX87StateFromFSW_Helper(NewFSW);
{
// FTW
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1));
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
}
}
@@ -86,7 +86,7 @@ void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
}
void OpDispatchBuilder::FLDF64_Const(OpcodeArgs, uint64_t Num) {
auto Data = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(Num, ConstPad::NoPad));
auto Data = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(Num));
_PushStack(Data, Data, OpSize::i64Bit);
}
@@ -376,21 +376,21 @@ void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) {
Ref Gpr = _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, Node, 0);
// zero case
Ref ExpZV = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(0xfff0'0000'0000'0000UL, ConstPad::NoPad));
Ref ExpZV = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(0xfff0'0000'0000'0000UL));
Ref SigZV = Node;
// non zero case
Ref ExpNZ = _Bfe(OpSize::i64Bit, 11, 52, Gpr);
ExpNZ = Sub(OpSize::i64Bit, ExpNZ, Constant(1023, ConstPad::NoPad));
ExpNZ = Sub(OpSize::i64Bit, ExpNZ, Constant(1023));
Ref ExpNZV = _Float_FromGPR_S(OpSize::i64Bit, OpSize::i64Bit, ExpNZ);
Ref SigNZ = _And(OpSize::i64Bit, Gpr, Constant(0x800f'ffff'ffff'ffffLL, ConstPad::NoPad));
SigNZ = _Or(OpSize::i64Bit, SigNZ, Constant(0x3ff0'0000'0000'0000LL, ConstPad::NoPad));
Ref SigNZ = _And(OpSize::i64Bit, Gpr, Constant(0x800f'ffff'ffff'ffffLL));
SigNZ = _Or(OpSize::i64Bit, SigNZ, Constant(0x3ff0'0000'0000'0000LL));
Ref SigNZV = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, SigNZ);
// Comparison and select to push onto stack
SaveNZCV();
_TestNZ(OpSize::i64Bit, Gpr, Constant(0x7fff'ffff'ffff'ffffUL, ConstPad::NoPad));
_TestNZ(OpSize::i64Bit, Gpr, Constant(0x7fff'ffff'ffff'ffffUL));
Ref Sig = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, SigZV, SigNZV);
Ref Exp = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, ExpZV, ExpNZV);
+1 -1
View File
@@ -936,7 +936,7 @@
]
},
"GPR = Constant i64:$Constant, ConstPad:$Pad, i32:$MaxBytes{0}": {
"GPR = Constant i64:$Constant, ConstPad:$Pad{IR::ConstPad::NoPad}, i32:$MaxBytes{0}": {
"Desc": ["Generates a 64bit constant inside of a GPR",
"Unsupported to create a constant in FPR"
],
+5 -5
View File
@@ -109,10 +109,10 @@ public:
return _Jump(InvalidNode);
}
IRPair<IROp_CondJump> _CondJump(Ref ssa0, CondClass cond = CondClass::NEQ) {
return _CondJump(ssa0, _Constant(0, ConstPad::NoPad), InvalidNode, InvalidNode, cond, GetOpSize(ssa0));
return _CondJump(ssa0, _Constant(0), InvalidNode, InvalidNode, cond, GetOpSize(ssa0));
}
IRPair<IROp_CondJump> _CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClass cond = CondClass::NEQ) {
return _CondJump(ssa0, _Constant(0, ConstPad::NoPad), ssa1, ssa2, cond, GetOpSize(ssa0));
return _CondJump(ssa0, _Constant(0), ssa1, ssa2, cond, GetOpSize(ssa0));
}
IRPair<IROp_LoadContext> _LoadContextGPR(OpSize ByteSize, uint32_t Offset) {
@@ -184,7 +184,7 @@ public:
}
IRPair<IROp_Select> To01(FEXCore::IR::OpSize CompareSize, OrderedNode* Cmp1) {
return Select01(CompareSize, CondClass::NEQ, Cmp1, Constant(0, ConstPad::NoPad));
return Select01(CompareSize, CondClass::NEQ, Cmp1, Constant(0));
}
IRPair<IROp_NZCVSelect> _NZCVSelect01(CondClass Cond) {
@@ -203,7 +203,7 @@ public:
Src2 = -Src2;
}
auto Dest = _Add(Size, Src1, Constant(Src2, ConstPad::NoPad));
auto Dest = _Add(Size, Src1, Constant(Src2));
Dest.first->Header.Op = Op;
return Dest;
}
@@ -249,7 +249,7 @@ public:
Ref ConstantRefs[32];
uint32_t NrConstants;
Ref Constant(int64_t Value, ConstPad Pad, int32_t MaxBytes = 0) {
Ref Constant(int64_t Value, ConstPad Pad = IR::ConstPad::NoPad, int32_t MaxBytes = 0) {
const ConstantData Data {
.Value = Value,
.Pad = Pad,
@@ -514,8 +514,8 @@ bool ConstrainedRAPass::TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header*
if (CPUID->DoesXCRFunctionReportConstantData(ConstantFunction)) {
const auto Result = CPUID->RunXCRFunction(ConstantFunction);
IREmit->SetWriteCursorBefore(CodeNode);
IREmit->_Constant(Result.eax, ConstPad::NoPad).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
IREmit->_Constant(Result.edx, ConstPad::NoPad).Node->Reg = PhysicalRegister(Op->OutEDX).Raw;
IREmit->_Constant(Result.eax).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
IREmit->_Constant(Result.edx).Node->Reg = PhysicalRegister(Op->OutEDX).Raw;
IREmit->RemovePostRA(CodeNode);
return false;
}
@@ -533,10 +533,10 @@ bool ConstrainedRAPass::TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header*
IREmit->SetWriteCursorBefore(CodeNode);
IREmit->_Fence(IR::FenceType::Inst);
IREmit->_Constant(Result.eax, ConstPad::NoPad).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
IREmit->_Constant(Result.ebx, ConstPad::NoPad).Node->Reg = PhysicalRegister(Op->OutEBX).Raw;
IREmit->_Constant(Result.ecx, ConstPad::NoPad).Node->Reg = PhysicalRegister(Op->OutECX).Raw;
IREmit->_Constant(Result.edx, ConstPad::NoPad).Node->Reg = PhysicalRegister(Op->OutEDX).Raw;
IREmit->_Constant(Result.eax).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
IREmit->_Constant(Result.ebx).Node->Reg = PhysicalRegister(Op->OutEBX).Raw;
IREmit->_Constant(Result.ecx).Node->Reg = PhysicalRegister(Op->OutECX).Raw;
IREmit->_Constant(Result.edx).Node->Reg = PhysicalRegister(Op->OutEDX).Raw;
IREmit->RemovePostRA(CodeNode);
return false;
}
@@ -387,11 +387,11 @@ inline void X87StackOptimization::Reset() {
inline Ref X87StackOptimization::GetConstant(ssize_t Offset) {
if (Offset < 0 || Offset >= X87StackOptimization::ConstantPool.size()) {
// not dealt by pool
return IREmit->_Constant(Offset, ConstPad::NoPad);
return IREmit->_Constant(Offset);
}
if (ConstantPool[Offset] == nullptr) {
ConstantPool[Offset] = IREmit->_Constant(Offset, ConstPad::NoPad);
ConstantPool[Offset] = IREmit->_Constant(Offset);
}
return ConstantPool[Offset];
}