mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 14:00:16 +02:00
FEXCore: Switch constant emission to default to NoPad
Most constants don't need to be padded for relocations. So now that these have all been audited, switch to defaulting to NoPad to reduce verbosity. The number of constant that need to be explicitly padded are now marked and with all the prior changes, this allows bisecting if something has gone wrong.
This commit is contained in:
22 files changed
+264
-276
No files matched your search
@@ -11,7 +11,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Offset) : IREmit->Constant(A.Offset, IR::ConstPad::NoPad);
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Offset) : IREmit->Constant(A.Offset);
|
||||
}
|
||||
|
||||
if (A.Index) {
|
||||
@@ -21,7 +21,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
|
||||
if (Tmp) {
|
||||
Tmp = IREmit->_AddShift(GPRSize, Tmp, A.Index, ShiftType::LSL, Log2);
|
||||
} else {
|
||||
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->Constant(Log2, IR::ConstPad::NoPad));
|
||||
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->Constant(Log2));
|
||||
}
|
||||
} else {
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
@@ -40,7 +40,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
|
||||
} else if (A.Offset) {
|
||||
uint64_t X = A.Offset;
|
||||
X &= (1ull << Bits) - 1;
|
||||
Tmp = IREmit->Constant(X, IR::ConstPad::NoPad);
|
||||
Tmp = IREmit->Constant(X);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -48,7 +48,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPR
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
}
|
||||
|
||||
return Tmp ?: IREmit->Constant(0, IR::ConstPad::NoPad);
|
||||
return Tmp ?: IREmit->Constant(0);
|
||||
}
|
||||
|
||||
AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSize GPRSize, bool HostSupportsTSOImm9, bool AtomicTSO,
|
||||
@@ -102,7 +102,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->Constant(A.Offset, ConstPad::NoPad),
|
||||
.Index = IREmit->Constant(A.Offset),
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
@@ -135,7 +135,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, const AddressMode& A, IR::OpSiz
|
||||
|
||||
return {
|
||||
.Base = LoadEffectiveAddress(IREmit, B, GPRSize, true /* AddSegmentBase */, false),
|
||||
.Index = IREmit->Constant(A.Offset, ConstPad::NoPad),
|
||||
.Index = IREmit->Constant(A.Offset),
|
||||
.IndexType = MemOffsetType::SXTX,
|
||||
.IndexScale = 1,
|
||||
};
|
||||
|
||||
@@ -117,7 +117,7 @@ public:
|
||||
// Choose to pad or not depending on if code-caching is enabled.
|
||||
AUTOPAD,
|
||||
};
|
||||
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, PadType Pad, int MaxBytes = 0);
|
||||
void LoadConstant(ARMEmitter::Size s, ARMEmitter::Register Reg, uint64_t Constant, PadType Pad = PadType::NOPAD, int MaxBytes = 0);
|
||||
|
||||
protected:
|
||||
FEXCore::Context::ContextImpl* EmitterCTX;
|
||||
|
||||
@@ -970,15 +970,13 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
|
||||
// Thunk entry-points don't get cached, don't need to be padded.
|
||||
if (GPRSize == IR::OpSize::i64Bit) {
|
||||
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint, IR::ConstPad::NoPad), GPRSize);
|
||||
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint), GPRSize);
|
||||
R->Reg = IR::PhysicalRegister(IR::RegClass::GPRFixed, X86State::REG_R11).Raw;
|
||||
} else {
|
||||
emit->_StoreContextFPR(GPRSize,
|
||||
emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint, IR::ConstPad::NoPad)),
|
||||
emit->_StoreContextFPR(GPRSize, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint, IR::ConstPad::NoPad), IR::BranchHint::None,
|
||||
emit->Invalid(), emit->Invalid());
|
||||
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint), IR::BranchHint::None, emit->Invalid(), emit->Invalid());
|
||||
},
|
||||
ThunkHandler, (void*)GuestThunkEntrypoint);
|
||||
|
||||
|
||||
@@ -188,7 +188,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
if (std::popcount(VirtualMemorySize) == 1) {
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), VirtualMemorySize - 1);
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP4, VirtualMemorySize);
|
||||
and_(ARMEmitter::Size::i64Bit, TMP4, RipReg.R(), TMP4);
|
||||
}
|
||||
|
||||
@@ -261,7 +261,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
|
||||
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
@@ -429,7 +429,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
PopCalleeSavedRegisters();
|
||||
ret();
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, 0);
|
||||
ldr(ARMEmitter::XReg::x1, ARMEmitter::Reg::r1);
|
||||
}
|
||||
}
|
||||
@@ -488,7 +488,7 @@ void Dispatcher::EmitDispatcher() {
|
||||
|
||||
// Now push the callback return trampoline to the guest stack
|
||||
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->SignalDelegation->GetThunkCallbackRET(), CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, CTX->SignalDelegation->GetThunkCallbackRET());
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r2, ARMEmitter::Reg::r2, CTX->Config.Is64BitMode ? 16 : 12);
|
||||
|
||||
@@ -1297,7 +1297,7 @@ DEF_OP(MaskGenerateFromBitWidth) {
|
||||
auto Op = IROp->C<IR::IROp_MaskGenerateFromBitWidth>();
|
||||
auto BitWidth = GetReg(Op->BitWidth);
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, -1, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, -1);
|
||||
cmp(ARMEmitter::Size::i64Bit, BitWidth, 0);
|
||||
lslv(ARMEmitter::Size::i64Bit, TMP2, TMP1, BitWidth);
|
||||
csinv(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1, TMP2, ARMEmitter::Condition::CC_EQ);
|
||||
|
||||
@@ -271,7 +271,7 @@ DEF_OP(Syscall) {
|
||||
// Still without overwriting registers that matter
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GPRSpillMask & 0xFFFF, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, GPRSpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
uint64_t SPOffset = AlignUp(FEXCore::HLE::SyscallArguments::MAX_ARGS * 8, 16);
|
||||
@@ -362,29 +362,29 @@ DEF_OP(ValidateCode) {
|
||||
|
||||
EmitCheck(8, [&]() {
|
||||
ldr(TMP1, Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, *(const uint64_t*)(OldCode + Offset), CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP2, *(const uint64_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
EmitCheck(4, [&]() {
|
||||
ldr(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint32_t*)(OldCode + Offset), CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint32_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
EmitCheck(2, [&]() {
|
||||
ldrh(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint16_t*)(OldCode + Offset), CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint16_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
EmitCheck(1, [&]() {
|
||||
ldrb(TMP1.W(), Base, Offset);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint8_t*)(OldCode + Offset), CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP2, *(const uint8_t*)(OldCode + Offset));
|
||||
});
|
||||
|
||||
ARMEmitter::ForwardLabel End;
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 0);
|
||||
b_OrRestart(&End);
|
||||
BindOrRestart(&Fail);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, Dst, 1);
|
||||
BindOrRestart(&End);
|
||||
}
|
||||
|
||||
|
||||
@@ -135,7 +135,7 @@ DEF_OP(VAESKeyGenAssist) {
|
||||
if (Op->RCON) {
|
||||
tbl(Dst.Q(), Dst.Q(), Swizzle.Q());
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, static_cast<uint64_t>(Op->RCON) << 32, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, static_cast<uint64_t>(Op->RCON) << 32);
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP2.Q(), TMP1);
|
||||
eor(Dst.Q(), Dst.Q(), VTMP2.Q());
|
||||
} else {
|
||||
|
||||
@@ -758,14 +758,14 @@ void Arm64JITCore::EmitTFCheck() {
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Constant);
|
||||
str(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
ldr(TMP1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.GuestSignal_SIGTRAP));
|
||||
br(TMP1);
|
||||
|
||||
(void)Bind(&l_TFBlocked);
|
||||
// If TF was blocked for this instruction, unblock it for the next.
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 0b11);
|
||||
strb(TMP1, STATE_PTR(CpuStateFrame, State.flags[X86State::RFLAG_TF_RAW_LOC]));
|
||||
(void)Bind(&l_TFUnset);
|
||||
}
|
||||
@@ -805,7 +805,7 @@ void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool C
|
||||
if (ARMEmitter::IsImmAddSub(TotalSpillSlotsSize)) {
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
|
||||
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
}
|
||||
}
|
||||
@@ -1163,7 +1163,7 @@ void Arm64JITCore::ResetStack() {
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
|
||||
} else {
|
||||
// Too big to fit in a 12bit immediate
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -378,7 +378,7 @@ DEF_OP(SpillRegister) {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
if (SlotOffset > LSByteMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
strb(Src, ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
strb(Src, ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -387,7 +387,7 @@ DEF_OP(SpillRegister) {
|
||||
}
|
||||
case IR::OpSize::i16Bit: {
|
||||
if (SlotOffset > LSHalfMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
strh(Src, ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
strh(Src, ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -396,7 +396,7 @@ DEF_OP(SpillRegister) {
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
if (SlotOffset > LSWordMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
str(Src.W(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
str(Src.W(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -405,7 +405,7 @@ DEF_OP(SpillRegister) {
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (SlotOffset > LSDWordMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
str(Src.X(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
str(Src.X(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -420,7 +420,7 @@ DEF_OP(SpillRegister) {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: {
|
||||
if (SlotOffset > LSWordMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
str(Src.S(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
str(Src.S(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -429,7 +429,7 @@ DEF_OP(SpillRegister) {
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (SlotOffset > LSDWordMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
str(Src.D(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
str(Src.D(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -438,7 +438,7 @@ DEF_OP(SpillRegister) {
|
||||
}
|
||||
case IR::OpSize::i128Bit: {
|
||||
if (SlotOffset > LSQWordMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
str(Src.Q(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
str(Src.Q(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -467,7 +467,7 @@ DEF_OP(FillRegister) {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i8Bit: {
|
||||
if (SlotOffset > LSByteMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
ldrb(Dst, ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
ldrb(Dst, ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -476,7 +476,7 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
case IR::OpSize::i16Bit: {
|
||||
if (SlotOffset > LSHalfMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
ldrh(Dst, ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
ldrh(Dst, ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -485,7 +485,7 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
case IR::OpSize::i32Bit: {
|
||||
if (SlotOffset > LSWordMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
ldr(Dst.W(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
ldr(Dst.W(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -494,7 +494,7 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (SlotOffset > LSDWordMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
ldr(Dst.X(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
ldr(Dst.X(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -509,7 +509,7 @@ DEF_OP(FillRegister) {
|
||||
switch (OpSize) {
|
||||
case IR::OpSize::i32Bit: {
|
||||
if (SlotOffset > LSWordMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
ldr(Dst.S(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
ldr(Dst.S(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -518,7 +518,7 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
case IR::OpSize::i64Bit: {
|
||||
if (SlotOffset > LSDWordMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
ldr(Dst.D(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
ldr(Dst.D(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -527,7 +527,7 @@ DEF_OP(FillRegister) {
|
||||
}
|
||||
case IR::OpSize::i128Bit: {
|
||||
if (SlotOffset > LSQWordMaxUnsignedOffset) {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, SlotOffset);
|
||||
ldr(Dst.Q(), ARMEmitter::Reg::rsp, TMP1.R(), ARMEmitter::ExtendedType::LSL_64, 0);
|
||||
} else {
|
||||
ldr(Dst.Q(), ARMEmitter::Reg::rsp, SlotOffset);
|
||||
@@ -609,7 +609,7 @@ ARMEmitter::Register Arm64JITCore::ApplyMemOperand(IR::OpSize AccessSize, ARMEmi
|
||||
if (Const == 0) {
|
||||
return Base;
|
||||
}
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Tmp, Const, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, Tmp, Const);
|
||||
add(ARMEmitter::Size::i64Bit, Tmp, Base, Tmp, ARMEmitter::ShiftType::LSL, FEXCore::ilog2(OffsetScale));
|
||||
} else {
|
||||
auto RegOffset = GetReg(Offset);
|
||||
@@ -1213,7 +1213,7 @@ DEF_OP(VLoadVectorGatherMasked) {
|
||||
AddrReg = GetReg(Op->AddrBase);
|
||||
} else {
|
||||
///< OpcodeDispatcher didn't provide a Base address while SVE requires one.
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, AddrReg, 0, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, AddrReg, 0);
|
||||
}
|
||||
MemDst = ARMEmitter::SVEMemOperand(AddrReg.X(), VectorIndexLow.Z(), ModType, SVEScale);
|
||||
}
|
||||
@@ -1299,7 +1299,7 @@ DEF_OP(VLoadVectorGatherMaskedQPS) {
|
||||
AddrReg = *BaseAddr;
|
||||
} else {
|
||||
///< OpcodeDispatcher didn't provide a Base address while SVE requires one.
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, AddrReg, 0, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, AddrReg, 0);
|
||||
}
|
||||
MemDst = ARMEmitter::SVEMemOperand(AddrReg.X(), VectorIndex.Z(), ModType, SVEScale);
|
||||
}
|
||||
|
||||
@@ -73,7 +73,7 @@ DEF_OP(Break) {
|
||||
uint64_t Constant {};
|
||||
memcpy(&Constant, &State, sizeof(State));
|
||||
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Constant, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r1, Constant);
|
||||
str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData));
|
||||
|
||||
switch (Op->Reason.Signal) {
|
||||
@@ -234,7 +234,7 @@ DEF_OP(ProcessorID) {
|
||||
// 16bit LoadConstant to be a single instruction
|
||||
// We must always spill at least one register (x8) so this value always has a bit set
|
||||
// This gives the signal handler a value to check to see if we are in a syscall at all
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SpillMask & 0xFFFF, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, SpillMask & 0xFFFF);
|
||||
str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CpuStateFrame, InSyscallInfo));
|
||||
|
||||
// Allocate some temporary space for storing the uint32_t CPU and Node IDs
|
||||
@@ -247,7 +247,7 @@ DEF_OP(ProcessorID) {
|
||||
#else
|
||||
constexpr auto GetCPUSyscallNum = SYS_getcpu;
|
||||
#endif
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, GetCPUSyscallNum, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r8, GetCPUSyscallNum);
|
||||
|
||||
// CPU pointer in x0
|
||||
add(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, ARMEmitter::Reg::rsp, 0);
|
||||
@@ -307,7 +307,7 @@ DEF_OP(MonoBackpatcherWrite) {
|
||||
|
||||
#ifdef _M_ARM_64EC
|
||||
ldr(TMP2, ARMEmitter::XReg::x18, TEB_CPU_AREA_OFFSET);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i32Bit, TMP1, 1);
|
||||
strb(TMP1.W(), TMP2, CPU_AREA_IN_SYSCALL_CALLBACK_OFFSET);
|
||||
#endif
|
||||
|
||||
|
||||
@@ -939,7 +939,7 @@ DEF_OP(VectorImm) {
|
||||
LOGMAN_THROW_A_FMT(Op->ShiftAmount == 0, "SVE VectorImm doesn't support a shift");
|
||||
if (ElementSize > IR::OpSize::i8Bit && (Op->Immediate & 0x80)) {
|
||||
// SVE dup uses sign extension where VectorImm wants zext
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Immediate, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, Op->Immediate);
|
||||
dup(SubRegSize, Dst.Z(), TMP1);
|
||||
} else {
|
||||
dup_imm(SubRegSize, Dst.Z(), static_cast<int8_t>(Op->Immediate));
|
||||
@@ -947,7 +947,7 @@ DEF_OP(VectorImm) {
|
||||
} else {
|
||||
if (ElementSize == IR::OpSize::i64Bit) {
|
||||
// movi with 64bit element size doesn't do what we want here
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, static_cast<uint64_t>(Op->Immediate) << Op->ShiftAmount, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, static_cast<uint64_t>(Op->Immediate) << Op->ShiftAmount);
|
||||
dup(SubRegSize, Dst.Q(), TMP1.R());
|
||||
} else {
|
||||
movi(SubRegSize, Dst.Q(), Op->Immediate, Op->ShiftAmount);
|
||||
@@ -2521,7 +2521,7 @@ DEF_OP(VUShl) {
|
||||
movi(SubRegSize, VTMP1.Q(), MaxShift);
|
||||
umin(SubRegSize, VTMP1.Q(), VTMP1.Q(), ShiftVector.Q());
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift);
|
||||
dup(SubRegSize, VTMP1.Q(), TMP1.R());
|
||||
|
||||
// UMIN is silly on Adv.SIMD and doesn't have a variant that handles 64-bit elements
|
||||
@@ -2577,7 +2577,7 @@ DEF_OP(VUShr) {
|
||||
movi(SubRegSize, VTMP1.Q(), MaxShift);
|
||||
umin(SubRegSize, VTMP1.Q(), VTMP1.Q(), ShiftVector.Q());
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift);
|
||||
dup(SubRegSize, VTMP1.Q(), TMP1.R());
|
||||
|
||||
// UMIN is silly on Adv.SIMD and doesn't have a variant that handles 64-bit elements
|
||||
@@ -2636,7 +2636,7 @@ DEF_OP(VSShr) {
|
||||
movi(SubRegSize, VTMP1.Q(), MaxShift);
|
||||
umin(SubRegSize, VTMP1.Q(), VTMP1.Q(), ShiftVector.Q());
|
||||
} else {
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift, CPU::Arm64Emitter::PadType::NOPAD);
|
||||
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, MaxShift);
|
||||
dup(SubRegSize, VTMP1.Q(), TMP1.R());
|
||||
|
||||
// UMIN is silly on Adv.SIMD and doesn't have a variant that handles 64-bit elements
|
||||
|
||||
@@ -475,8 +475,7 @@ void OpDispatchBuilder::POPSegmentOp(OpcodeArgs, uint32_t SegmentReg) {
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX:
|
||||
// Unset the 'active' bit in the packed TF, skipping the single step exception after this instruction
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_TF_RAW_LOC>(
|
||||
_And(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::RFLAG_TF_RAW_LOC), Constant(1, ConstPad::NoPad)));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_TF_RAW_LOC>(_And(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::RFLAG_TF_RAW_LOC), Constant(1)));
|
||||
_StoreContextGPR(DstSize, NewSegment, offsetof(FEXCore::Core::CPUState, ss_idx));
|
||||
break;
|
||||
case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX:
|
||||
@@ -1169,7 +1168,7 @@ void OpDispatchBuilder::SAHFOp(OpcodeArgs) {
|
||||
Ref Src = LoadGPRRegister(X86State::REG_RAX, OpSize::i8Bit, 8);
|
||||
|
||||
// Clear bits that aren't supposed to be set
|
||||
Src = _Andn(OpSize::i64Bit, Src, Constant(0b101000, ConstPad::NoPad));
|
||||
Src = _Andn(OpSize::i64Bit, Src, Constant(0b101000));
|
||||
|
||||
// Set the bit that is always set here
|
||||
Src = _Or(OpSize::i64Bit, Src, _InlineConstant(0b10));
|
||||
@@ -1194,17 +1193,17 @@ void OpDispatchBuilder::FLAGControlOp(OpcodeArgs) {
|
||||
CarryInvert();
|
||||
break;
|
||||
case 0xF8: // CLC
|
||||
SetCFInverted(Constant(1, ConstPad::NoPad));
|
||||
SetCFInverted(Constant(1));
|
||||
break;
|
||||
case 0xF9: // STC
|
||||
SetCFInverted(Constant(0, ConstPad::NoPad));
|
||||
SetCFInverted(Constant(0));
|
||||
break;
|
||||
case 0xFC: // CLD
|
||||
// Transformed
|
||||
StoreDF(Constant(1, ConstPad::NoPad));
|
||||
StoreDF(Constant(1));
|
||||
break;
|
||||
case 0xFD: // STD
|
||||
StoreDF(Constant(-1, ConstPad::NoPad));
|
||||
StoreDF(Constant(-1));
|
||||
break;
|
||||
}
|
||||
}
|
||||
@@ -1300,7 +1299,7 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs, bool ToSeg) {
|
||||
case FEXCore::X86State::REG_RBP: // GS
|
||||
case FEXCore::X86State::REG_R13: // GS
|
||||
if (Is64BitMode) {
|
||||
Segment = Constant(0, ConstPad::NoPad);
|
||||
Segment = Constant(0);
|
||||
} else {
|
||||
Segment = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, gs_idx));
|
||||
}
|
||||
@@ -1308,7 +1307,7 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs, bool ToSeg) {
|
||||
case FEXCore::X86State::REG_RSP: // FS
|
||||
case FEXCore::X86State::REG_R12: // FS
|
||||
if (Is64BitMode) {
|
||||
Segment = Constant(0, ConstPad::NoPad);
|
||||
Segment = Constant(0);
|
||||
} else {
|
||||
Segment = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, fs_idx));
|
||||
}
|
||||
@@ -1406,7 +1405,7 @@ void OpDispatchBuilder::SHLImmediateOp(OpcodeArgs, bool SHL1Bit) {
|
||||
uint64_t Shift = GetConstantShift(Op, SHL1Bit);
|
||||
const auto Size = GetSrcBitSize(Op);
|
||||
|
||||
Ref Result = _Lshl(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, Constant(Shift, ConstPad::NoPad));
|
||||
Ref Result = _Lshl(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, Constant(Shift));
|
||||
|
||||
CalculateFlags_ShiftLeftImmediate(OpSizeFromSrc(Op), Result, Dest, Shift);
|
||||
CalculateDeferredFlags();
|
||||
@@ -1427,7 +1426,7 @@ void OpDispatchBuilder::SHRImmediateOp(OpcodeArgs, bool SHR1Bit) {
|
||||
auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32});
|
||||
|
||||
uint64_t Shift = GetConstantShift(Op, SHR1Bit);
|
||||
auto ALUOp = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, Constant(Shift, ConstPad::NoPad));
|
||||
auto ALUOp = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, Constant(Shift));
|
||||
|
||||
CalculateFlags_ShiftRightImmediate(OpSizeFromSrc(Op), ALUOp, Dest, Shift);
|
||||
CalculateDeferredFlags();
|
||||
@@ -1456,7 +1455,7 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) {
|
||||
|
||||
// a64 masks the bottom bits, so if we're using a native 32/64-bit shift, we
|
||||
// can negate to do the subtract (it's congruent), which saves a constant.
|
||||
auto ShiftRight = Size >= 32 ? _Neg(OpSize::i64Bit, Shift) : Sub(OpSize::i64Bit, Constant(Size, ConstPad::NoPad), Shift);
|
||||
auto ShiftRight = Size >= 32 ? _Neg(OpSize::i64Bit, Shift) : Sub(OpSize::i64Bit, Constant(Size), Shift);
|
||||
|
||||
auto Tmp1 = _Lshl(OpSize::i64Bit, Dest, Shift);
|
||||
auto Tmp2 = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Src, ShiftRight);
|
||||
@@ -1472,7 +1471,7 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) {
|
||||
//
|
||||
// TODO: This whole function wants to be wrapped in the if. Maybe b/w pass is
|
||||
// a good idea after all.
|
||||
Res = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Shift, Constant(0, ConstPad::NoPad), Dest, Res);
|
||||
Res = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Shift, Constant(0), Dest, Res);
|
||||
|
||||
HandleShift(Op, Res, Dest, ShiftType::LSL, Shift);
|
||||
}
|
||||
@@ -1487,11 +1486,11 @@ void OpDispatchBuilder::SHLDImmediateOp(OpcodeArgs) {
|
||||
if (Shift != 0) {
|
||||
Ref Res {};
|
||||
if (Size < 32) {
|
||||
Ref ShiftLeft = Constant(Shift, ConstPad::NoPad);
|
||||
Ref ShiftLeft = Constant(Shift);
|
||||
auto ShiftRight = Size - Shift;
|
||||
|
||||
auto Tmp1 = _Lshl(OpSize::i64Bit, Dest, ShiftLeft);
|
||||
Ref Tmp2 = ShiftRight ? _Lshr(OpSize::i32Bit, Src, Constant(ShiftRight, ConstPad::NoPad)) : Src;
|
||||
Ref Tmp2 = ShiftRight ? _Lshr(OpSize::i32Bit, Src, Constant(ShiftRight)) : Src;
|
||||
|
||||
Res = _Or(OpSize::i64Bit, Tmp1, Tmp2);
|
||||
} else {
|
||||
@@ -1527,7 +1526,7 @@ void OpDispatchBuilder::SHRDOp(OpcodeArgs) {
|
||||
Shift = _And(OpSize::i64Bit, Shift, _InlineConstant(0x1F));
|
||||
}
|
||||
|
||||
auto ShiftLeft = Sub(OpSize::i64Bit, Constant(Size, ConstPad::NoPad), Shift);
|
||||
auto ShiftLeft = Sub(OpSize::i64Bit, Constant(Size), Shift);
|
||||
|
||||
auto Tmp1 = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, Shift);
|
||||
auto Tmp2 = _Lshl(OpSize::i64Bit, Src, ShiftLeft);
|
||||
@@ -1537,7 +1536,7 @@ void OpDispatchBuilder::SHRDOp(OpcodeArgs) {
|
||||
// If shift count was zero then output doesn't change
|
||||
// Needs to be checked for the 32bit operand case
|
||||
// where shift = 0 and the source register still gets Zext
|
||||
Res = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Shift, Constant(0, ConstPad::NoPad), Dest, Res);
|
||||
Res = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Shift, Constant(0), Dest, Res);
|
||||
|
||||
HandleShift(Op, Res, Dest, ShiftType::LSR, Shift);
|
||||
}
|
||||
@@ -1552,8 +1551,8 @@ void OpDispatchBuilder::SHRDImmediateOp(OpcodeArgs) {
|
||||
if (Shift != 0) {
|
||||
Ref Res {};
|
||||
if (Size < 32) {
|
||||
Ref ShiftRight = Constant(Shift, ConstPad::NoPad);
|
||||
auto ShiftLeft = Constant(Size - Shift, ConstPad::NoPad);
|
||||
Ref ShiftRight = Constant(Shift);
|
||||
auto ShiftLeft = Constant(Size - Shift);
|
||||
|
||||
auto Tmp1 = _Lshr(OpSize::i32Bit, Dest, ShiftRight);
|
||||
auto Tmp2 = _Lshl(OpSize::i64Bit, Src, ShiftLeft);
|
||||
@@ -1588,7 +1587,7 @@ void OpDispatchBuilder::ASHROp(OpcodeArgs, bool Immediate, bool SHR1Bit) {
|
||||
|
||||
if (Immediate) {
|
||||
uint64_t Shift = GetConstantShift(Op, SHR1Bit);
|
||||
Ref Result = _Ashr(OpSize, Dest, Constant(Shift, ConstPad::NoPad));
|
||||
Ref Result = _Ashr(OpSize, Dest, Constant(Shift));
|
||||
|
||||
CalculateFlags_SignShiftRightImmediate(OpSizeFromSrc(Op), Result, Dest, Shift);
|
||||
CalculateDeferredFlags();
|
||||
@@ -1682,21 +1681,21 @@ void OpDispatchBuilder::BEXTRBMIOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
const auto SrcSize = IR::OpSizeAsBits(Size);
|
||||
const auto MaxSrcBit = SrcSize - 1;
|
||||
auto MaxSrcBitOp = Constant(MaxSrcBit, ConstPad::NoPad);
|
||||
auto MaxSrcBitOp = Constant(MaxSrcBit);
|
||||
|
||||
// Shift the operand down to the starting bit
|
||||
auto Start = _Bfe(OpSizeFromSrc(Op), 8, 0, Src2);
|
||||
auto Shifted = _Lshr(Size, Src1, Start);
|
||||
|
||||
// Shifts larger than operand size need to be set to zero.
|
||||
auto SanitizedShifted = _Select(Size, Size, CondClass::ULE, Start, MaxSrcBitOp, Shifted, Constant(0, ConstPad::NoPad));
|
||||
auto SanitizedShifted = _Select(Size, Size, CondClass::ULE, Start, MaxSrcBitOp, Shifted, Constant(0));
|
||||
|
||||
// Now handle the length specifier.
|
||||
auto Length = _Bfe(Size, 8, 8, Src2);
|
||||
|
||||
// Now build up the mask
|
||||
// (1 << Length) - 1 = ~(~0 << Length)
|
||||
auto AllOnes = Constant(~0ull, ConstPad::NoPad);
|
||||
auto AllOnes = Constant(~0ull);
|
||||
auto InvertedMask = _Lshl(Size, AllOnes, Length);
|
||||
|
||||
// Now put it all together and make the result.
|
||||
@@ -1811,7 +1810,7 @@ void OpDispatchBuilder::BZHI(OpcodeArgs) {
|
||||
// Clear the high bits specified by the index. A64 only considers bottom bits
|
||||
// of the shift, so we don't need to mask bottom 8-bits ourselves.
|
||||
// Out-of-bounds results ignored after.
|
||||
auto Mask = _Lshl(Size, Constant(-1, ConstPad::NoPad), Index);
|
||||
auto Mask = _Lshl(Size, Constant(-1), Index);
|
||||
auto MaskResult = _Andn(Size, Src, Mask);
|
||||
|
||||
// If the index is above OperandSize, we don't clear anything. BZHI only
|
||||
@@ -1820,7 +1819,7 @@ void OpDispatchBuilder::BZHI(OpcodeArgs) {
|
||||
//
|
||||
// Because we're clobbering flags internally we ignore all carry invert
|
||||
// shenanigans and use the raw versions here.
|
||||
_TestNZ(OpSize::i64Bit, Index, Constant(0xFF & ~(OperandSize - 1), ConstPad::NoPad));
|
||||
_TestNZ(OpSize::i64Bit, Index, Constant(0xFF & ~(OperandSize - 1)));
|
||||
auto Result = _NZCVSelect(Size, CondClass::NEQ, Src, MaskResult);
|
||||
StoreResultGPR(Op, Result);
|
||||
|
||||
@@ -1914,7 +1913,7 @@ void OpDispatchBuilder::ADXOp(OpcodeArgs) {
|
||||
|
||||
// Handles ADCX and ADOX
|
||||
const bool IsADCX = Op->OP == 0x1F6;
|
||||
auto Zero = Constant(0, ConstPad::NoPad);
|
||||
auto Zero = Constant(0);
|
||||
|
||||
// Before we go trashing NZCV, save the current NZCV state.
|
||||
Ref OldNZCV = GetNZCV();
|
||||
@@ -2334,7 +2333,7 @@ void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) {
|
||||
|
||||
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
|
||||
Ref Tmp = Constant(0, ConstPad::NoPad);
|
||||
Ref Tmp = Constant(0);
|
||||
|
||||
for (size_t i = 0; i < (32 + Size + 1); i += (Size + 1)) {
|
||||
// Insert incoming value
|
||||
@@ -2713,9 +2712,9 @@ void OpDispatchBuilder::NOTOp(OpcodeArgs) {
|
||||
|
||||
Ref MaskConst {};
|
||||
if (Size == OpSize::i64Bit) {
|
||||
MaskConst = Constant(~0ULL, ConstPad::NoPad);
|
||||
MaskConst = Constant(~0ULL);
|
||||
} else {
|
||||
MaskConst = Constant((1ULL << SizeBits) - 1, ConstPad::NoPad);
|
||||
MaskConst = Constant((1ULL << SizeBits) - 1);
|
||||
}
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
@@ -2734,7 +2733,7 @@ void OpDispatchBuilder::NOTOp(OpcodeArgs) {
|
||||
auto Dest = Op->Dest;
|
||||
if (Dest.Data.GPR.HighBits) {
|
||||
LOGMAN_THROW_A_FMT(Size == OpSize::i8Bit, "Only 8-bit GPRs get high bits");
|
||||
MaskConst = Constant(0xFF00, ConstPad::NoPad);
|
||||
MaskConst = Constant(0xFF00);
|
||||
Dest.Data.GPR.HighBits = false;
|
||||
}
|
||||
|
||||
@@ -2796,8 +2795,8 @@ void OpDispatchBuilder::PopcountOp(OpcodeArgs) {
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateAFForDecimal(Ref A) {
|
||||
auto Nibble = _And(OpSize::i64Bit, A, Constant(0xF, ConstPad::NoPad));
|
||||
auto Greater = Select01(OpSize::i64Bit, CondClass::UGT, Nibble, Constant(9, ConstPad::NoPad));
|
||||
auto Nibble = _And(OpSize::i64Bit, A, Constant(0xF));
|
||||
auto Greater = Select01(OpSize::i64Bit, CondClass::UGT, Nibble, Constant(9));
|
||||
|
||||
return _Or(OpSize::i64Bit, LoadAF(), Greater);
|
||||
}
|
||||
@@ -2809,13 +2808,13 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
auto AF = CalculateAFForDecimal(AL);
|
||||
|
||||
// CF |= (AL > 0x99);
|
||||
CFInv = _And(OpSize::i64Bit, CFInv, Select01(OpSize::i64Bit, CondClass::ULE, AL, Constant(0x99, ConstPad::NoPad)));
|
||||
CFInv = _And(OpSize::i64Bit, CFInv, Select01(OpSize::i64Bit, CondClass::ULE, AL, Constant(0x99)));
|
||||
|
||||
// AL = AF ? (AL + 0x6) : AL;
|
||||
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::NEQ, AF, Constant(0, ConstPad::NoPad), Add(OpSize::i64Bit, AL, 0x6), AL);
|
||||
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::NEQ, AF, Constant(0), Add(OpSize::i64Bit, AL, 0x6), AL);
|
||||
|
||||
// AL = CF ? (AL + 0x60) : AL;
|
||||
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, CFInv, Constant(0, ConstPad::NoPad), Add(OpSize::i64Bit, AL, 0x60), AL);
|
||||
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, CFInv, Constant(0), Add(OpSize::i64Bit, AL, 0x60), AL);
|
||||
|
||||
// SF, ZF, PF set according to result. CF set per above. OF undefined.
|
||||
StoreGPRRegister(X86State::REG_RAX, AL, OpSize::i8Bit);
|
||||
@@ -2832,16 +2831,16 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
auto AF = CalculateAFForDecimal(AL);
|
||||
|
||||
// CF |= (AL > 0x99);
|
||||
CF = _Or(OpSize::i64Bit, CF, Select01(OpSize::i64Bit, CondClass::UGT, AL, Constant(0x99, ConstPad::NoPad)));
|
||||
CF = _Or(OpSize::i64Bit, CF, Select01(OpSize::i64Bit, CondClass::UGT, AL, Constant(0x99)));
|
||||
|
||||
// NewCF = CF | (AF && (Borrow from AL - 6))
|
||||
auto NewCF = _Or(OpSize::i32Bit, CF, _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::ULT, AL, Constant(6, ConstPad::NoPad), AF, CF));
|
||||
auto NewCF = _Or(OpSize::i32Bit, CF, _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::ULT, AL, Constant(6), AF, CF));
|
||||
|
||||
// AL = AF ? (AL - 0x6) : AL;
|
||||
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::NEQ, AF, Constant(0, ConstPad::NoPad), Sub(OpSize::i64Bit, AL, 0x6), AL);
|
||||
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::NEQ, AF, Constant(0), Sub(OpSize::i64Bit, AL, 0x6), AL);
|
||||
|
||||
// AL = CF ? (AL - 0x60) : AL;
|
||||
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::NEQ, CF, Constant(0, ConstPad::NoPad), Sub(OpSize::i64Bit, AL, 0x60), AL);
|
||||
AL = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::NEQ, CF, Constant(0), Sub(OpSize::i64Bit, AL, 0x60), AL);
|
||||
|
||||
// SF, ZF, PF set according to result. CF set per above. OF undefined.
|
||||
StoreGPRRegister(X86State::REG_RAX, AL, OpSize::i8Bit);
|
||||
@@ -2864,7 +2863,7 @@ void OpDispatchBuilder::AAAOp(OpcodeArgs) {
|
||||
A = NZCVSelect(OpSize::i32Bit, CondClass::UGE /* CF = 1 */, Add(OpSize::i32Bit, A, 0x106), A);
|
||||
|
||||
// AL = AL & 0x0F
|
||||
A = _And(OpSize::i32Bit, A, Constant(0xFF0F, ConstPad::NoPad));
|
||||
A = _And(OpSize::i32Bit, A, Constant(0xFF0F));
|
||||
StoreGPRRegister(X86State::REG_RAX, A, OpSize::i16Bit);
|
||||
}
|
||||
|
||||
@@ -2881,13 +2880,13 @@ void OpDispatchBuilder::AASOp(OpcodeArgs) {
|
||||
A = NZCVSelect(OpSize::i32Bit, CondClass::UGE /* CF = 1 */, Sub(OpSize::i32Bit, A, 0x106), A);
|
||||
|
||||
// AL = AL & 0x0F
|
||||
A = _And(OpSize::i32Bit, A, Constant(0xFF0F, ConstPad::NoPad));
|
||||
A = _And(OpSize::i32Bit, A, Constant(0xFF0F));
|
||||
StoreGPRRegister(X86State::REG_RAX, A, OpSize::i16Bit);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AAMOp(OpcodeArgs) {
|
||||
auto AL = LoadGPRRegister(X86State::REG_RAX, OpSize::i8Bit);
|
||||
auto Imm8 = Constant(Op->Src[0].Literal() & 0xFF, ConstPad::NoPad);
|
||||
auto Imm8 = Constant(Op->Src[0].Literal() & 0xFF);
|
||||
Ref Quotient = _AllocateGPR(true);
|
||||
Ref Remainder = _AllocateGPR(true);
|
||||
_UDiv(OpSize::i64Bit, AL, Invalid(), Imm8, Quotient, Remainder);
|
||||
@@ -2901,10 +2900,10 @@ void OpDispatchBuilder::AAMOp(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::AADOp(OpcodeArgs) {
|
||||
auto A = LoadGPRRegister(X86State::REG_RAX);
|
||||
auto AH = _Lshr(OpSize::i32Bit, A, Constant(8, ConstPad::NoPad));
|
||||
auto Imm8 = Constant(Op->Src[0].Literal() & 0xFF, ConstPad::NoPad);
|
||||
auto AH = _Lshr(OpSize::i32Bit, A, Constant(8));
|
||||
auto Imm8 = Constant(Op->Src[0].Literal() & 0xFF);
|
||||
auto NewAL = Add(OpSize::i64Bit, A, _Mul(OpSize::i64Bit, AH, Imm8));
|
||||
auto Result = _And(OpSize::i64Bit, NewAL, Constant(0xFF, ConstPad::NoPad));
|
||||
auto Result = _And(OpSize::i64Bit, NewAL, Constant(0xFF));
|
||||
StoreGPRRegister(X86State::REG_RAX, Result, OpSize::i16Bit);
|
||||
|
||||
SetNZ_ZeroCV(OpSize::i8Bit, Result);
|
||||
@@ -3001,9 +3000,8 @@ void OpDispatchBuilder::SGDTOp(OpcodeArgs) {
|
||||
GDTStoreSize = OpSize::i32Bit;
|
||||
}
|
||||
|
||||
_StoreMemGPRAutoTSO(OpSize::i16Bit, DestAddress, Constant(0, ConstPad::NoPad));
|
||||
_StoreMemGPRAutoTSO(GDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit},
|
||||
Constant(GDTAddress, ConstPad::NoPad));
|
||||
_StoreMemGPRAutoTSO(OpSize::i16Bit, DestAddress, Constant(0));
|
||||
_StoreMemGPRAutoTSO(GDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, Constant(GDTAddress));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SIDTOp(OpcodeArgs) {
|
||||
@@ -3018,30 +3016,28 @@ void OpDispatchBuilder::SIDTOp(OpcodeArgs) {
|
||||
IDTStoreSize = OpSize::i32Bit;
|
||||
}
|
||||
|
||||
_StoreMemGPRAutoTSO(OpSize::i16Bit, DestAddress, Constant(0xfff, ConstPad::NoPad));
|
||||
_StoreMemGPRAutoTSO(IDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit},
|
||||
Constant(IDTAddress, ConstPad::NoPad));
|
||||
_StoreMemGPRAutoTSO(OpSize::i16Bit, DestAddress, Constant(0xfff));
|
||||
_StoreMemGPRAutoTSO(IDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, Constant(IDTAddress));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SMSWOp(OpcodeArgs) {
|
||||
const bool IsMemDst = DestIsMem(Op);
|
||||
|
||||
IR::OpSize DstSize {OpSize::iInvalid};
|
||||
Ref Const = Constant((1U << 31) | ///< PG - Paging
|
||||
(0U << 30) | ///< CD - Cache Disable
|
||||
(0U << 29) | ///< NW - Not Writethrough (Legacy, now ignored)
|
||||
///< [28:19] - Reserved
|
||||
(1U << 18) | ///< AM - Alignment Mask
|
||||
///< 17 - Reserved
|
||||
(1U << 16) | ///< WP - Write Protect
|
||||
///< [15:6] - Reserved
|
||||
(1U << 5) | ///< NE - Numeric Error
|
||||
(1U << 4) | ///< ET - Extension Type (Legacy, now reserved and 1)
|
||||
(0U << 3) | ///< TS - Task Switched
|
||||
(0U << 2) | ///< EM - Emulation
|
||||
(1U << 1) | ///< MP - Monitor Coprocessor
|
||||
(1U << 0), ///< PE - Protection Enabled
|
||||
ConstPad::NoPad);
|
||||
Ref Const = Constant((1U << 31) | ///< PG - Paging
|
||||
(0U << 30) | ///< CD - Cache Disable
|
||||
(0U << 29) | ///< NW - Not Writethrough (Legacy, now ignored)
|
||||
///< [28:19] - Reserved
|
||||
(1U << 18) | ///< AM - Alignment Mask
|
||||
///< 17 - Reserved
|
||||
(1U << 16) | ///< WP - Write Protect
|
||||
///< [15:6] - Reserved
|
||||
(1U << 5) | ///< NE - Numeric Error
|
||||
(1U << 4) | ///< ET - Extension Type (Legacy, now reserved and 1)
|
||||
(0U << 3) | ///< TS - Task Switched
|
||||
(0U << 2) | ///< EM - Emulation
|
||||
(1U << 1) | ///< MP - Monitor Coprocessor
|
||||
(1U << 0)); ///< PE - Protection Enabled
|
||||
const auto OpAddr = X86Tables::DecodeFlags::GetOpAddr(Op->Flags, 0);
|
||||
if (Is64BitMode) {
|
||||
DstSize = OpAddr == X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST ? OpSize::i16Bit :
|
||||
@@ -3072,8 +3068,8 @@ OpDispatchBuilder::CycleCounterPair OpDispatchBuilder::CycleCounter(bool SelfSyn
|
||||
Ref CounterHigh {};
|
||||
auto Counter = _CycleCounter(SelfSynchronizingLoads);
|
||||
if (CTX->Config.TSCScale) {
|
||||
CounterLow = _Lshl(OpSize::i32Bit, Counter, Constant(CTX->Config.TSCScale, ConstPad::NoPad));
|
||||
CounterHigh = _Lshr(OpSize::i64Bit, Counter, Constant(32 - CTX->Config.TSCScale, ConstPad::NoPad));
|
||||
CounterLow = _Lshl(OpSize::i32Bit, Counter, Constant(CTX->Config.TSCScale));
|
||||
CounterHigh = _Lshr(OpSize::i64Bit, Counter, Constant(32 - CTX->Config.TSCScale));
|
||||
} else {
|
||||
CounterLow = _Bfe(OpSize::i64Bit, 32, 0, Counter);
|
||||
CounterHigh = _Bfe(OpSize::i64Bit, 32, 32, Counter);
|
||||
@@ -3101,7 +3097,7 @@ void OpDispatchBuilder::INCOp(OpcodeArgs) {
|
||||
HandledLock = true;
|
||||
|
||||
Ref DestAddress = MakeSegmentAddress(Op, Op->Dest);
|
||||
Dest = _AtomicFetchAdd(OpSizeFromSrc(Op), Constant(1, ConstPad::NoPad), DestAddress);
|
||||
Dest = _AtomicFetchAdd(OpSizeFromSrc(Op), Constant(1), DestAddress);
|
||||
} else {
|
||||
Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32});
|
||||
}
|
||||
@@ -3112,7 +3108,7 @@ void OpDispatchBuilder::INCOp(OpcodeArgs) {
|
||||
// Addition producing upper garbage
|
||||
Result = Add(OpSize::i32Bit, Dest, 1);
|
||||
CalculatePF(Result);
|
||||
CalculateAF(Dest, Constant(1, ConstPad::NoPad));
|
||||
CalculateAF(Dest, Constant(1));
|
||||
|
||||
// Correctly set NZ flags, preserving C
|
||||
HandleNZCV_RMW();
|
||||
@@ -3122,7 +3118,7 @@ void OpDispatchBuilder::INCOp(OpcodeArgs) {
|
||||
// getting a negative. So compare the sign bits to calculate V.
|
||||
_RmifNZCV(_Andn(OpSize::i32Bit, Result, Dest), Size - 1, 1);
|
||||
} else {
|
||||
Result = CalculateFlags_ADD(OpSizeFromSrc(Op), Dest, Constant(1, ConstPad::NoPad), false);
|
||||
Result = CalculateFlags_ADD(OpSizeFromSrc(Op), Dest, Constant(1), false);
|
||||
}
|
||||
|
||||
if (!IsLocked) {
|
||||
@@ -3142,7 +3138,7 @@ void OpDispatchBuilder::DECOp(OpcodeArgs) {
|
||||
Ref DestAddress = MakeSegmentAddress(Op, Op->Dest);
|
||||
|
||||
// Use Add instead of Sub to avoid a NEG
|
||||
Dest = _AtomicFetchAdd(OpSizeFromSrc(Op), Constant(Size == 64 ? -1 : ((1ULL << Size) - 1), ConstPad::NoPad), DestAddress);
|
||||
Dest = _AtomicFetchAdd(OpSizeFromSrc(Op), Constant(Size == 64 ? -1 : ((1ULL << Size) - 1)), DestAddress);
|
||||
} else {
|
||||
Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32});
|
||||
}
|
||||
@@ -3153,7 +3149,7 @@ void OpDispatchBuilder::DECOp(OpcodeArgs) {
|
||||
// Subtraction producing upper garbage
|
||||
Result = Sub(OpSize::i32Bit, Dest, 1);
|
||||
CalculatePF(Result);
|
||||
CalculateAF(Dest, Constant(1, ConstPad::NoPad));
|
||||
CalculateAF(Dest, Constant(1));
|
||||
|
||||
// Correctly set NZ flags, preserving C
|
||||
HandleNZCV_RMW();
|
||||
@@ -3163,7 +3159,7 @@ void OpDispatchBuilder::DECOp(OpcodeArgs) {
|
||||
// getting a positive. So compare the sign bits to calculate V.
|
||||
_RmifNZCV(_Andn(OpSize::i32Bit, Dest, Result), Size - 1, 1);
|
||||
} else {
|
||||
Result = CalculateFlags_SUB(OpSizeFromSrc(Op), Dest, Constant(1, ConstPad::NoPad), false);
|
||||
Result = CalculateFlags_SUB(OpSizeFromSrc(Op), Dest, Constant(1), false);
|
||||
}
|
||||
|
||||
if (!IsLocked) {
|
||||
@@ -3211,7 +3207,7 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) {
|
||||
Ref Counter = LoadGPRRegister(X86State::REG_RCX);
|
||||
|
||||
auto Result = _MemSet(CTX->IsAtomicTSOEnabled(), Size, Segment ?: InvalidNode, Dest, Src, Counter, LoadDir(1));
|
||||
StoreGPRRegister(X86State::REG_RCX, Constant(0, ConstPad::NoPad));
|
||||
StoreGPRRegister(X86State::REG_RCX, Constant(0));
|
||||
StoreGPRRegister(X86State::REG_RDI, Result);
|
||||
}
|
||||
}
|
||||
@@ -3254,7 +3250,7 @@ void OpDispatchBuilder::MOVSOp(OpcodeArgs) {
|
||||
Result_Src = Sub(OpSize::i64Bit, Result_Src, SrcSegment);
|
||||
}
|
||||
|
||||
StoreGPRRegister(X86State::REG_RCX, Constant(0, ConstPad::NoPad));
|
||||
StoreGPRRegister(X86State::REG_RCX, Constant(0));
|
||||
StoreGPRRegister(X86State::REG_RDI, Result_Dst);
|
||||
StoreGPRRegister(X86State::REG_RSI, Result_Src);
|
||||
} else {
|
||||
@@ -3610,7 +3606,7 @@ void OpDispatchBuilder::BSWAPOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
if (Size == OpSize::i16Bit) {
|
||||
// BSWAP of 16bit is undef. ZEN+ causes the lower 16bits to get zero'd
|
||||
Dest = Constant(0, ConstPad::NoPad);
|
||||
Dest = Constant(0);
|
||||
} else {
|
||||
Dest = LoadSourceGPR_WithOpSize(Op, Op->Dest, GetGPROpSize(), Op->Flags);
|
||||
Dest = _Rev(Size, Dest);
|
||||
@@ -3632,7 +3628,7 @@ void OpDispatchBuilder::POPFOp(OpcodeArgs) {
|
||||
// Bit 1 is always 1
|
||||
// Bit 9 is always 1 because we always have interrupts enabled
|
||||
|
||||
Src = _Or(OpSize::i64Bit, Src, Constant(0x202, ConstPad::NoPad));
|
||||
Src = _Or(OpSize::i64Bit, Src, Constant(0x202));
|
||||
|
||||
SetPackedRFLAG(false, Src);
|
||||
|
||||
@@ -3645,7 +3641,7 @@ void OpDispatchBuilder::NEGOp(OpcodeArgs) {
|
||||
HandledLock = (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK) != 0;
|
||||
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
auto ZeroConst = Constant(0, ConstPad::NoPad);
|
||||
auto ZeroConst = Constant(0);
|
||||
|
||||
if (DestIsLockedMem(Op)) {
|
||||
Ref DestMem = MakeSegmentAddress(Op, Op->Dest);
|
||||
@@ -4151,14 +4147,14 @@ void OpDispatchBuilder::UpdatePrefixFromSegment(Ref Segment, uint32_t SegmentReg
|
||||
// In some cases the upper 16-bits of the 32-bit GPR contain garbage to ignore.
|
||||
auto GDT = _Bfe(OpSize::i32Bit, 1, 2, Segment);
|
||||
// Fun quirk, if we mask the selector then it is premultiplied by 8 which we need to do for accessing anyway.
|
||||
auto SegmentOffset = _And(OpSize::i32Bit, Segment, _Constant(0xfff8, ConstPad::NoPad));
|
||||
auto SegmentOffset = _And(OpSize::i32Bit, Segment, _Constant(0xfff8));
|
||||
Ref SegmentBase = _LoadContextGPRIndexed(GDT, OpSize::i64Bit, offsetof(FEXCore::Core::CPUState, segment_arrays[0]), 8);
|
||||
Ref NewSegment = _LoadMemGPR(OpSize::i64Bit, SegmentBase, SegmentOffset, OpSize::i8Bit, MemOffsetType::UXTW, 1);
|
||||
CheckLegacySegmentWrite(NewSegment, SegmentReg);
|
||||
|
||||
// Extract the 32-bit base from the GDT segment.
|
||||
auto Upper32 = _Lshr(OpSize::i64Bit, NewSegment, _Constant(32, ConstPad::NoPad));
|
||||
auto Masked = _And(OpSize::i32Bit, Upper32, _Constant(0xFF00'0000, ConstPad::NoPad));
|
||||
auto Upper32 = _Lshr(OpSize::i64Bit, NewSegment, _Constant(32));
|
||||
auto Masked = _And(OpSize::i32Bit, Upper32, _Constant(0xFF00'0000));
|
||||
Ref Merged = _Orlshr(OpSize::i32Bit, Masked, NewSegment, 16);
|
||||
NewSegment = _Bfi(OpSize::i32Bit, 8, 16, Merged, Upper32);
|
||||
|
||||
@@ -4333,7 +4329,7 @@ Ref OpDispatchBuilder::LoadGPRRegister(uint32_t GPR, IR::OpSize Size, uint8_t Of
|
||||
// Extract the subregister if requested.
|
||||
const auto OpSize = std::max(OpSize::i32Bit, Size);
|
||||
if (AllowUpperGarbage) {
|
||||
Reg = _Lshr(OpSize, Reg, Constant(Offset, ConstPad::NoPad));
|
||||
Reg = _Lshr(OpSize, Reg, Constant(Offset));
|
||||
} else {
|
||||
Reg = _Bfe(OpSize, IR::OpSizeAsBits(Size), Offset, Reg);
|
||||
}
|
||||
@@ -4440,7 +4436,7 @@ void OpDispatchBuilder::StoreResult_WithOpSize(RegClass Class, FEXCore::X86Table
|
||||
// For X87 extended doubles, split before storing
|
||||
_StoreMemFPR(OpSize::i64Bit, MemStoreDst, Src, Align);
|
||||
auto Upper = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, Src, 1);
|
||||
_StoreMemGPR(OpSize::i16Bit, Upper, MemStoreDst, Constant(8, ConstPad::NoPad), std::min(Align, OpSize::i64Bit), MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(OpSize::i16Bit, Upper, MemStoreDst, Constant(8), std::min(Align, OpSize::i64Bit), MemOffsetType::SXTX, 1);
|
||||
}
|
||||
} else {
|
||||
_StoreMemAutoTSO(Class, OpSize, A, Src, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
@@ -4516,7 +4512,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
FlushRegisterCache();
|
||||
|
||||
// Move 0 into the register
|
||||
StoreResultGPR(Op, Constant(0, ConstPad::NoPad));
|
||||
StoreResultGPR(Op, Constant(0));
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -4551,7 +4547,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
// adjusted constant here will inline into the arm64 and instruction, so if
|
||||
// flags are not needed, we save an instruction overall.
|
||||
if (ALUIROp == IR::IROps::OP_ANDWITHFLAGS) {
|
||||
Src = Constant(Const | ~((1ull << IR::OpSizeAsBits(Size)) - 1), ConstPad::NoPad);
|
||||
Src = Constant(Const | ~((1ull << IR::OpSizeAsBits(Size)) - 1));
|
||||
ALUIROp = IR::IROps::OP_AND;
|
||||
}
|
||||
}
|
||||
@@ -4606,7 +4602,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
void OpDispatchBuilder::LSLOp(OpcodeArgs) {
|
||||
// Emulate by always returning failure, this deviates from both Linux and Windows but
|
||||
// shouldn't be depended on by anything.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(Constant(0, ConstPad::NoPad));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_ZF_RAW_LOC>(Constant(0));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::INTOp(OpcodeArgs) {
|
||||
|
||||
@@ -244,7 +244,7 @@ public:
|
||||
template<typename F>
|
||||
void ForeachDirection(F&& Routine) {
|
||||
// Otherwise, prepare to branch.
|
||||
auto Zero = Constant(0, ConstPad::NoPad);
|
||||
auto Zero = Constant(0);
|
||||
|
||||
// If the shift is zero, do not touch the flags.
|
||||
auto ForwardBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
@@ -1172,7 +1172,7 @@ public:
|
||||
auto Const = Header->C<IR::IROp_LoadNamedVectorConstant>();
|
||||
|
||||
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
|
||||
Ref Zero = _Constant(0, ConstPad::NoPad);
|
||||
Ref Zero = _Constant(0);
|
||||
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, RegClass::GPR, Zero, Zero, Offset);
|
||||
|
||||
// XXX: This works around InlineConstant not having an associated
|
||||
@@ -1263,7 +1263,7 @@ public:
|
||||
StoreContextHelper(Size, Class, Value, Offset);
|
||||
// If Partial and MMX register, then we need to store all 1s in bits 64-80
|
||||
if (Partial && Index >= MM0Index && Index <= MM7Index) {
|
||||
_StoreContextGPR(OpSize::i16Bit, Constant(0xFFFF, ConstPad::NoPad), Offset + 8);
|
||||
_StoreContextGPR(OpSize::i16Bit, Constant(0xFFFF), Offset + 8);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -1697,7 +1697,7 @@ private:
|
||||
}
|
||||
|
||||
void ZeroNZCV() {
|
||||
CachedNZCV = Constant(0, ConstPad::NoPad);
|
||||
CachedNZCV = Constant(0);
|
||||
NZCVDirty = true;
|
||||
}
|
||||
|
||||
@@ -1712,7 +1712,7 @@ private:
|
||||
if (SetPF) {
|
||||
CalculatePF(SubWithFlags(SrcSize, Res, (uint64_t)0));
|
||||
} else {
|
||||
_SubNZCV(SrcSize, Res, Constant(0, ConstPad::NoPad));
|
||||
_SubNZCV(SrcSize, Res, Constant(0));
|
||||
}
|
||||
|
||||
CFInverted = true;
|
||||
@@ -1783,7 +1783,7 @@ private:
|
||||
} else {
|
||||
// Invert as a GPR
|
||||
unsigned Bit = IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC);
|
||||
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), Constant(1u << Bit, ConstPad::NoPad)));
|
||||
SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), Constant(1u << Bit)));
|
||||
CalculateDeferredFlags();
|
||||
}
|
||||
|
||||
@@ -1821,7 +1821,7 @@ private:
|
||||
}
|
||||
|
||||
HandleNZCVWrite();
|
||||
_SubNZCV(OpSize::i32Bit, Constant(0, ConstPad::NoPad), Value);
|
||||
_SubNZCV(OpSize::i32Bit, Constant(0), Value);
|
||||
CFInverted = true;
|
||||
}
|
||||
|
||||
@@ -1846,14 +1846,14 @@ private:
|
||||
StoreRegister(Core::CPUState::AF_AS_GREG, false, Value);
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
|
||||
// For DF, we need to transform 0/1 into 1/-1
|
||||
StoreDF(_SubShift(OpSize::i64Bit, Constant(1, ConstPad::NoPad), Value, ShiftType::LSL, 1));
|
||||
StoreDF(_SubShift(OpSize::i64Bit, Constant(1), Value, ShiftType::LSL, 1));
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_TF_RAW_LOC) {
|
||||
auto PackedTF = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
// An exception should still be raised after an instruction that unsets TF, leave the unblocked bit set but unset
|
||||
// the TF bit to cause such behaviour. The handling code at the start of the next block will then unset the
|
||||
// unblocked bit before raising the exception.
|
||||
auto NewPackedTF = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Value, Constant(0, ConstPad::NoPad),
|
||||
_And(OpSize::i32Bit, PackedTF, Constant(~1, ConstPad::NoPad)), Constant(1, ConstPad::NoPad));
|
||||
auto NewPackedTF =
|
||||
_Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, Value, Constant(0), _And(OpSize::i32Bit, PackedTF, Constant(~1)), Constant(1));
|
||||
_StoreContextGPR(OpSize::i8Bit, NewPackedTF, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
} else {
|
||||
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags[BitOffset]));
|
||||
@@ -1865,7 +1865,7 @@ private:
|
||||
// bits. This allows us to defer the extract in the usual case. When it is
|
||||
// read, bit 4 is extracted. In order to write a constant value of AF, that
|
||||
// means we need to left-shift here to compensate.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(K << 4, ConstPad::NoPad));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(K << 4));
|
||||
}
|
||||
|
||||
void ZeroPF_AF();
|
||||
@@ -2089,7 +2089,7 @@ private:
|
||||
auto Value = _Bfe(OpSize::i32Bit, 1, IndexNZCV(BitOffset), GetNZCV());
|
||||
|
||||
if (Invert) {
|
||||
return _Xor(OpSize::i32Bit, Value, Constant(1, ConstPad::NoPad));
|
||||
return _Xor(OpSize::i32Bit, Value, Constant(1));
|
||||
} else {
|
||||
return Value;
|
||||
}
|
||||
@@ -2104,7 +2104,7 @@ private:
|
||||
return LoadGPR(Core::CPUState::AF_AS_GREG);
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
|
||||
// Recover the sign bit, it is the logical DF value
|
||||
return _Lshr(OpSize::i64Bit, LoadDF(), Constant(63, ConstPad::NoPad));
|
||||
return _Lshr(OpSize::i64Bit, LoadDF(), Constant(63));
|
||||
} else {
|
||||
return _LoadContextGPR(OpSize::i8Bit, offsetof(Core::CPUState, flags[BitOffset]));
|
||||
}
|
||||
@@ -2176,7 +2176,7 @@ private:
|
||||
// Zero AF. Note that the comparison sets the raw PF to 0/1 above, so
|
||||
// PF[4] is 0 so the XOR with PF will have no effect, so setting the AF
|
||||
// byte to zero will indeed zero AF as intended.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0, ConstPad::NoPad));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(Constant(0));
|
||||
}
|
||||
|
||||
// Convert NZCV from the Arm representation to an eXternal representation
|
||||
@@ -2210,7 +2210,7 @@ private:
|
||||
}
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(C);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0, ConstPad::NoPad));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(V);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(Z);
|
||||
}
|
||||
@@ -2332,7 +2332,7 @@ private:
|
||||
}
|
||||
|
||||
// Otherwise, prepare to branch.
|
||||
auto Zero = Constant(0, ConstPad::NoPad);
|
||||
auto Zero = Constant(0);
|
||||
|
||||
// If the shift is zero, do not touch the flags.
|
||||
auto SetBlock = CreateNewCodeBlockAfter(GetCurrentBlock());
|
||||
@@ -2405,8 +2405,8 @@ private:
|
||||
void ChgStateX87_MMX() override {
|
||||
LOGMAN_THROW_A_FMT(MMXState == MMXState_X87, "Expected state to be x87");
|
||||
_StackForceSlow();
|
||||
SetX87Top(Constant(0, ConstPad::NoPad)); // top reset to zero
|
||||
_StoreContextGPR(OpSize::i8Bit, Constant(0xFFFFUL, ConstPad::NoPad), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
SetX87Top(Constant(0)); // top reset to zero
|
||||
_StoreContextGPR(OpSize::i8Bit, Constant(0xFFFFUL), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
MMXState = MMXState_MMX;
|
||||
}
|
||||
|
||||
@@ -2639,11 +2639,11 @@ private:
|
||||
}
|
||||
|
||||
ArithRef And(uint64_t K) {
|
||||
return IsConstant ? ArithRef(E, C & K) : ArithRef(E, E->_And(OpSize::i64Bit, R, E->Constant(K, ConstPad::NoPad)));
|
||||
return IsConstant ? ArithRef(E, C & K) : ArithRef(E, E->_And(OpSize::i64Bit, R, E->Constant(K)));
|
||||
}
|
||||
|
||||
ArithRef Presub(uint64_t K) {
|
||||
return IsConstant ? ArithRef(E, K - C) : ArithRef(E, E->Sub(OpSize::i64Bit, E->Constant(K, ConstPad::NoPad), R));
|
||||
return IsConstant ? ArithRef(E, K - C) : ArithRef(E, E->Sub(OpSize::i64Bit, E->Constant(K), R));
|
||||
}
|
||||
|
||||
ArithRef Lshl(uint64_t Shift) {
|
||||
@@ -2652,7 +2652,7 @@ private:
|
||||
} else if (IsConstant) {
|
||||
return ArithRef(E, C << Shift);
|
||||
} else {
|
||||
return ArithRef(E, E->_Lshl(OpSize::i64Bit, R, E->Constant(Shift, ConstPad::NoPad)));
|
||||
return ArithRef(E, E->_Lshl(OpSize::i64Bit, R, E->Constant(Shift)));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2692,7 +2692,7 @@ private:
|
||||
}
|
||||
|
||||
if (IsConstant) {
|
||||
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, E->Constant(C, ConstPad::NoPad));
|
||||
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, E->Constant(C));
|
||||
} else {
|
||||
return E->_Bfi(OpSize::i64Bit, Size, Start, Bitfield, R);
|
||||
}
|
||||
@@ -2708,12 +2708,12 @@ private:
|
||||
|
||||
return ArithRef(E, Result);
|
||||
} else {
|
||||
return ArithRef(E, E->_Lshl(Size, E->Constant(1, ConstPad::NoPad), R));
|
||||
return ArithRef(E, E->_Lshl(Size, E->Constant(1), R));
|
||||
}
|
||||
}
|
||||
|
||||
Ref Ref() {
|
||||
return IsConstant ? E->Constant(C, ConstPad::NoPad) : R;
|
||||
return IsConstant ? E->Constant(C) : R;
|
||||
}
|
||||
|
||||
bool IsDefinitelyZero() const {
|
||||
|
||||
@@ -845,7 +845,7 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
// Inserting the full lower 32-bits offset 31 so the sign bit ends up at offset 63.
|
||||
GPR = _Bfi(OpSize::i64Bit, 32, 31, GPR, GPR);
|
||||
// Shift right to only get the two sign bits we care about.
|
||||
return _Lshr(OpSize::i64Bit, GPR, Constant(62, ConstPad::NoPad));
|
||||
return _Lshr(OpSize::i64Bit, GPR, Constant(62));
|
||||
};
|
||||
|
||||
auto Mask4Byte = [this](Ref Src) {
|
||||
@@ -1838,7 +1838,7 @@ void OpDispatchBuilder::AVX128_VPERMD(OpcodeArgs) {
|
||||
RefPair Result {};
|
||||
|
||||
Ref IndexMask = _VectorImm(OpSize::i128Bit, OpSize::i32Bit, 0b111);
|
||||
Ref AddConst = Constant(0x03020100, ConstPad::NoPad);
|
||||
Ref AddConst = Constant(0x03020100);
|
||||
Ref Repeating3210 = _VDupFromGPR(OpSize::i128Bit, OpSize::i32Bit, AddConst);
|
||||
|
||||
Result.Low = DoPerm(Src, Indices.Low, IndexMask, Repeating3210);
|
||||
@@ -2035,7 +2035,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpcodeArgs, Op
|
||||
if (BaseAddr && VSIB.Displacement) {
|
||||
BaseAddr = Add(OpSize::i64Bit, BaseAddr, VSIB.Displacement);
|
||||
} else if (VSIB.Displacement) {
|
||||
BaseAddr = Constant(VSIB.Displacement, ConstPad::NoPad);
|
||||
BaseAddr = Constant(VSIB.Displacement);
|
||||
} else if (!BaseAddr) {
|
||||
BaseAddr = Invalid();
|
||||
}
|
||||
@@ -2133,7 +2133,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(OpcodeArgs,
|
||||
if (BaseAddr && VSIB.Displacement) {
|
||||
BaseAddr = Add(OpSize::i64Bit, BaseAddr, VSIB.Displacement);
|
||||
} else if (VSIB.Displacement) {
|
||||
BaseAddr = Constant(VSIB.Displacement, ConstPad::NoPad);
|
||||
BaseAddr = Constant(VSIB.Displacement);
|
||||
} else if (!BaseAddr) {
|
||||
BaseAddr = Invalid();
|
||||
}
|
||||
|
||||
@@ -28,7 +28,7 @@ constexpr std::array<uint32_t, 17> FlagOffsets = {
|
||||
|
||||
void OpDispatchBuilder::ZeroPF_AF() {
|
||||
// PF is stored inverted, so invert it when we zero.
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(Constant(1, ConstPad::NoPad));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(Constant(1));
|
||||
SetAF(0);
|
||||
}
|
||||
|
||||
@@ -247,7 +247,7 @@ void OpDispatchBuilder::CalculateAF(Ref Src1, Ref Src2) {
|
||||
// We store the XOR of the arguments. At read time, we XOR with the
|
||||
// appropriate bit of the result (available as the PF flag) and extract the
|
||||
// appropriate bit. Again 64-bit to avoid masking.
|
||||
Ref XorRes = Src1 == Src2 ? Constant(0, ConstPad::NoPad) : _Xor(OpSize::i64Bit, Src1, Src2);
|
||||
Ref XorRes = Src1 == Src2 ? Constant(0) : _Xor(OpSize::i64Bit, Src1, Src2);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
|
||||
|
||||
@@ -740,7 +740,7 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
// Inserting the full lower 32-bits offset 31 so the sign bit ends up at offset 63.
|
||||
GPR = _Bfi(OpSize::i64Bit, 32, 31, GPR, GPR);
|
||||
// Shift right to only get the two sign bits we care about.
|
||||
GPR = _Lshr(OpSize::i64Bit, GPR, Constant(62, ConstPad::NoPad));
|
||||
GPR = _Lshr(OpSize::i64Bit, GPR, Constant(62));
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize());
|
||||
} else if (Size == OpSize::i128Bit && ElementSize == OpSize::i32Bit) {
|
||||
// Shift all the sign bits to the bottom of their respective elements.
|
||||
@@ -755,7 +755,7 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
Ref GPR = _VExtractToGPR(Size, OpSize::i32Bit, Src, 0);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize());
|
||||
} else {
|
||||
Ref CurrentVal = Constant(0, ConstPad::NoPad);
|
||||
Ref CurrentVal = Constant(0);
|
||||
|
||||
for (unsigned i = 0; i < NumElements; ++i) {
|
||||
// Extract the top bit of the element
|
||||
@@ -2121,7 +2121,7 @@ Ref OpDispatchBuilder::CVTFPR_To_GPRImpl(OpcodeArgs, Ref Src, IR::OpSize SrcElem
|
||||
Ref Converted = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
|
||||
bool Dst32 = GPRSize == OpSize::i32Bit;
|
||||
Ref MaxI = Dst32 ? Constant(0x80000000, ConstPad::NoPad) : Constant(0x8000000000000000, ConstPad::NoPad);
|
||||
Ref MaxI = Dst32 ? Constant(0x80000000) : Constant(0x8000000000000000);
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcElementSize, (SrcElementSize == OpSize::i32Bit) ?
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F32_I32 : NAMED_VECTOR_CVTMAX_F32_I64) :
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F64_I32 : NAMED_VECTOR_CVTMAX_F64_I64));
|
||||
@@ -2552,7 +2552,7 @@ void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) {
|
||||
|
||||
// XSTATE_BV section of the header is 8 bytes in size, but we only really
|
||||
// care about setting at most 3 bits in the first byte. We zero out the rest.
|
||||
_StoreMemGPR(OpSize::i64Bit, RequestedFeatures, Base, Constant(512, ConstPad::NoPad), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(OpSize::i64Bit, RequestedFeatures, Base, Constant(512), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -2578,12 +2578,12 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
|
||||
_StoreMemGPR(OpSize::i16Bit, MemBase, FCW, OpSize::i16Bit);
|
||||
}
|
||||
|
||||
{ _StoreMemGPR(OpSize::i16Bit, ReconstructFSW_Helper(), MemBase, Constant(2, ConstPad::NoPad), OpSize::i16Bit, MemOffsetType::SXTX, 1); }
|
||||
{ _StoreMemGPR(OpSize::i16Bit, ReconstructFSW_Helper(), MemBase, Constant(2), OpSize::i16Bit, MemOffsetType::SXTX, 1); }
|
||||
|
||||
{
|
||||
// Abridged FTW
|
||||
auto FTW = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
_StoreMemGPR(OpSize::i8Bit, FTW, MemBase, Constant(4, ConstPad::NoPad), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(OpSize::i8Bit, FTW, MemBase, Constant(4), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
// BYTE | 0 1 | 2 3 | 4 | 5 | 6 7 | 8 9 | a b | c d | e f |
|
||||
@@ -2633,7 +2633,7 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
|
||||
//
|
||||
// x87 registers are stored rotated depending on the current TOP.
|
||||
Ref Top = GetX87Top();
|
||||
auto SevenConst = Constant(7, ConstPad::NoPad);
|
||||
auto SevenConst = Constant(7);
|
||||
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
|
||||
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
|
||||
@@ -2641,7 +2641,7 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
_StoreMemFPR(OpSize::i128Bit, data, MemBase, Constant(16 * i + 32, ConstPad::NoPad), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i128Bit, data, MemBase, Constant(16 * i + 32), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
}
|
||||
@@ -2656,7 +2656,7 @@ void OpDispatchBuilder::SaveSSEState(Ref MemBase) {
|
||||
|
||||
void OpDispatchBuilder::SaveMXCSRState(Ref MemBase) {
|
||||
// Store MXCSR and the mask for all bits.
|
||||
_StoreMemPairGPR(OpSize::i32Bit, GetMXCSR(), Constant(0xFFFF, ConstPad::NoPad), MemBase, 24);
|
||||
_StoreMemPairGPR(OpSize::i32Bit, GetMXCSR(), Constant(0xFFFF), MemBase, 24);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SaveAVXState(Ref MemBase) {
|
||||
@@ -2674,7 +2674,7 @@ Ref OpDispatchBuilder::GetMXCSR() {
|
||||
Ref MXCSR = _LoadContextGPR(OpSize::i32Bit, offsetof(FEXCore::Core::CPUState, mxcsr));
|
||||
// Mask out unsupported bits
|
||||
// Keeps FZ, RC, exception masks, and DAZ
|
||||
MXCSR = _And(OpSize::i32Bit, MXCSR, Constant(0xFFC0, ConstPad::NoPad));
|
||||
MXCSR = _And(OpSize::i32Bit, MXCSR, Constant(0xFFC0));
|
||||
return MXCSR;
|
||||
}
|
||||
|
||||
@@ -2684,7 +2684,7 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) {
|
||||
RestoreX87State(Mem);
|
||||
RestoreSSEState(Mem);
|
||||
|
||||
Ref MXCSR = _LoadMemGPR(OpSize::i32Bit, Mem, Constant(24, ConstPad::NoPad), OpSize::i32Bit, MemOffsetType::SXTX, 1);
|
||||
Ref MXCSR = _LoadMemGPR(OpSize::i32Bit, Mem, Constant(24), OpSize::i32Bit, MemOffsetType::SXTX, 1);
|
||||
RestoreMXCSRState(MXCSR);
|
||||
}
|
||||
|
||||
@@ -2701,7 +2701,7 @@ void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
|
||||
// Note: we rematerialize Base/Mask in each block to avoid crossblock
|
||||
// liveness.
|
||||
Ref Base = XSaveBase(Op);
|
||||
Ref Mask = _LoadMemGPR(OpSize::i64Bit, Base, Constant(512, ConstPad::NoPad), OpSize::i64Bit, MemOffsetType::SXTX, 1);
|
||||
Ref Mask = _LoadMemGPR(OpSize::i64Bit, Base, Constant(512), OpSize::i64Bit, MemOffsetType::SXTX, 1);
|
||||
|
||||
Ref BitFlag = _Bfe(OpSize, FieldSize, BitIndex, Mask);
|
||||
auto CondJump_ = CondJump(BitFlag, CondClass::NEQ);
|
||||
@@ -2745,7 +2745,7 @@ void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) {
|
||||
1,
|
||||
[this, Op] {
|
||||
Ref Base = XSaveBase(Op);
|
||||
Ref MXCSR = _LoadMemGPR(OpSize::i32Bit, Base, Constant(24, ConstPad::NoPad), OpSize::i32Bit, MemOffsetType::SXTX, 1);
|
||||
Ref MXCSR = _LoadMemGPR(OpSize::i32Bit, Base, Constant(24), OpSize::i32Bit, MemOffsetType::SXTX, 1);
|
||||
RestoreMXCSRState(MXCSR);
|
||||
},
|
||||
[] { /* Intentionally do nothing*/ }, 2);
|
||||
@@ -2759,13 +2759,13 @@ void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
{
|
||||
auto NewFSW = _LoadMemGPR(OpSize::i16Bit, MemBase, Constant(2, ConstPad::NoPad), OpSize::i16Bit, MemOffsetType::SXTX, 1);
|
||||
auto NewFSW = _LoadMemGPR(OpSize::i16Bit, MemBase, Constant(2), OpSize::i16Bit, MemOffsetType::SXTX, 1);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
}
|
||||
|
||||
{
|
||||
// Abridged FTW
|
||||
auto NewFTW = _LoadMemGPR(OpSize::i8Bit, MemBase, Constant(4, ConstPad::NoPad), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
auto NewFTW = _LoadMemGPR(OpSize::i8Bit, MemBase, Constant(4), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
_StoreContextGPR(OpSize::i8Bit, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
|
||||
@@ -2789,7 +2789,7 @@ void OpDispatchBuilder::RestoreSSEState(Ref MemBase) {
|
||||
|
||||
void OpDispatchBuilder::RestoreMXCSRState(Ref MXCSR) {
|
||||
// Mask out unsupported bits
|
||||
MXCSR = _And(OpSize::i32Bit, MXCSR, Constant(0xFFC0, ConstPad::NoPad));
|
||||
MXCSR = _And(OpSize::i32Bit, MXCSR, Constant(0xFFC0));
|
||||
|
||||
_StoreContextGPR(OpSize::i32Bit, MXCSR, offsetof(FEXCore::Core::CPUState, mxcsr));
|
||||
// We only support the rounding mode and FTZ bit being set
|
||||
@@ -3988,7 +3988,7 @@ void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref
|
||||
const auto ElementSizeInBits = IR::OpSizeAsBits(ElementSize);
|
||||
const auto MaskConstant = uint64_t {1} << (ElementSizeInBits - 1);
|
||||
|
||||
Ref Mask = _VDupFromGPR(SrcSize, ElementSize, Constant(MaskConstant, ConstPad::NoPad));
|
||||
Ref Mask = _VDupFromGPR(SrcSize, ElementSize, Constant(MaskConstant));
|
||||
|
||||
Ref AndTest = _VAnd(SrcSize, OpSize::i8Bit, Src2, Src1);
|
||||
Ref AndNotTest = _VAndn(SrcSize, OpSize::i8Bit, Src2, Src1);
|
||||
@@ -4589,7 +4589,7 @@ void OpDispatchBuilder::VPERMDOp(OpcodeArgs) {
|
||||
// Get rid of any junk unrelated to the relevant selector index bits (bits [2:0])
|
||||
Ref IndexMask = _VectorImm(DstSize, OpSize::i32Bit, 0b111);
|
||||
|
||||
Ref AddConst = Constant(0x03020100, ConstPad::NoPad);
|
||||
Ref AddConst = Constant(0x03020100);
|
||||
Ref Repeating3210 = _VDupFromGPR(DstSize, OpSize::i32Bit, AddConst);
|
||||
Ref FinalIndices = VPERMDIndices(OpSizeFromDst(Op), Indices, IndexMask, Repeating3210);
|
||||
|
||||
@@ -4824,7 +4824,7 @@ Ref OpDispatchBuilder::VPERMILRegOpImpl(OpSize DstSize, IR::OpSize ElementSize,
|
||||
Ref ShiftedIndices = _VShlI(DstSize, OpSize::i8Bit, IndexTrn3, IndexShift);
|
||||
|
||||
uint64_t VConstant = IsPD ? 0x0706050403020100 : 0x03020100;
|
||||
Ref VectorConst = _VDupFromGPR(DstSize, ElementSize, Constant(VConstant, ConstPad::NoPad));
|
||||
Ref VectorConst = _VDupFromGPR(DstSize, ElementSize, Constant(VConstant));
|
||||
Ref FinalIndices {};
|
||||
|
||||
if (Is256Bit) {
|
||||
@@ -4883,7 +4883,7 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
|
||||
IntermediateResult = _VPCMPISTRX(Src1, Src2, Control);
|
||||
}
|
||||
|
||||
Ref ZeroConst = Constant(0, ConstPad::NoPad);
|
||||
Ref ZeroConst = Constant(0);
|
||||
|
||||
if (IsMask) {
|
||||
// For the masked variant of the instructions, if control[6] is set, then we
|
||||
@@ -4920,7 +4920,7 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask
|
||||
|
||||
Ref ResultNoFlags = _Bfe(OpSize::i32Bit, 16, 0, IntermediateResult);
|
||||
|
||||
Ref IfZero = Constant(16 >> (Control & 1), ConstPad::NoPad);
|
||||
Ref IfZero = Constant(16 >> (Control & 1));
|
||||
Ref IfNotZero = UseMSBIndex ? _FindMSB(IR::OpSize::i32Bit, ResultNoFlags) : _FindLSB(IR::OpSize::i32Bit, ResultNoFlags);
|
||||
Ref Result = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, ResultNoFlags, ZeroConst, IfZero, IfNotZero);
|
||||
|
||||
@@ -5110,7 +5110,7 @@ void OpDispatchBuilder::VPGATHER(OpcodeArgs) {
|
||||
if (BaseAddr && VSIB.Displacement) {
|
||||
BaseAddr = Add(OpSize::i64Bit, BaseAddr, VSIB.Displacement);
|
||||
} else if (VSIB.Displacement) {
|
||||
BaseAddr = Constant(VSIB.Displacement, ConstPad::NoPad);
|
||||
BaseAddr = Constant(VSIB.Displacement);
|
||||
} else if (!BaseAddr) {
|
||||
BaseAddr = Invalid();
|
||||
}
|
||||
@@ -5156,7 +5156,7 @@ void OpDispatchBuilder::Extrq_imm(OpcodeArgs) {
|
||||
}
|
||||
|
||||
const uint64_t Mask = ~0ULL >> (MaskWidth == 0 ? 0 : (64 - MaskWidth));
|
||||
const Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask, ConstPad::NoPad));
|
||||
const Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask));
|
||||
Result = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Result, MaskVector);
|
||||
|
||||
StoreResultFPR(Op, Result);
|
||||
@@ -5170,7 +5170,7 @@ void OpDispatchBuilder::Insertq_imm(OpcodeArgs) {
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
const uint64_t Mask = ~0ULL >> (MaskWidth == 0 ? 0 : (64 - MaskWidth));
|
||||
Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask, ConstPad::NoPad));
|
||||
Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask));
|
||||
|
||||
// Mask incoming source.
|
||||
Src = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Src, MaskVector);
|
||||
|
||||
@@ -41,13 +41,13 @@ void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
|
||||
// Invert FTW and clear the odd bits. Even bits are 1 if the pair
|
||||
// is not equal to 11, and odd bits are 0.
|
||||
FTW = _Andn(OpSize::i32Bit, Constant(0x55555555, ConstPad::NoPad), FTW);
|
||||
FTW = _Andn(OpSize::i32Bit, Constant(0x55555555), FTW);
|
||||
|
||||
// All that's left is to compact away the odd bits. That is a Morton
|
||||
// deinterleave operation, which has a standard solution. See
|
||||
// https://stackoverflow.com/questions/3137266/how-to-de-interleave-bits-unmortonizing
|
||||
FTW = _And(OpSize::i32Bit, _Orlshr(OpSize::i32Bit, FTW, FTW, 1), Constant(0x33333333, ConstPad::NoPad));
|
||||
FTW = _And(OpSize::i32Bit, _Orlshr(OpSize::i32Bit, FTW, FTW, 2), Constant(0x0f0f0f0f, ConstPad::NoPad));
|
||||
FTW = _And(OpSize::i32Bit, _Orlshr(OpSize::i32Bit, FTW, FTW, 1), Constant(0x33333333));
|
||||
FTW = _And(OpSize::i32Bit, _Orlshr(OpSize::i32Bit, FTW, FTW, 2), Constant(0x0f0f0f0f));
|
||||
FTW = _Orlshr(OpSize::i32Bit, FTW, FTW, 4);
|
||||
|
||||
// ...and that's it. StoreContext implicitly does the final masking.
|
||||
@@ -107,16 +107,16 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
SaveNZCV();
|
||||
|
||||
// Extract sign and make integer absolute
|
||||
auto zero = Constant(0, ConstPad::NoPad);
|
||||
auto zero = Constant(0);
|
||||
_SubNZCV(OpSize::i64Bit, Data, zero);
|
||||
auto sign = _NZCVSelect(OpSize::i64Bit, CondClass::SLT, Constant(0x8000, ConstPad::NoPad), zero);
|
||||
auto sign = _NZCVSelect(OpSize::i64Bit, CondClass::SLT, Constant(0x8000), zero);
|
||||
auto absolute = _Neg(OpSize::i64Bit, Data, CondClass::MI);
|
||||
|
||||
// left justify the absolute integer
|
||||
auto shift = Sub(OpSize::i64Bit, Constant(63, ConstPad::NoPad), _FindMSB(IR::OpSize::i64Bit, absolute));
|
||||
auto shift = Sub(OpSize::i64Bit, Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
|
||||
auto shifted = _Lshl(OpSize::i64Bit, absolute, shift);
|
||||
|
||||
auto adjusted_exponent = Sub(OpSize::i64Bit, Constant(0x3fff + 63, ConstPad::NoPad), shift);
|
||||
auto adjusted_exponent = Sub(OpSize::i64Bit, Constant(0x3fff + 63), shift);
|
||||
auto zeroed_exponent = _Select(OpSize::i64Bit, OpSize::i64Bit, CondClass::EQ, absolute, zero, zero, adjusted_exponent);
|
||||
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
|
||||
|
||||
@@ -159,11 +159,11 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
// Extract the 80-bit float value to check for special cases
|
||||
// Get the upper 64 bits which contain sign and exponent and then the exponent from upper.
|
||||
Ref Upper = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, Data, 1);
|
||||
Ref Exponent = _And(OpSize::i64Bit, Upper, Constant(0x7fff, ConstPad::NoPad));
|
||||
Ref Exponent = _And(OpSize::i64Bit, Upper, Constant(0x7fff));
|
||||
|
||||
// Check for NaN/Infinity: exponent = 0x7fff
|
||||
SaveNZCV();
|
||||
_TestNZ(OpSize::i64Bit, Exponent, Constant(0x7fff, ConstPad::NoPad));
|
||||
_TestNZ(OpSize::i64Bit, Exponent, Constant(0x7fff));
|
||||
Ref IsSpecial = _NZCVSelect01(CondClass::EQ);
|
||||
|
||||
// For overflow detection, check if exponent indicates a value >= 2^15
|
||||
@@ -340,17 +340,17 @@ Ref OpDispatchBuilder::GetX87FTW_Helper() {
|
||||
// https://graphics.stanford.edu/~seander/bithacks.html#InterleaveBMN
|
||||
Ref X = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 4);
|
||||
X = _And(OpSize::i32Bit, X, Constant(0x0f0f0f0f, ConstPad::NoPad));
|
||||
X = _And(OpSize::i32Bit, X, Constant(0x0f0f0f0f));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 2);
|
||||
X = _And(OpSize::i32Bit, X, Constant(0x33333333, ConstPad::NoPad));
|
||||
X = _And(OpSize::i32Bit, X, Constant(0x33333333));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 1);
|
||||
X = _And(OpSize::i32Bit, X, Constant(0x55555555, ConstPad::NoPad));
|
||||
X = _And(OpSize::i32Bit, X, Constant(0x55555555));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 1);
|
||||
|
||||
// The above sequence sets valid to 11 and empty to 00, so invert to finalize.
|
||||
static_assert(static_cast<uint8_t>(FPState::X87Tag::Valid) == 0b00);
|
||||
static_assert(static_cast<uint8_t>(FPState::X87Tag::Empty) == 0b11);
|
||||
return _Xor(OpSize::i32Bit, X, Constant(0xffff, ConstPad::NoPad));
|
||||
return _Xor(OpSize::i32Bit, X, Constant(0xffff));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
@@ -387,33 +387,33 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
_StoreMemGPR(Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1); }
|
||||
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
|
||||
|
||||
auto ZeroConst = Constant(0, ConstPad::NoPad);
|
||||
auto ZeroConst = Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -485,44 +485,43 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
_StoreMemGPR(Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1); }
|
||||
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
|
||||
|
||||
auto ZeroConst = Constant(0, ConstPad::NoPad);
|
||||
auto ZeroConst = Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
auto SevenConst = Constant(7, ConstPad::NoPad);
|
||||
auto SevenConst = Constant(7);
|
||||
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
_StoreMemFPR(OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i), ConstPad::NoPad), OpSize::i8Bit,
|
||||
MemOffsetType::SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
|
||||
@@ -534,11 +533,9 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMemFPR(OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10), ConstPad::NoPad), OpSize::i8Bit,
|
||||
MemOffsetType::SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
auto topBytes = _VDupElement(OpSize::i128Bit, OpSize::i16Bit, data, 4);
|
||||
_StoreMemFPR(OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8, ConstPad::NoPad), OpSize::i8Bit,
|
||||
MemOffsetType::SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
|
||||
// reset to default
|
||||
FNINIT(Op);
|
||||
@@ -555,28 +552,27 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = NewFCW;
|
||||
auto roundShift = Constant(10, ConstPad::NoPad);
|
||||
auto roundMask = Constant(3, ConstPad::NoPad);
|
||||
auto roundShift = Constant(10);
|
||||
auto roundMask = Constant(3);
|
||||
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
|
||||
roundingMode = _And(OpSize::i32Bit, roundingMode, roundMask);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
}
|
||||
|
||||
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 1, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
|
||||
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1);
|
||||
Ref Top = ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1));
|
||||
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
|
||||
}
|
||||
|
||||
auto SevenConst = Constant(7, ConstPad::NoPad);
|
||||
auto low = Constant(~0ULL, ConstPad::NoPad);
|
||||
auto high = Constant(0xFFFF, ConstPad::NoPad);
|
||||
auto SevenConst = Constant(7);
|
||||
auto low = Constant(~0ULL);
|
||||
auto high = Constant(0xFFFF);
|
||||
Ref Mask = _VLoadTwoGPRs(low, high);
|
||||
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i), ConstPad::NoPad), OpSize::i8Bit,
|
||||
MemOffsetType::SXTX, 1);
|
||||
Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask);
|
||||
if (ReducedPrecisionMode) {
|
||||
@@ -592,10 +588,8 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
Ref Reg =
|
||||
_LoadMemFPR(OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7), ConstPad::NoPad), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Ref RegHigh = _LoadMemFPR(OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8, ConstPad::NoPad), OpSize::i8Bit,
|
||||
MemOffsetType::SXTX, 1);
|
||||
Ref Reg = _LoadMemFPR(OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Ref RegHigh = _LoadMemFPR(OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Reg = _VInsElement(OpSize::i128Bit, OpSize::i16Bit, 4, 0, Reg, RegHigh);
|
||||
if (ReducedPrecisionMode) {
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg); // Convert to double precision
|
||||
@@ -624,13 +618,13 @@ void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
if (Offset != 0) {
|
||||
_F80StackXchange(Offset);
|
||||
}
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0, ConstPad::NoPad));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FYL2X(OpcodeArgs, bool IsFYL2XP1) {
|
||||
if (IsFYL2XP1) {
|
||||
// create an add between top of stack and 1.
|
||||
Ref One = ReducedPrecisionMode ? _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(0x3FF0000000000000, ConstPad::NoPad)) :
|
||||
Ref One = ReducedPrecisionMode ? _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(0x3FF0000000000000)) :
|
||||
LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NamedVectorConstant::NAMED_VECTOR_X87_ONE);
|
||||
_F80AddValue(0, One);
|
||||
}
|
||||
@@ -671,7 +665,7 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
|
||||
if (WhichFlags == FCOMIFlags::FLAGS_X87) {
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0, ConstPad::NoPad));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
} else {
|
||||
@@ -681,7 +675,7 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
|
||||
// PF is stored inverted, so invert from the host flag.
|
||||
// TODO: This could perhaps be optimized?
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, Constant(1, ConstPad::NoPad));
|
||||
auto PF = _Xor(OpSize::i32Bit, HostFlag_Unordered, Constant(1));
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_PF_RAW_LOC>(PF);
|
||||
}
|
||||
|
||||
@@ -706,7 +700,7 @@ void OpDispatchBuilder::FTST(OpcodeArgs) {
|
||||
HostFlag_ZF = _Or(OpSize::i32Bit, HostFlag_ZF, HostFlag_Unordered);
|
||||
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(HostFlag_CF);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0, ConstPad::NoPad));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Constant(0));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(HostFlag_Unordered);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(HostFlag_ZF);
|
||||
|
||||
@@ -717,7 +711,7 @@ void OpDispatchBuilder::FTST(OpcodeArgs) {
|
||||
void OpDispatchBuilder::X87OpHelper(OpcodeArgs, FEXCore::IR::IROps IROp, bool ZeroC2) {
|
||||
DeriveOp(Result, IROp, _F80SCALEStack());
|
||||
if (ZeroC2) {
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(Constant(0, ConstPad::NoPad));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(Constant(0));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -743,7 +737,7 @@ void OpDispatchBuilder::X87ModifySTP(OpcodeArgs, bool Inc) {
|
||||
Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
|
||||
// Start with the top value
|
||||
auto Top = T ? T : GetX87Top();
|
||||
Ref FSW = _Lshl(OpSize::i64Bit, Top, Constant(11, ConstPad::NoPad));
|
||||
Ref FSW = _Lshl(OpSize::i64Bit, Top, Constant(11));
|
||||
|
||||
// We must construct the FSW from our various bits
|
||||
auto C0 = GetRFLAG(FEXCore::X86State::X87FLAG_C0_LOC);
|
||||
@@ -775,20 +769,20 @@ void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
|
||||
// Clear the exception flag bit
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Constant(0, ConstPad::NoPad));
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_IE_LOC>(_Constant(0));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
_SyncStackToSlow(); // Invalidate x87 register caches
|
||||
|
||||
auto Zero = Constant(0, ConstPad::NoPad);
|
||||
auto Zero = Constant(0);
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
_SetRoundingMode(Zero, false, Zero);
|
||||
}
|
||||
|
||||
// Init FCW to 0x037F
|
||||
auto NewFCW = Constant(0x037F, ConstPad::NoPad);
|
||||
auto NewFCW = Constant(0x037F);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Set top to zero
|
||||
@@ -867,7 +861,7 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
auto TopValid = _StackValidTag(0);
|
||||
|
||||
// In the case of top being invalid then C3:C2:C0 is 0b101
|
||||
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1, ConstPad::NoPad));
|
||||
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1));
|
||||
|
||||
auto C2 = TopValid;
|
||||
auto C0 = C3; // Mirror C3 until something other than zero is supported
|
||||
|
||||
@@ -36,12 +36,12 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size), ConstPad::NoPad), Size, MemOffsetType::SXTX, 1);
|
||||
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MemOffsetType::SXTX, 1);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2, ConstPad::NoPad), Size, MemOffsetType::SXTX, 1));
|
||||
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -86,7 +86,7 @@ void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FLDF64_Const(OpcodeArgs, uint64_t Num) {
|
||||
auto Data = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(Num, ConstPad::NoPad));
|
||||
auto Data = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(Num));
|
||||
_PushStack(Data, Data, OpSize::i64Bit);
|
||||
}
|
||||
|
||||
@@ -376,21 +376,21 @@ void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) {
|
||||
Ref Gpr = _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, Node, 0);
|
||||
|
||||
// zero case
|
||||
Ref ExpZV = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(0xfff0'0000'0000'0000UL, ConstPad::NoPad));
|
||||
Ref ExpZV = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, Constant(0xfff0'0000'0000'0000UL));
|
||||
Ref SigZV = Node;
|
||||
|
||||
// non zero case
|
||||
Ref ExpNZ = _Bfe(OpSize::i64Bit, 11, 52, Gpr);
|
||||
ExpNZ = Sub(OpSize::i64Bit, ExpNZ, Constant(1023, ConstPad::NoPad));
|
||||
ExpNZ = Sub(OpSize::i64Bit, ExpNZ, Constant(1023));
|
||||
Ref ExpNZV = _Float_FromGPR_S(OpSize::i64Bit, OpSize::i64Bit, ExpNZ);
|
||||
|
||||
Ref SigNZ = _And(OpSize::i64Bit, Gpr, Constant(0x800f'ffff'ffff'ffffLL, ConstPad::NoPad));
|
||||
SigNZ = _Or(OpSize::i64Bit, SigNZ, Constant(0x3ff0'0000'0000'0000LL, ConstPad::NoPad));
|
||||
Ref SigNZ = _And(OpSize::i64Bit, Gpr, Constant(0x800f'ffff'ffff'ffffLL));
|
||||
SigNZ = _Or(OpSize::i64Bit, SigNZ, Constant(0x3ff0'0000'0000'0000LL));
|
||||
Ref SigNZV = _VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, SigNZ);
|
||||
|
||||
// Comparison and select to push onto stack
|
||||
SaveNZCV();
|
||||
_TestNZ(OpSize::i64Bit, Gpr, Constant(0x7fff'ffff'ffff'ffffUL, ConstPad::NoPad));
|
||||
_TestNZ(OpSize::i64Bit, Gpr, Constant(0x7fff'ffff'ffff'ffffUL));
|
||||
|
||||
Ref Sig = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, SigZV, SigNZV);
|
||||
Ref Exp = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, ExpZV, ExpNZV);
|
||||
|
||||
@@ -936,7 +936,7 @@
|
||||
]
|
||||
},
|
||||
|
||||
"GPR = Constant i64:$Constant, ConstPad:$Pad, i32:$MaxBytes{0}": {
|
||||
"GPR = Constant i64:$Constant, ConstPad:$Pad{IR::ConstPad::NoPad}, i32:$MaxBytes{0}": {
|
||||
"Desc": ["Generates a 64bit constant inside of a GPR",
|
||||
"Unsupported to create a constant in FPR"
|
||||
],
|
||||
|
||||
@@ -109,10 +109,10 @@ public:
|
||||
return _Jump(InvalidNode);
|
||||
}
|
||||
IRPair<IROp_CondJump> _CondJump(Ref ssa0, CondClass cond = CondClass::NEQ) {
|
||||
return _CondJump(ssa0, _Constant(0, ConstPad::NoPad), InvalidNode, InvalidNode, cond, GetOpSize(ssa0));
|
||||
return _CondJump(ssa0, _Constant(0), InvalidNode, InvalidNode, cond, GetOpSize(ssa0));
|
||||
}
|
||||
IRPair<IROp_CondJump> _CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClass cond = CondClass::NEQ) {
|
||||
return _CondJump(ssa0, _Constant(0, ConstPad::NoPad), ssa1, ssa2, cond, GetOpSize(ssa0));
|
||||
return _CondJump(ssa0, _Constant(0), ssa1, ssa2, cond, GetOpSize(ssa0));
|
||||
}
|
||||
|
||||
IRPair<IROp_LoadContext> _LoadContextGPR(OpSize ByteSize, uint32_t Offset) {
|
||||
@@ -184,7 +184,7 @@ public:
|
||||
}
|
||||
|
||||
IRPair<IROp_Select> To01(FEXCore::IR::OpSize CompareSize, OrderedNode* Cmp1) {
|
||||
return Select01(CompareSize, CondClass::NEQ, Cmp1, Constant(0, ConstPad::NoPad));
|
||||
return Select01(CompareSize, CondClass::NEQ, Cmp1, Constant(0));
|
||||
}
|
||||
|
||||
IRPair<IROp_NZCVSelect> _NZCVSelect01(CondClass Cond) {
|
||||
@@ -203,7 +203,7 @@ public:
|
||||
Src2 = -Src2;
|
||||
}
|
||||
|
||||
auto Dest = _Add(Size, Src1, Constant(Src2, ConstPad::NoPad));
|
||||
auto Dest = _Add(Size, Src1, Constant(Src2));
|
||||
Dest.first->Header.Op = Op;
|
||||
return Dest;
|
||||
}
|
||||
@@ -249,7 +249,7 @@ public:
|
||||
Ref ConstantRefs[32];
|
||||
uint32_t NrConstants;
|
||||
|
||||
Ref Constant(int64_t Value, ConstPad Pad, int32_t MaxBytes = 0) {
|
||||
Ref Constant(int64_t Value, ConstPad Pad = IR::ConstPad::NoPad, int32_t MaxBytes = 0) {
|
||||
const ConstantData Data {
|
||||
.Value = Value,
|
||||
.Pad = Pad,
|
||||
|
||||
@@ -514,8 +514,8 @@ bool ConstrainedRAPass::TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header*
|
||||
if (CPUID->DoesXCRFunctionReportConstantData(ConstantFunction)) {
|
||||
const auto Result = CPUID->RunXCRFunction(ConstantFunction);
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
IREmit->_Constant(Result.eax, ConstPad::NoPad).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
|
||||
IREmit->_Constant(Result.edx, ConstPad::NoPad).Node->Reg = PhysicalRegister(Op->OutEDX).Raw;
|
||||
IREmit->_Constant(Result.eax).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
|
||||
IREmit->_Constant(Result.edx).Node->Reg = PhysicalRegister(Op->OutEDX).Raw;
|
||||
IREmit->RemovePostRA(CodeNode);
|
||||
return false;
|
||||
}
|
||||
@@ -533,10 +533,10 @@ bool ConstrainedRAPass::TryPostRAMerge(Ref LastNode, Ref CodeNode, IROp_Header*
|
||||
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
IREmit->_Fence(IR::FenceType::Inst);
|
||||
IREmit->_Constant(Result.eax, ConstPad::NoPad).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
|
||||
IREmit->_Constant(Result.ebx, ConstPad::NoPad).Node->Reg = PhysicalRegister(Op->OutEBX).Raw;
|
||||
IREmit->_Constant(Result.ecx, ConstPad::NoPad).Node->Reg = PhysicalRegister(Op->OutECX).Raw;
|
||||
IREmit->_Constant(Result.edx, ConstPad::NoPad).Node->Reg = PhysicalRegister(Op->OutEDX).Raw;
|
||||
IREmit->_Constant(Result.eax).Node->Reg = PhysicalRegister(Op->OutEAX).Raw;
|
||||
IREmit->_Constant(Result.ebx).Node->Reg = PhysicalRegister(Op->OutEBX).Raw;
|
||||
IREmit->_Constant(Result.ecx).Node->Reg = PhysicalRegister(Op->OutECX).Raw;
|
||||
IREmit->_Constant(Result.edx).Node->Reg = PhysicalRegister(Op->OutEDX).Raw;
|
||||
IREmit->RemovePostRA(CodeNode);
|
||||
return false;
|
||||
}
|
||||
|
||||
@@ -387,11 +387,11 @@ inline void X87StackOptimization::Reset() {
|
||||
inline Ref X87StackOptimization::GetConstant(ssize_t Offset) {
|
||||
if (Offset < 0 || Offset >= X87StackOptimization::ConstantPool.size()) {
|
||||
// not dealt by pool
|
||||
return IREmit->_Constant(Offset, ConstPad::NoPad);
|
||||
return IREmit->_Constant(Offset);
|
||||
}
|
||||
if (ConstantPool[Offset] == nullptr) {
|
||||
|
||||
ConstantPool[Offset] = IREmit->_Constant(Offset, ConstPad::NoPad);
|
||||
ConstantPool[Offset] = IREmit->_Constant(Offset);
|
||||
}
|
||||
return ConstantPool[Offset];
|
||||
}
|
||||
|
||||
Reference in new issue
Block a user