mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 11:00:19 +02:00
IR: Migrate to new helpers
Reduces a bunch of noise related to the register classes and hoists them out so that converting the classes over to enums should be fairly straightforward.
This commit is contained in:
1 parent
c879650c4b
commit
fc63dcedb6
10 files changed
+1015
-1010
No files matched your search
@@ -949,8 +949,8 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
|
||||
IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint), GPRSize);
|
||||
R->Reg = IR::PhysicalRegister(IR::GPRFixedClass, X86State::REG_R11).Raw;
|
||||
} else {
|
||||
emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
emit->_StoreContextFPR(GPRSize, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)),
|
||||
offsetof(Core::CPUState, mm[0][0]));
|
||||
}
|
||||
emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint), IR::BranchHint::None, emit->Invalid(), emit->Invalid());
|
||||
},
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
@@ -1599,8 +1599,7 @@ private:
|
||||
StoreResult(FPRClass, Op, Operand, Src, Align, AccessType);
|
||||
}
|
||||
|
||||
void StoreResult(RegisterClassType Class, X86Tables::DecodedOp Op, Ref Src, OpSize Align,
|
||||
MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResult(RegisterClassType Class, X86Tables::DecodedOp Op, Ref Src, OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT);
|
||||
void StoreResultGPR(X86Tables::DecodedOp Op, Ref Src, OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) {
|
||||
StoreResult(GPRClass, Op, Src, Align, AccessType);
|
||||
}
|
||||
|
||||
@@ -46,9 +46,9 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize(
|
||||
}
|
||||
|
||||
if (NeedsHigh) {
|
||||
return _LoadMemPairAutoTSO(FPRClass, OpSize::i128Bit, A, OpSize::i8Bit);
|
||||
return _LoadMemPairFPRAutoTSO(OpSize::i128Bit, A, OpSize::i8Bit);
|
||||
} else {
|
||||
return {.Low = _LoadMemAutoTSO(FPRClass, OpSize::i128Bit, A, OpSize::i8Bit)};
|
||||
return {.Low = _LoadMemFPRAutoTSO(OpSize::i128Bit, A, OpSize::i8Bit)};
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -95,9 +95,9 @@ void OpDispatchBuilder::AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::Decode
|
||||
AddressMode A = DecodeAddress(Op, Operand, AccessType, false /* IsLoad */);
|
||||
|
||||
if (Src.High) {
|
||||
_StoreMemPairAutoTSO(FPRClass, OpSize::i128Bit, A, Src.Low, Src.High, OpSize::i8Bit);
|
||||
_StoreMemPairFPRAutoTSO(OpSize::i128Bit, A, Src.Low, Src.High, OpSize::i8Bit);
|
||||
} else {
|
||||
_StoreMemAutoTSO(FPRClass, OpSize::i128Bit, A, Src.Low, OpSize::i8Bit);
|
||||
_StoreMemFPRAutoTSO(OpSize::i128Bit, A, Src.Low, OpSize::i8Bit);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -151,13 +151,13 @@ void OpDispatchBuilder::AVX128_VMOVScalarImpl(OpcodeArgs, IR::OpSize ElementSize
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Result, .High = High});
|
||||
} else if (Op->Dest.IsGPR()) {
|
||||
// VMOVSS/SD xmm1, mem32/mem64
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], ElementSize, Op->Flags);
|
||||
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[1], ElementSize, Op->Flags);
|
||||
auto High = LoadZeroVector(OpSize::i128Bit);
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Src, .High = High});
|
||||
} else {
|
||||
// VMOVSS/SD mem32/mem64, xmm1
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src.Low, ElementSize, OpSize::iInvalid);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src.Low, ElementSize, OpSize::iInvalid);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -351,7 +351,7 @@ void OpDispatchBuilder::AVX128_MOVVectorNT(OpcodeArgs) {
|
||||
if (Op->Dest.IsGPR()) {
|
||||
///< MOVNTDQA load non-temporal comes from SSE4.1 and is extended by AVX/AVX2.
|
||||
RefPair Src {};
|
||||
Ref SrcAddr = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Ref SrcAddr = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Src.Low = _VLoadNonTemporal(OpSize::i128Bit, SrcAddr, 0);
|
||||
|
||||
if (Is128Bit) {
|
||||
@@ -362,7 +362,7 @@ void OpDispatchBuilder::AVX128_MOVVectorNT(OpcodeArgs) {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
|
||||
} else {
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit, MemoryAccessType::STREAM);
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
|
||||
if (Is128Bit) {
|
||||
// Single store non-temporal for 128-bit operations.
|
||||
@@ -379,7 +379,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) {
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
} else {
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
|
||||
}
|
||||
|
||||
// This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 256bit
|
||||
@@ -390,7 +390,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) {
|
||||
Src.High = ZeroVector;
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src);
|
||||
} else {
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src.Low, OpSize::i64Bit, OpSize::i64Bit);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src.Low, OpSize::i64Bit, OpSize::i64Bit);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -399,7 +399,7 @@ void OpDispatchBuilder::AVX128_VMOVLP(OpcodeArgs) {
|
||||
|
||||
if (!Op->Dest.IsGPR()) {
|
||||
///< VMOVLPS/PD mem64, xmm1
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src1.Low, OpSize::i64Bit, OpSize::i64Bit);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Src1.Low, OpSize::i64Bit, OpSize::i64Bit);
|
||||
} else if (!Op->Src[1].IsGPR()) {
|
||||
///< VMOVLPS/PD xmm1, xmm2, mem64
|
||||
// Bits[63:0] come from Src2[63:0]
|
||||
@@ -463,7 +463,7 @@ void OpDispatchBuilder::AVX128_VMOVDDUP(OpcodeArgs) {
|
||||
// 128-bit operation only loads 8-bytes.
|
||||
// 256-bit operation loads a full 32-bytes.
|
||||
if (Is128Bit) {
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i64Bit, Op->Flags);
|
||||
} else {
|
||||
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, true);
|
||||
}
|
||||
@@ -558,18 +558,18 @@ void OpDispatchBuilder::AVX128_InsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstEle
|
||||
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
// If the source is a GPR then convert directly from the GPR.
|
||||
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Op->Src[1], GetGPROpSize(), Op->Flags);
|
||||
auto Src2 = LoadSourceGPR_WithOpSize(Op, Op->Src[1], GetGPROpSize(), Op->Flags);
|
||||
Result.Low = _VSToFGPRInsert(OpSize::i128Bit, DstElementSize, SrcSize, Src1.Low, Src2, false);
|
||||
} else if (SrcSize != DstElementSize) {
|
||||
// If the source is from memory but the Source size and destination size aren't the same,
|
||||
// then it is more optimal to load in to a GPR and convert between GPR->FPR.
|
||||
// ARM GPR->FPR conversion supports different size source and destinations while FPR->FPR doesn't.
|
||||
auto Src2 = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags);
|
||||
auto Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags);
|
||||
Result.Low = _VSToFGPRInsert(DstSize, DstElementSize, SrcSize, Src1.Low, Src2, false);
|
||||
} else {
|
||||
// In the case of cvtsi2s{s,d} where the source and destination are the same size,
|
||||
// then it is more optimal to load in to the FPR register directly and convert there.
|
||||
auto Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
auto Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
// Always signed
|
||||
Result.Low = _VSToFVectorInsert(DstSize, DstElementSize, DstElementSize, Src1.Low, Src2, false, false);
|
||||
}
|
||||
@@ -589,11 +589,11 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSi
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
} else {
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcElementSize, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcElementSize, Op->Flags);
|
||||
}
|
||||
|
||||
Ref Result = CVTFPR_To_GPRImpl(Op, Src.Low, SrcElementSize, HostRoundingMode);
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) {
|
||||
@@ -636,7 +636,7 @@ void OpDispatchBuilder::AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
} else {
|
||||
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
Comiss(ElementSize, Src1.Low, Src2.Low);
|
||||
@@ -653,7 +653,7 @@ void OpDispatchBuilder::AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IR
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
} else {
|
||||
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
@@ -690,7 +690,7 @@ void OpDispatchBuilder::AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSi
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
} else {
|
||||
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
const uint8_t CompType = Op->Src[2].Literal();
|
||||
@@ -708,12 +708,12 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) {
|
||||
RefPair Result {};
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
// Loading from GPR and moving to Vector.
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], GetGPROpSize(), Op->Flags);
|
||||
Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], GetGPROpSize(), Op->Flags);
|
||||
// zext to 128bit
|
||||
Result.Low = _VCastFromGPR(OpSize::i128Bit, OpSizeFromSrc(Op), Src);
|
||||
} else {
|
||||
// Loading from Memory as a scalar. Zero extend
|
||||
Result.Low = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Result.Low = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
|
||||
Result.High = LoadZeroVector(OpSize::i128Bit);
|
||||
@@ -726,11 +726,11 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) {
|
||||
auto ElementSize = OpSizeFromDst(Op);
|
||||
// Extract element from GPR. Zero extending in the process.
|
||||
Src.Low = _VExtractToGPR(OpSizeFromSrc(Op), ElementSize, Src.Low, 0);
|
||||
StoreResult(GPRClass, Op, Op->Dest, Src.Low, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, Op->Dest, Src.Low, OpSize::iInvalid);
|
||||
} else {
|
||||
// Storing first element to memory.
|
||||
Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreMem(FPRClass, OpSizeFromDst(Op), Dest, Src.Low, OpSize::i8Bit);
|
||||
Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
_StoreMemFPR(OpSizeFromDst(Op), Dest, Src.Low, OpSize::i8Bit);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -758,7 +758,7 @@ void OpDispatchBuilder::AVX128_PExtr(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
// Extract already zero extends the result.
|
||||
Ref Result = _VExtractToGPR(OpSize::i128Bit, OverridenElementSize, Src.Low, Index);
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, OpSize::iInvalid);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Result, GPRSize, OpSize::iInvalid);
|
||||
return;
|
||||
}
|
||||
|
||||
@@ -779,7 +779,7 @@ void OpDispatchBuilder::AVX128_ExtendVectorElements(OpcodeArgs, IR::OpSize Eleme
|
||||
const auto SrcSize = OpSizeFromSrc(Op);
|
||||
const auto LoadSize = Is256Bit ? IR::SizeToOpSize(IR::OpSizeToSize(SrcSize) * 2) : SrcSize;
|
||||
|
||||
return LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags);
|
||||
return LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags);
|
||||
}
|
||||
};
|
||||
|
||||
@@ -868,7 +868,7 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
auto GPRHigh = Mask8Byte(Src.High);
|
||||
GPR = _Orlshl(OpSize::i64Bit, GPRLow, GPRHigh, 2);
|
||||
}
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, GPR, GetGPROpSize(), OpSize::iInvalid);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize(), OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
|
||||
@@ -897,7 +897,7 @@ void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) {
|
||||
Result = _Orlshl(OpSize::i64Bit, Result, ResultHigh, 16);
|
||||
}
|
||||
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, const X86Tables::DecodedOperand& Src1Op,
|
||||
@@ -910,7 +910,7 @@ void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, con
|
||||
|
||||
if (Src2Op.IsGPR()) {
|
||||
// If the source is a GPR then convert directly from the GPR.
|
||||
auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, GetGPROpSize(), Op->Flags);
|
||||
auto Src2 = LoadSourceGPR_WithOpSize(Op, Src2Op, GetGPROpSize(), Op->Flags);
|
||||
Result.Low = _VInsGPR(OpSize::i128Bit, ElementSize, Index, Src1.Low, Src2);
|
||||
} else {
|
||||
// If loading from memory then we only load the element size
|
||||
@@ -1047,7 +1047,7 @@ void OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::O
|
||||
// Then zero extends the top 128-bit.
|
||||
const auto SrcSize = Op->Src[1].IsGPR() ? OpSize::i128Bit : SrcElementSize;
|
||||
auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false);
|
||||
Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true});
|
||||
|
||||
Ref Result = _VFToFScalarInsert(OpSize::i128Bit, DstElementSize, SrcElementSize, Src1.Low, Src2, false);
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result));
|
||||
@@ -1076,7 +1076,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize
|
||||
} else {
|
||||
// Handle 64-bit memory source.
|
||||
// In the case of cvtps2pd xmm, m64.
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags);
|
||||
}
|
||||
|
||||
RefPair Result {};
|
||||
@@ -1154,7 +1154,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float(OpcodeArgs, IR::OpSize Sr
|
||||
// unnecessarily zero extend the vector. Otherwise, if
|
||||
// memory, then we want to load the element size exactly.
|
||||
const auto LoadSize = IR::SizeToOpSize(8 * (IR::OpSizeToSize(Size) / 16));
|
||||
return RefPair {.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags)};
|
||||
return RefPair {.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags)};
|
||||
} else {
|
||||
return AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit);
|
||||
}
|
||||
@@ -1304,7 +1304,7 @@ void OpDispatchBuilder::AVX128_InsertScalarRound(OpcodeArgs, IR::OpSize ElementS
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false);
|
||||
} else {
|
||||
Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
// If OpSize == ElementSize then it only does the lower scalar op
|
||||
@@ -1616,11 +1616,11 @@ void OpDispatchBuilder::AVX128_MASKMOV(OpcodeArgs) {
|
||||
// RDI source (DS prefix by default)
|
||||
auto MemDest = MakeSegmentAddress(X86State::REG_RDI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
|
||||
Ref XMMReg = _LoadMem(FPRClass, Size, MemDest, OpSize::i8Bit);
|
||||
Ref XMMReg = _LoadMemFPR(Size, MemDest, OpSize::i8Bit);
|
||||
|
||||
// If the Mask element high bit is set then overwrite the element with the source, else keep the memory variant
|
||||
XMMReg = _VBSL(Size, MaskSrc.Low, VectorSrc.Low, XMMReg);
|
||||
_StoreMem(FPRClass, Size, MemDest, XMMReg, OpSize::i8Bit);
|
||||
_StoreMemFPR(Size, MemDest, XMMReg, OpSize::i8Bit);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize) {
|
||||
@@ -1660,7 +1660,7 @@ void OpDispatchBuilder::AVX128_SaveAVXState(Ref MemBase) {
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
RefPair Pair = LoadContextPair(OpSize::i128Bit, AVXHigh0Index + i);
|
||||
_StoreMemPair(FPRClass, OpSize::i128Bit, Pair.Low, Pair.High, MemBase, i * 16 + 576);
|
||||
_StoreMemPairFPR(OpSize::i128Bit, Pair.Low, Pair.High, MemBase, i * 16 + 576);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1668,7 +1668,7 @@ void OpDispatchBuilder::AVX128_RestoreAVXState(Ref MemBase) {
|
||||
const auto NumRegs = Is64BitMode ? 16U : 8U;
|
||||
|
||||
for (uint32_t i = 0; i < NumRegs; i += 2) {
|
||||
auto YMMHRegs = LoadMemPair(FPRClass, OpSize::i128Bit, MemBase, i * 16 + 576);
|
||||
auto YMMHRegs = LoadMemPairFPR(OpSize::i128Bit, MemBase, i * 16 + 576);
|
||||
|
||||
AVX128_StoreXMMRegister(i, YMMHRegs.Low, true);
|
||||
AVX128_StoreXMMRegister(i + 1, YMMHRegs.High, true);
|
||||
@@ -1968,7 +1968,7 @@ void OpDispatchBuilder::AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Sr
|
||||
if (Op->Src[1].IsGPR()) {
|
||||
Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false).Low;
|
||||
} else {
|
||||
Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
Ref Sources[3] = {Dest, Src1, Src2};
|
||||
@@ -2240,7 +2240,7 @@ void OpDispatchBuilder::AVX128_VCVTPH2PS(OpcodeArgs) {
|
||||
// In the event that a memory operand is used as the source operand,
|
||||
// the access width will always be half the size of the destination vector width
|
||||
// (i.e. 128-bit vector -> 64-bit mem, 256-bit vector -> 128-bit mem)
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
}
|
||||
|
||||
RefPair Result {};
|
||||
@@ -2295,7 +2295,7 @@ void OpDispatchBuilder::AVX128_VCVTPS2PH(OpcodeArgs) {
|
||||
}
|
||||
|
||||
if (!Op->Dest.IsGPR()) {
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result.Low, StoreSize, OpSize::iInvalid);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, Result.Low, StoreSize, OpSize::iInvalid);
|
||||
} else {
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
}
|
||||
|
||||
@@ -23,8 +23,8 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30.
|
||||
// This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this.
|
||||
@@ -36,7 +36,7 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) {
|
||||
auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode);
|
||||
auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
@@ -44,15 +44,15 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1);
|
||||
|
||||
// [W0, W1, W2, W3] ^ [W2, W3, W4, W5]
|
||||
Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
@@ -60,8 +60,8 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
// ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0.
|
||||
auto Src1 = SHADataShuffle(Dest);
|
||||
@@ -70,7 +70,7 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) {
|
||||
// The result is swizzled differently than expected
|
||||
auto Result = SHADataShuffle(_VSha1SU1(Src1, Src2));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
@@ -79,8 +79,8 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
return;
|
||||
}
|
||||
const uint64_t Imm8 = Op->Src[1].Literal() & 0b11;
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
Ref Result {};
|
||||
Ref ConstantVector {};
|
||||
@@ -112,7 +112,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) {
|
||||
case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break;
|
||||
}
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
@@ -120,12 +120,12 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto Result = _VSha256U0(Dest, Src);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
@@ -133,8 +133,8 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
|
||||
auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3);
|
||||
auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3);
|
||||
@@ -142,7 +142,7 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) {
|
||||
|
||||
auto Result = _VSha256U1(Src1, Src2);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
@@ -150,8 +150,8 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
// Hardcoded to XMM0
|
||||
auto XMM0 = LoadXMMRegister(0);
|
||||
|
||||
@@ -177,7 +177,7 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) {
|
||||
auto B = _VSha256H2(EFGH, ABCD, Key);
|
||||
auto Result = shuffle_abcd(A, B);
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
@@ -185,9 +185,9 @@ void OpDispatchBuilder::AESImcOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESImc(Src);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
@@ -195,10 +195,10 @@ void OpDispatchBuilder::AESEncOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
@@ -208,11 +208,11 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
@@ -220,10 +220,10 @@ void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
@@ -233,11 +233,11 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESENCLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
@@ -245,10 +245,10 @@ void OpDispatchBuilder::AESDecOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
@@ -258,11 +258,11 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDEC.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
@@ -270,10 +270,10 @@ void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit));
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
@@ -283,15 +283,15 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) {
|
||||
// TODO: Handle 256-bit VAESDECLAST.
|
||||
LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented");
|
||||
|
||||
Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize));
|
||||
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) {
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
const uint64_t RCON = Op->Src[1].Literal();
|
||||
|
||||
auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE);
|
||||
@@ -305,7 +305,7 @@ void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) {
|
||||
}
|
||||
|
||||
Ref Result = AESKeyGenAssistImpl(Op);
|
||||
StoreResult(FPRClass, Op, Result, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
@@ -313,12 +313,12 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) {
|
||||
UnimplementedOp(Op);
|
||||
return;
|
||||
}
|
||||
Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags);
|
||||
Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[1].Literal());
|
||||
|
||||
auto Res = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Res, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
@@ -328,12 +328,12 @@ void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) {
|
||||
}
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags);
|
||||
Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags);
|
||||
const auto Selector = static_cast<uint8_t>(Op->Src[2].Literal());
|
||||
|
||||
Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector & 0b1'0001);
|
||||
StoreResult(FPRClass, Op, Res, OpSize::iInvalid);
|
||||
StoreResultFPR(Op, Res, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
File diff suppressed because it is too large.
Load diff
@@ -28,7 +28,7 @@ class OrderedNode;
|
||||
Ref OpDispatchBuilder::GetX87Top() {
|
||||
// Yes, we are storing 3 bits in a single flag register.
|
||||
// Deal with it
|
||||
return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
return _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
@@ -52,18 +52,18 @@ void OpDispatchBuilder::SetX87FTW(Ref FTW) {
|
||||
FTW = _Orlshr(OpSize::i32Bit, FTW, FTW, 4);
|
||||
|
||||
// ...and that's it. StoreContext implicitly does the final masking.
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, FTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
_StoreContextGPR(OpSize::i8Bit, FTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SetX87Top(Ref Value) {
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
_StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
|
||||
// Float LoaD operation with memory operand
|
||||
void OpDispatchBuilder::FLD(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags);
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
|
||||
Ref ConvertedData = Data;
|
||||
// Convert to 80bit float
|
||||
if (Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
@@ -79,14 +79,14 @@ void OpDispatchBuilder::FLDFromStack(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::FBLD(OpcodeArgs) {
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
_PushStack(ConvertedData, Data, OpSize::i128Bit, true);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FBSTP(OpcodeArgs) {
|
||||
Ref converted = _F80BCDStore(_ReadStackValue(0));
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
_PopStackDestroy();
|
||||
}
|
||||
|
||||
@@ -99,7 +99,7 @@ void OpDispatchBuilder::FLD_Const(OpcodeArgs, NamedVectorConstant K) {
|
||||
void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
const auto ReadWidth = OpSizeFromSrc(Op);
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
Ref Data = LoadSourceGPR_WithOpSize(Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
|
||||
// Sign extend to 64bits
|
||||
if (ReadWidth != OpSize::i64Bit) {
|
||||
@@ -180,7 +180,7 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
|
||||
Data = _F80CVTInt(Size, Data, Truncate);
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Data, Size, OpSize::i8Bit);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, Data, Size, OpSize::i8Bit);
|
||||
|
||||
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
_PopStackDestroy();
|
||||
@@ -206,10 +206,10 @@ void OpDispatchBuilder::FADD(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width);
|
||||
}
|
||||
|
||||
@@ -236,10 +236,10 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width);
|
||||
}
|
||||
|
||||
@@ -273,10 +273,10 @@ void OpDispatchBuilder::FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re
|
||||
// We have one memory argument
|
||||
Ref arg {};
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _F80CVTTo(arg, Width);
|
||||
}
|
||||
|
||||
@@ -314,10 +314,10 @@ void OpDispatchBuilder::FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re
|
||||
// We have one memory argument
|
||||
Ref Arg {};
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTToInt(Arg, Width);
|
||||
} else {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _F80CVTTo(Arg, Width);
|
||||
}
|
||||
|
||||
@@ -340,7 +340,7 @@ Ref OpDispatchBuilder::GetX87FTW_Helper() {
|
||||
// bytes, we use the well-known bit twiddling algorithm:
|
||||
//
|
||||
// https://graphics.stanford.edu/~seander/bithacks.html#InterleaveBMN
|
||||
Ref X = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
Ref X = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 4);
|
||||
X = _And(OpSize::i32Bit, X, Constant(0x0f0f0f0f));
|
||||
X = _Orlshl(OpSize::i32Bit, X, X, 2);
|
||||
@@ -381,41 +381,41 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
_SyncStackToSlow();
|
||||
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Ref Mem = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
{
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMemGPR(Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
|
||||
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -441,20 +441,20 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Ref Mem = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false});
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 1);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
auto NewFSW = _LoadMemGPR(Size, MemLocation, Size);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 2);
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
SetX87FTW(_LoadMemGPR(Size, MemLocation, Size));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -483,61 +483,61 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Dest);
|
||||
Ref Top = GetX87Top();
|
||||
{
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMem(GPRClass, Size, Mem, FCW, Size);
|
||||
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreMemGPR(Size, Mem, FCW, Size);
|
||||
}
|
||||
|
||||
{ _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
|
||||
{ _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); }
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
|
||||
{
|
||||
// FTW
|
||||
_StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction Offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Instruction CS selector (+ Opcode)
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer offset
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
{
|
||||
// Data pointer selector
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
|
||||
_StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
|
||||
auto SevenConst = Constant(7);
|
||||
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
_StoreMem(FPRClass, OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
if (ReducedPrecisionMode) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
_StoreMem(FPRClass, OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
auto topBytes = _VDupElement(OpSize::i128Bit, OpSize::i16Bit, data, 4);
|
||||
_StoreMem(FPRClass, OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
_StoreMemFPR(OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
|
||||
// reset to default
|
||||
FNINIT(Op);
|
||||
@@ -548,8 +548,8 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
if (ReducedPrecisionMode) {
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
@@ -561,11 +561,11 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
}
|
||||
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1);
|
||||
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1);
|
||||
Ref Top = ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
|
||||
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
|
||||
}
|
||||
|
||||
auto SevenConst = Constant(7);
|
||||
@@ -574,14 +574,14 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
Ref Mask = _VLoadTwoGPRs(low, high);
|
||||
const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
// Mask off the top bits
|
||||
Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask);
|
||||
if (ReducedPrecisionMode) {
|
||||
// Convert to double precision
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg);
|
||||
}
|
||||
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
_StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
@@ -590,20 +590,19 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
// ST7 broken in to two parts
|
||||
// Lower 64bits [63:0]
|
||||
// upper 16 bits [79:64]
|
||||
Ref Reg = _LoadMem(FPRClass, OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Ref RegHigh =
|
||||
_LoadMem(FPRClass, OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Ref Reg = _LoadMemFPR(OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Ref RegHigh = _LoadMemFPR(OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1);
|
||||
Reg = _VInsElement(OpSize::i128Bit, OpSize::i16Bit, 4, 0, Reg, RegHigh);
|
||||
if (ReducedPrecisionMode) {
|
||||
Reg = _F80CVT(OpSize::i64Bit, Reg); // Convert to double precision
|
||||
}
|
||||
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
_StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit));
|
||||
}
|
||||
|
||||
// Load / Store Control Word
|
||||
void OpDispatchBuilder::X87FSTCW(OpcodeArgs) {
|
||||
auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
StoreResult(GPRClass, Op, FCW, OpSize::iInvalid);
|
||||
auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
StoreResultGPR(Op, FCW, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
@@ -611,8 +610,8 @@ void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
// to switch for now to slow mode whenever these are manually changed.
|
||||
// Remove the next line and try DF_04.asm in fast path.
|
||||
_StackForceSlow();
|
||||
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
Ref NewFCW = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FXCH(OpcodeArgs) {
|
||||
@@ -648,10 +647,10 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
// Memory arg
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTToInt(arg, Width);
|
||||
} else {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
b = _F80CVTTo(arg, Width);
|
||||
}
|
||||
} else {
|
||||
@@ -767,7 +766,7 @@ Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) {
|
||||
void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) {
|
||||
Ref TopValue = _SyncStackToSlow();
|
||||
Ref StatusWord = ReconstructFSW_Helper(TopValue);
|
||||
StoreResult(GPRClass, Op, StatusWord, OpSize::iInvalid);
|
||||
StoreResultGPR(Op, StatusWord, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::FNCLEX(OpcodeArgs) {
|
||||
@@ -786,12 +785,12 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
|
||||
// Init FCW to 0x037F
|
||||
auto NewFCW = Constant(0x037F);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Set top to zero
|
||||
SetX87Top(Zero);
|
||||
// Tags all get marked as invalid
|
||||
_StoreContext(OpSize::i8Bit, GPRClass, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
_StoreContextGPR(OpSize::i8Bit, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
|
||||
// Reinits the simulated stack
|
||||
_InitStack();
|
||||
|
||||
@@ -29,38 +29,38 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
Ref Mem = MakeSegmentAddress(Op, Op->Src[0]);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MemOffsetType::SXTX, 1);
|
||||
auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MemOffsetType::SXTX, 1);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
|
||||
SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
_StackForceSlow();
|
||||
|
||||
Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref NewFCW = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
// ignore the rounding precision, we're always 64-bit in F64.
|
||||
// extract rounding mode
|
||||
Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW);
|
||||
_SetRoundingMode(roundingMode, false, roundingMode);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
// F64 ops
|
||||
// Float load op with memory operand
|
||||
void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
|
||||
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags);
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
|
||||
// Convert to 64bit float
|
||||
Ref ConvertedData = Data;
|
||||
if (Width == OpSize::i32Bit) {
|
||||
@@ -73,7 +73,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
|
||||
|
||||
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
|
||||
Ref ConvertedData = _F80BCDLoad(Data);
|
||||
ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData);
|
||||
_PushStack(ConvertedData, Data, OpSize::i64Bit, true);
|
||||
@@ -82,7 +82,7 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
|
||||
void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
|
||||
Ref converted = _F80CVTTo(_ReadStackValue(0), OpSize::i64Bit);
|
||||
converted = _F80BCDStore(converted);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
StoreResultFPR_WithOpSize(Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit);
|
||||
_PopStackDestroy();
|
||||
}
|
||||
|
||||
@@ -95,7 +95,7 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
|
||||
const auto ReadWidth = OpSizeFromSrc(Op);
|
||||
|
||||
// Read from memory
|
||||
Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
Ref Data = LoadSourceGPR_WithOpSize(Op, Op->Src[0], ReadWidth, Op->Flags);
|
||||
if (ReadWidth == OpSize::i16Bit) {
|
||||
Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data);
|
||||
}
|
||||
@@ -112,7 +112,7 @@ void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
|
||||
} else {
|
||||
data = _Float_ToGPR_S(Size == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, OpSize::i64Bit, data);
|
||||
}
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, OpSize::i8Bit);
|
||||
StoreResultGPR_WithOpSize(Op, Op->Dest, data, Size, OpSize::i8Bit);
|
||||
|
||||
if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) {
|
||||
_PopStackDestroy();
|
||||
@@ -138,16 +138,16 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
|
||||
Ref arg {};
|
||||
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
@@ -176,16 +176,16 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi
|
||||
Ref arg {};
|
||||
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
}
|
||||
@@ -228,16 +228,16 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
|
||||
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
Arg = _Sbfe(OpSize::i64Bit, 16, 0, Arg);
|
||||
}
|
||||
Arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, Arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
Arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, Arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
@@ -285,16 +285,16 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool
|
||||
|
||||
if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
@@ -332,16 +332,16 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpD
|
||||
} else if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
|
||||
// Memory arg
|
||||
if (Integer) {
|
||||
arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags);
|
||||
if (Width == OpSize::i16Bit) {
|
||||
arg = _Sbfe(OpSize::i64Bit, 16, 0, arg);
|
||||
}
|
||||
b = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i32Bit) {
|
||||
arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
b = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg);
|
||||
} else if (Width == OpSize::i64Bit) {
|
||||
b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
b = LoadSourceFPR(Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
} else {
|
||||
FEX_UNREACHABLE;
|
||||
|
||||
@@ -180,6 +180,9 @@ public:
|
||||
return _StoreMem(FPRClass, Size, Value, Addr, Offset, Align, OffsetType, OffsetScale);
|
||||
}
|
||||
|
||||
IRPair<IROp_StoreMemPair> _StoreMemPairGPR(OpSize Size, Ref Value1, Ref Value2, Ref Addr, uint32_t Offset) {
|
||||
return _StoreMemPair(GPRClass, Size, Value1, Value2, Addr, Offset);
|
||||
}
|
||||
IRPair<IROp_StoreMemPair> _StoreMemPairFPR(OpSize Size, Ref Value1, Ref Value2, Ref Addr, uint32_t Offset) {
|
||||
return _StoreMemPair(FPRClass, Size, Value1, Value2, Addr, Offset);
|
||||
}
|
||||
|
||||
@@ -182,7 +182,7 @@ private:
|
||||
MemOffsetType OffsetType = Op->OffsetType;
|
||||
uint8_t OffsetScale = Op->OffsetScale;
|
||||
|
||||
IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
IREmit->_StoreMemFPR(OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
|
||||
|
||||
// Store the Upper part of the register (the remaining 2 bytes) into memory.
|
||||
@@ -193,7 +193,7 @@ private:
|
||||
.Offset = 8,
|
||||
.AddrSize = OpSize::i64Bit};
|
||||
A = SelectAddressMode(IREmit, A, GPROpSize, Features.SupportsTSOImm9, false, false, OpSize::i16Bit);
|
||||
IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MemOffsetType::SXTX, A.IndexScale);
|
||||
IREmit->_StoreMemGPR(OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MemOffsetType::SXTX, A.IndexScale);
|
||||
}
|
||||
|
||||
void StoreStackMem_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
|
||||
@@ -211,7 +211,7 @@ private:
|
||||
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
|
||||
StackNode = SilenceNaN(StackNode);
|
||||
}
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
IREmit->_StoreMemFPR(Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -252,7 +252,7 @@ private:
|
||||
[[fallthrough]];
|
||||
}
|
||||
case OpSize::i64Bit: {
|
||||
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
IREmit->_StoreMemFPR(Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
|
||||
break;
|
||||
}
|
||||
|
||||
@@ -412,8 +412,7 @@ inline void X87StackOptimization::MigrateToSlowPathIf(bool ShouldMigrate) {
|
||||
|
||||
inline Ref X87StackOptimization::GetTopWithCache_Slow() {
|
||||
if (!TopOffsetCache[0]) {
|
||||
TopOffsetCache[0] =
|
||||
IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
TopOffsetCache[0] = IREmit->_LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
}
|
||||
return TopOffsetCache[0];
|
||||
}
|
||||
@@ -458,7 +457,7 @@ inline void X87StackOptimization::SetTopWithCache_Slow(Ref Value) {
|
||||
|
||||
inline Ref X87StackOptimization::GetFTW() {
|
||||
if (!FTWCached) {
|
||||
FTWCached = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
FTWCached = IREmit->_LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
}
|
||||
return FTWCached;
|
||||
}
|
||||
@@ -481,8 +480,7 @@ inline Ref X87StackOptimization::LoadStackValueAtOffset_Slow(uint8_t Offset) {
|
||||
OrderedNode* TopOffsetAddress = GetOffsetTopAddressWithCache_Slow(Offset);
|
||||
auto Size = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
if (!TopValueCache[Offset]) {
|
||||
TopValueCache[Offset] =
|
||||
IREmit->_LoadMem(FPRClass, Size, TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1);
|
||||
TopValueCache[Offset] = IREmit->_LoadMemFPR(Size, TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1);
|
||||
}
|
||||
return TopValueCache[Offset];
|
||||
}
|
||||
@@ -616,7 +614,7 @@ inline void X87StackOptimization::UpdateTopForPush_Slow() {
|
||||
|
||||
void X87StackOptimization::FlushCachedRegs() {
|
||||
if (FlushTopPending) {
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, TopOffsetCache[0], offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
IREmit->_StoreContextGPR(OpSize::i8Bit, TopOffsetCache[0], offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC);
|
||||
FlushTopPending = false;
|
||||
}
|
||||
|
||||
@@ -624,7 +622,7 @@ void X87StackOptimization::FlushCachedRegs() {
|
||||
for (size_t i = 0; i < FlushValuesPending.size(); i++) {
|
||||
if (FlushValuesPending[i]) {
|
||||
OrderedNode* TopOffsetAddress = GetOffsetTopAddressWithCache_Slow(i);
|
||||
IREmit->_StoreMem(FPRClass, Size, TopValueCache[i], TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1);
|
||||
IREmit->_StoreMemFPR(Size, TopValueCache[i], TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1);
|
||||
// store
|
||||
FlushValuesPending[i] = false;
|
||||
}
|
||||
@@ -675,7 +673,7 @@ void X87StackOptimization::FlushCachedRegs() {
|
||||
}
|
||||
}();
|
||||
|
||||
IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
IREmit->_StoreContextGPR(OpSize::i8Bit, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
|
||||
FTWCached = NewFTW;
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user