From ebb21e86b05f8b84134e0fd391a1be95d9f4afd5 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Thu, 8 Jul 2021 10:06:20 -0700 Subject: [PATCH 1/2] OpcodeDispatcher: Split the opcode handling to multiple files These are some fairly large separate files still. Don't want these to be too terribly small but helps significantly with compile time when working in the OpcodeDispatcher --- External/FEXCore/Source/CMakeLists.txt | 4 + .../Interface/Core/OpcodeDispatcher.cpp | 3949 +---------------- .../Source/Interface/Core/OpcodeDispatcher.h | 26 +- .../Core/OpcodeDispatcher/Crypto.cpp | 58 + .../Interface/Core/OpcodeDispatcher/Flags.cpp | 810 ++++ .../Core/OpcodeDispatcher/Vector.cpp | 2383 ++++++++++ .../Interface/Core/OpcodeDispatcher/X87.cpp | 1276 ++++++ 7 files changed, 4572 insertions(+), 3934 deletions(-) create mode 100644 External/FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp create mode 100644 External/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp create mode 100644 External/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp create mode 100644 External/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp diff --git a/External/FEXCore/Source/CMakeLists.txt b/External/FEXCore/Source/CMakeLists.txt index f4153dc96..2a87d044e 100644 --- a/External/FEXCore/Source/CMakeLists.txt +++ b/External/FEXCore/Source/CMakeLists.txt @@ -81,6 +81,10 @@ set (SRCS Interface/Core/Frontend.cpp Interface/Core/GdbServer.cpp Interface/Core/HostFeatures.cpp + Interface/Core/OpcodeDispatcher/Crypto.cpp + Interface/Core/OpcodeDispatcher/Flags.cpp + Interface/Core/OpcodeDispatcher/Vector.cpp + Interface/Core/OpcodeDispatcher/X87.cpp Interface/Core/OpcodeDispatcher.cpp Interface/Core/X86Tables.cpp Interface/Core/X86DebugInfo.cpp diff --git a/External/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/External/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index 55b0e408a..b91c4bb45 100644 --- a/External/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/External/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -1540,7 +1540,7 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs) { Segment = _LoadContext(2, offsetof(FEXCore::Core::CPUState, fs), GPRClass); } break; - default: + default: LogMan::Msg::E("Unknown segment register: %d", Op->Dest.Data.GPR.GPR); DecodeFailure = true; return; @@ -1940,7 +1940,7 @@ void OpDispatchBuilder::ASHRImmediateOp(OpcodeArgs) { if (Size < 32) { Dest = _Sbfe(Size, 0, Dest); } - + OrderedNode *Src = _Constant(Size, Shift); OrderedNode *Result = _Ashr(Dest, Src); @@ -3892,505 +3892,6 @@ void OpDispatchBuilder::BSROp(OpcodeArgs) { SetRFLAG(ZFSelectOp); } -void OpDispatchBuilder::MOVAPSOp(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - StoreResult(FPRClass, Op, Src, -1); -} - -void OpDispatchBuilder::MOVUPSOp(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 1); - StoreResult(FPRClass, Op, Src, 1); -} - -void OpDispatchBuilder::MOVLHPSOp(OpcodeArgs) { - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, 8); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8); - auto Result = _VInsElement(16, 8, 1, 0, Dest, Src); - StoreResult(FPRClass, Op, Result, 8); -} - -void OpDispatchBuilder::MOVHPDOp(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - // This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 128bit - if (Op->Dest.IsGPR()) { - // If the destination is a GPR then the source is memory - // xmm1[127:64] = src - OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1); - auto Result = _VInsElement(16, 8, 1, 0, Dest, Src); - StoreResult(FPRClass, Op, Result, -1); - } - else { - // In this case memory is the destination and the high bits of the XMM are source - // Mem64 = xmm1[127:64] - auto Result = _VExtractToGPR(16, 8, Src, 1); - StoreResult(GPRClass, Op, Result, -1); - } -} - -void OpDispatchBuilder::MOVLPOp(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8); - if (Op->Dest.IsGPR()) { - // xmm, xmm is movhlps special case - if (Op->Src[0].IsGPR()) { - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, 8, 16); - Src = _VExtractElement(16, 8, Src, 1); - auto Result = _VInsScalarElement(16, 8, 0, Dest, Src); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 16, 16); - } - else { - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, 8, 16); - auto Result = _VInsScalarElement(16, 8, 0, Dest, Src); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 8, 16); - } - } - else { - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 8, 8); - } -} - -void OpDispatchBuilder::MOVSHDUPOp(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8); - OrderedNode *Result = _VInsElement(16, 4, 3, 3, Src, Src); - Result = _VInsElement(16, 4, 2, 3, Result, Src); - Result = _VInsElement(16, 4, 1, 1, Result, Src); - Result = _VInsElement(16, 4, 0, 1, Result, Src); - StoreResult(FPRClass, Op, Result, -1); -} - -void OpDispatchBuilder::MOVSLDUPOp(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8); - OrderedNode *Result = _VInsElement(16, 4, 3, 2, Src, Src); - Result = _VInsElement(16, 4, 2, 2, Result, Src); - Result = _VInsElement(16, 4, 1, 0, Result, Src); - Result = _VInsElement(16, 4, 0, 0, Result, Src); - StoreResult(FPRClass, Op, Result, -1); -} - -void OpDispatchBuilder::MOVSSOp(OpcodeArgs) { - if (Op->Dest.IsGPR() && Op->Src[0].IsGPR()) { - // MOVSS xmm1, xmm2 - OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1); - OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1); - auto Result = _VInsScalarElement(16, 4, 0, Dest, Src); - StoreResult(FPRClass, Op, Result, -1); - } - else if (Op->Dest.IsGPR()) { - // MOVSS xmm1, mem32 - // xmm1[127:0] <- zext(mem32) - OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1); - StoreResult(FPRClass, Op, Src, -1); - } - else { - // MOVSS mem32, xmm1 - OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 4, -1); - } -} - -void OpDispatchBuilder::MOVSDOp(OpcodeArgs) { - if (Op->Dest.IsGPR() && Op->Src[0].IsGPR()) { - // xmm1[63:0] <- xmm2[63:0] - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - auto Result = _VInsScalarElement(16, 8, 0, Dest, Src); - StoreResult(FPRClass, Op, Result, -1); - } - else if (Op->Dest.IsGPR()) { - // xmm1[127:0] <- zext(mem64) - OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 8, Op->Flags, -1); - StoreResult(FPRClass, Op, Src, -1); - } - else { - // In this case memory is the destination and the low bits of the XMM are source - // Mem64 = xmm2[63:0] - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 8, -1); - } -} - -template -void OpDispatchBuilder::PADDQOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - auto ALUOp = _VAdd(Size, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, ALUOp, -1); -} - -template -void OpDispatchBuilder::PSUBQOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - auto ALUOp = _VSub(Size, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, ALUOp, -1); -} - -template -void OpDispatchBuilder::VectorALUOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - auto ALUOp = _VAdd(Size, ElementSize, Dest, Src); - // Overwrite our IR's op type - ALUOp.first->Header.Op = IROp; - - StoreResult(FPRClass, Op, ALUOp, -1); -} - -template -void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - // If OpSize == ElementSize then it only does the lower scalar op - auto ALUOp = _VAdd(ElementSize, ElementSize, Dest, Src); - // Overwrite our IR's op type - ALUOp.first->Header.Op = IROp; - - OrderedNode* Result = ALUOp; - - if (Size != ElementSize) { - // Insert the lower bits - Result = _VInsScalarElement(Size, ElementSize, 0, Dest, Result); - } - - StoreResult(FPRClass, Op, Result, -1); -} - -template -void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - if constexpr (Scalar) { - Size = ElementSize; - } - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - auto ALUOp = _VFSqrt(Size, ElementSize, Src); - // Overwrite our IR's op type - ALUOp.first->Header.Op = IROp; - - if constexpr (Scalar) { - // Insert the lower bits - auto Result = _VInsScalarElement(GetSrcSize(Op), ElementSize, 0, Dest, ALUOp); - StoreResult(FPRClass, Op, Result, -1); - } - else { - StoreResult(FPRClass, Op, ALUOp, -1); - } -} - -void OpDispatchBuilder::MOVQOp(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - // This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 128bit - if (Op->Dest.IsGPR()) { - const auto gpr = Op->Dest.Data.GPR.GPR; - _StoreContext(FPRClass, 8, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][0]), Src); - auto Const = _Constant(0); - _StoreContext(GPRClass, 8, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][1]), Const); - } - else { - // This is simple, just store the result - StoreResult(FPRClass, Op, Src, -1); - } -} - -template -void OpDispatchBuilder::MOVMSKOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - uint8_t NumElements = Size / ElementSize; - - OrderedNode *CurrentVal = _Constant(0); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - for (unsigned i = 0; i < NumElements; ++i) { - // Extract the top bit of the element - OrderedNode *Tmp = _VExtractToGPR(16, ElementSize, Src, i); - Tmp = _Bfe(1, ElementSize * 8 - 1, Tmp); - - // Shift it to the correct location - Tmp = _Lshl(Tmp, _Constant(i)); - - // Or it with the current value - CurrentVal = _Or(CurrentVal, Tmp); - } - StoreResult(GPRClass, Op, CurrentVal, -1); -} - -void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - //TODO: We could remove this VCastFromGOR + VInsGPR pair if we had a VDUPFromGPR instruction that maps directly to AArch64. - auto M = _Constant(0x80'40'20'10'08'04'02'01ULL); - OrderedNode *VMask = _VCastFromGPR(16, 8, M); - VMask = _VInsGPR(16, 8, VMask, M, 1); - - auto VCMP = _VCMPLTZ(Src, 16, 1); - auto VAnd = _VAnd(VCMP, VMask, 16, 1); - - auto VAdd1 = _VAddP(VAnd, VAnd, 16, 1); - auto VAdd2 = _VAddP(VAdd1, VAdd1, 8, 1); - auto VAdd3 = _VAddP(VAdd2, VAdd2, 8, 1); - - StoreResult(GPRClass, Op, _VExtractToGPR(16, 2, VAdd3, 0), -1); -} - -template -void OpDispatchBuilder::PUNPCKLOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - auto ALUOp = _VZip(Size, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, ALUOp, -1); -} - -template -void OpDispatchBuilder::PUNPCKHOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - auto ALUOp = _VZip2(Size, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, ALUOp, -1); -} - -void OpDispatchBuilder::PSHUFBOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - // PSHUFB doesn't 100% match VTBL behaviour - // VTBL will set the element zero if the index is greater than the number of elements - // In the array - // Bit 7 is the only bit that is supposed to set elements to zero with PSHUFB - // Mask the selection bits and top bit correctly - // Bits [6:4] is reserved for 128bit - // Bits [6:3] is reserved for 64bit - if (Size == 8) { - auto MaskVector = _VectorImm(0b1000'0111, Size, 1); - Src = _VAnd(Size, Size, Src, MaskVector); - } - else { - auto MaskVector = _VectorImm(0b1000'1111, Size, 1); - Src = _VAnd(Size, Size, Src, MaskVector); - } - auto Res = _VTBL1(Size, Dest, Src); - StoreResult(FPRClass, Op, Res, -1); -} - -template -void OpDispatchBuilder::PSHUFDOp(OpcodeArgs) { - LOGMAN_THROW_A(ElementSize != 0, "What. No element size?"); - auto Size = GetSrcSize(Op); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - uint8_t Shuffle = Op->Src[1].Data.Literal.Value; - - uint8_t NumElements = Size / ElementSize; - - // 16bit is a bit special of a shuffle - // It only ever operates on half the register - // Then there is a high and low variant of the instruction to determine where the destination goes - // and where the source comes from - if constexpr (HalfSize) { - NumElements /= 2; - } - - uint8_t BaseElement = Low ? 0 : NumElements; - - auto Dest = Src; - for (uint8_t Element = 0; Element < NumElements; ++Element) { - Dest = _VInsElement(Size, ElementSize, BaseElement + Element, BaseElement + (Shuffle & 0b11), Dest, Src); - Shuffle >>= 2; - } - - StoreResult(FPRClass, Op, Dest, -1); -} - -template -void OpDispatchBuilder::SHUFOp(OpcodeArgs) { - LOGMAN_THROW_A(ElementSize != 0, "What. No element size?"); - auto Size = GetSrcSize(Op); - OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - uint8_t Shuffle = Op->Src[1].Data.Literal.Value; - - uint8_t NumElements = Size / ElementSize; - - auto Dest = Src1; - std::array Srcs = { - }; - - for (int i = 0; i < (NumElements >> 1); ++i) { - Srcs[i] = Src1; - } - - for (int i = (NumElements >> 1); i < NumElements; ++i) { - Srcs[i] = Src2; - } - - // 32bit: - // [31:0] = Src1[Selection] - // [63:32] = Src1[Selection] - // [95:64] = Src2[Selection] - // [127:96] = Src2[Selection] - // 64bit: - // [63:0] = Src1[Selection] - // [127:64] = Src2[Selection] - uint8_t SelectionMask = NumElements - 1; - uint8_t ShiftAmount = std::popcount(SelectionMask); - for (uint8_t Element = 0; Element < NumElements; ++Element) { - Dest = _VInsElement(Size, ElementSize, Element, Shuffle & SelectionMask, Dest, Srcs[Element]); - Shuffle >>= ShiftAmount; - } - - StoreResult(FPRClass, Op, Dest, -1); -} - -void OpDispatchBuilder::ANDNOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - // Dest = ~Src1 & Src2 - - Src1 = _VNot(Size, Size, Src1); - auto Dest = _VAnd(Size, Size, Src1, Src2); - - StoreResult(FPRClass, Op, Dest, -1); -} - -template -void OpDispatchBuilder::PINSROp(OpcodeArgs) { - auto Size = GetDstSize(Op); - - OrderedNode *Src{}; - if (Op->Src[0].IsGPR()) { - Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); - } - else { - // If loading from memory then we only load the element size - Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ElementSize, Op->Flags, -1); - } - OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetDstSize(Op), Op->Flags, -1); - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint64_t Index = Op->Src[1].Data.Literal.Value; - - uint8_t NumElements = Size / ElementSize; - Index &= NumElements - 1; - - // This maps 1:1 to an AArch64 NEON Op - auto ALUOp = _VInsGPR(Size, ElementSize, Dest, Src, Index); - StoreResult(FPRClass, Op, ALUOp, -1); -} - -void OpDispatchBuilder::InsertPSOp(OpcodeArgs) { - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint8_t Imm = Op->Src[1].Data.Literal.Value; - uint8_t CountS = (Imm >> 6); - uint8_t CountD = (Imm >> 4) & 0b11; - uint8_t ZMask = Imm & 0xF; - - OrderedNode *Dest{}; - if (ZMask != 0xF) { - // Only need to load destination if it isn't a full zero - Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetDstSize(Op), Op->Flags, -1); - } - - if (!(ZMask & (1 << CountD))) { - // In the case that ZMask overwrites the destination element, then don't even insert - OrderedNode *Src{}; - if (Op->Src[0].IsGPR()) { - Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - } - else { - // If loading from memory then CountS is forced to zero - CountS = 0; - Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1); - } - - Dest = _VInsElement(GetDstSize(Op), 4, CountD, CountS, Dest, Src); - } - - // ZMask happens after insert - if (ZMask == 0xF) { - Dest = _VectorImm(0, 16, 4); - } - else if (ZMask) { - auto Zero = _VectorImm(0, 16, 4); - for (size_t i = 0; i < 4; ++i) { - if (ZMask & (1 << i)) { - Dest = _VInsElement(GetDstSize(Op), 4, i, 0, Dest, Zero); - } - } - } - - StoreResult(FPRClass, Op, Dest, -1); -} - -template -void OpDispatchBuilder::PExtrOp(OpcodeArgs) { - const auto Size = GetSrcSize(Op); - - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint64_t Index = Op->Src[1].Data.Literal.Value; - - const uint8_t NumElements = Size / ElementSize; - Index &= NumElements - 1; - - OrderedNode *Result = _VExtractToGPR(16, ElementSize, Src, Index); - - if (Op->Dest.IsGPR()) { - const uint8_t GPRSize = CTX->GetGPRSize(); - - // If we are storing to a GPR then we zero extend it - if constexpr (ElementSize < 4) { - Result = _Bfe(GPRSize, ElementSize * 8, 0, Result); - } - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, -1); - } - else { - // If we are storing to memory then we store the size of the element extracted - StoreResult(GPRClass, Op, Result, -1); - } -} - -template -void OpDispatchBuilder::PSIGN(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - auto ZeroVec = _VectorZero(Size); - auto NegVec = _VNeg(Size, ElementSize, Dest); - - OrderedNode *CmpLT = _VCMPLTZ(Size, ElementSize, Src); - OrderedNode *CmpEQ = _VCMPEQZ(Size, ElementSize, Src); - OrderedNode *CmpGT = _VCMPGTZ(Size, ElementSize, Src); - - // Negative elements return -dest - CmpLT = _VAnd(Size, ElementSize, CmpLT, NegVec); - - // Zero elements return 0 - CmpEQ = _VAnd(Size, ElementSize, CmpEQ, ZeroVec); - - // Positive elements return dest - CmpGT = _VAnd(Size, ElementSize, CmpGT, Dest); - - // Or our results - OrderedNode *Res = _VOr(Size, ElementSize, CmpGT, _VOr(Size, ElementSize, CmpLT, CmpEQ)); - StoreResult(FPRClass, Op, Res, -1); -} - void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) { // CMPXCHG ModRM, reg, {RAX} // MemData = *ModRM.dest @@ -4477,7 +3978,7 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) { } else { HandledLock = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK; - + OrderedNode *Src3{}; OrderedNode *Src3Lower{}; if (GPRSize == 8 && Size == 4) { @@ -5017,817 +4518,6 @@ void OpDispatchBuilder::ResetWorkingList() { CurrentCodeBlock = nullptr; } -template -void OpDispatchBuilder::SetRFLAG(OrderedNode *Value) { - flagsOp = FLAGS_OP_NONE; - _StoreFlag(_Bfe(1, 0, Value), BitOffset); -} -void OpDispatchBuilder::SetRFLAG(OrderedNode *Value, unsigned BitOffset) { - flagsOp = FLAGS_OP_NONE; - _StoreFlag(_Bfe(1, 0, Value), BitOffset); -} - -OrderedNode *OpDispatchBuilder::GetRFLAG(unsigned BitOffset) { - return _LoadFlag(BitOffset); -} - -constexpr std::array FlagOffsets = { - FEXCore::X86State::RFLAG_CF_LOC, - FEXCore::X86State::RFLAG_PF_LOC, - FEXCore::X86State::RFLAG_AF_LOC, - FEXCore::X86State::RFLAG_ZF_LOC, - FEXCore::X86State::RFLAG_SF_LOC, - FEXCore::X86State::RFLAG_TF_LOC, - FEXCore::X86State::RFLAG_IF_LOC, - FEXCore::X86State::RFLAG_DF_LOC, - FEXCore::X86State::RFLAG_OF_LOC, - FEXCore::X86State::RFLAG_IOPL_LOC, - FEXCore::X86State::RFLAG_NT_LOC, - FEXCore::X86State::RFLAG_RF_LOC, - FEXCore::X86State::RFLAG_VM_LOC, - FEXCore::X86State::RFLAG_AC_LOC, - FEXCore::X86State::RFLAG_VIF_LOC, - FEXCore::X86State::RFLAG_VIP_LOC, - FEXCore::X86State::RFLAG_ID_LOC, -}; - -void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) { - uint8_t NumFlags = FlagOffsets.size(); - if (Lower8) { - NumFlags = 5; - } - auto OneConst = _Constant(1); - for (int i = 0; i < NumFlags; ++i) { - auto Tmp = _And(_Lshr(Src, _Constant(FlagOffsets[i])), OneConst); - SetRFLAG(Tmp, FlagOffsets[i]); - } -} - -OrderedNode *OpDispatchBuilder::GetPackedRFLAG(bool Lower8) { - OrderedNode *Original = _Constant(2); - uint8_t NumFlags = FlagOffsets.size(); - if (Lower8) { - NumFlags = 5; - } - - for (int i = 0; i < NumFlags; ++i) { - OrderedNode *Flag = _LoadFlag(FlagOffsets[i]); - Flag = _Bfe(4, 32, 0, Flag); - Flag = _Lshl(Flag, _Constant(FlagOffsets[i])); - Original = _Or(Original, Flag); - } - return Original; -} - -void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) { - auto Size = GetSrcSize(Op) * 8; - // AF - { - OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); - AFRes = _Bfe(1, 4, AFRes); - SetRFLAG(AFRes); - } - - // SF - { - auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); - - auto LshrOp = _Lshr(Res, SignBitConst); - SetRFLAG(LshrOp); - } - - // PF - if (!CTX->Config.ABINoPF) { - auto PopCountOp = _Popcount(_And(Res, _Constant(0xFF))); - - auto XorOp = _Xor(PopCountOp, _Constant(1)); - SetRFLAG(XorOp); - } else { - _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); - } - - // ZF - { - auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, _Constant(0), _Constant(1), _Constant(0)); - SetRFLAG(SelectOp); - } - - // CF - // Unsigned - { - auto SelectOpLT = _Select(FEXCore::IR::COND_ULT, Res, Src2, _Constant(1), _Constant(0)); - auto SelectOpLE = _Select(FEXCore::IR::COND_ULE, Res, Src2, _Constant(1), _Constant(0)); - auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, _Constant(1), SelectOpLE, SelectOpLT); - SetRFLAG(SelectCF); - } - - // OF - // Signed - { - auto NegOne = _Constant(~0ULL); - auto XorOp1 = _Xor(_Xor(Src1, Src2), NegOne); - auto XorOp2 = _Xor(Res, Src1); - OrderedNode *AndOp1 = _And(XorOp1, XorOp2); - - switch (Size) { - case 8: - AndOp1 = _Bfe(1, 7, AndOp1); - break; - case 16: - AndOp1 = _Bfe(1, 15, AndOp1); - break; - case 32: - AndOp1 = _Bfe(1, 31, AndOp1); - break; - case 64: - AndOp1 = _Bfe(1, 63, AndOp1); - break; - default: LOGMAN_MSG_A("Unknown BFESize: %d", Size); break; - } - SetRFLAG(AndOp1); - } -} - -void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) { - // AF - { - OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); - AFRes = _Bfe(1, 4, AFRes); - SetRFLAG(AFRes); - } - - // SF - { - auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); - - auto LshrOp = _Lshr(Res, SignBitConst); - SetRFLAG(LshrOp); - } - - // PF - if (!CTX->Config.ABINoPF) { - auto PopCountOp = _Popcount(_And(Res, _Constant(0xFF))); - - auto XorOp = _Xor(PopCountOp, _Constant(1)); - SetRFLAG(XorOp); - } else { - _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); - } - - // ZF - { - auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, _Constant(0), _Constant(1), _Constant(0)); - SetRFLAG(SelectOp); - } - - // CF - // Unsigned - { - auto SelectOpLT = _Select(FEXCore::IR::COND_UGT, Res, Src1, _Constant(1), _Constant(0)); - auto SelectOpLE = _Select(FEXCore::IR::COND_UGE, Res, Src1, _Constant(1), _Constant(0)); - auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, _Constant(1), SelectOpLE, SelectOpLT); - SetRFLAG(SelectCF); - } - - // OF - // Signed - { - auto XorOp1 = _Xor(Src1, Src2); - auto XorOp2 = _Xor(Res, Src1); - OrderedNode *AndOp1 = _And(XorOp1, XorOp2); - - switch (GetSrcSize(Op)) { - case 1: - AndOp1 = _Bfe(1, 7, AndOp1); - break; - case 2: - AndOp1 = _Bfe(1, 15, AndOp1); - break; - case 4: - AndOp1 = _Bfe(1, 31, AndOp1); - break; - case 8: - AndOp1 = _Bfe(1, 63, AndOp1); - break; - default: LOGMAN_MSG_A("Unknown BFESize: %d", GetSrcSize(Op)); break; - } - SetRFLAG(AndOp1); - } -} - -void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) { - // AF - { - OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); - AFRes = _Bfe(1, 4, AFRes); - SetRFLAG(AFRes); - } - - // SF - { - auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); - - auto LshrOp = _Lshr(Res, SignBitConst); - SetRFLAG(LshrOp); - } - - // PF - if (!CTX->Config.ABINoPF) { - auto EightBitMask = _Constant(0xFF); - auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, _Constant(1)); - SetRFLAG(XorOp); - } else { - _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); - } - - // ZF - { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, ZeroConst, OneConst, ZeroConst); - SetRFLAG(SelectOp); - } - - // CF - if (UpdateCF) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - - auto SelectOp = _Select(FEXCore::IR::COND_ULT, - Src1, Src2, OneConst, ZeroConst); - - SetRFLAG(SelectOp); - } - // OF - { - auto XorOp1 = _Xor(Src1, Src2); - auto XorOp2 = _Xor(Res, Src1); - OrderedNode *FinalAnd = _And(XorOp1, XorOp2); - - FinalAnd = _Bfe(1, GetSrcSize(Op) * 8 - 1, FinalAnd); - - SetRFLAG(FinalAnd); - } -} - -void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) { - // AF - { - OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); - AFRes = _Bfe(1, 4, AFRes); - SetRFLAG(AFRes); - } - - // SF - { - auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); - - auto LshrOp = _Lshr(Res, SignBitConst); - SetRFLAG(LshrOp); - } - - // PF - if (!CTX->Config.ABINoPF) { - auto EightBitMask = _Constant(0xFF); - auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, _Constant(1)); - SetRFLAG(XorOp); - } else { - _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); - } - - // ZF - { - auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, _Constant(0), _Constant(1), _Constant(0)); - SetRFLAG(SelectOp); - } - // CF - if (UpdateCF) { - auto SelectOp = _Select(FEXCore::IR::COND_ULT, Res, Src2, _Constant(1), _Constant(0)); - - SetRFLAG(SelectOp); - } - - // OF - { - auto NegOne = _Constant(~0ULL); - auto XorOp1 = _Xor(_Xor(Src1, Src2), NegOne); - auto XorOp2 = _Xor(Res, Src1); - - OrderedNode *AndOp1 = _And(XorOp1, XorOp2); - - switch (GetSrcSize(Op)) { - case 1: - AndOp1 = _Bfe(1, 7, AndOp1); - break; - case 2: - AndOp1 = _Bfe(1, 15, AndOp1); - break; - case 4: - AndOp1 = _Bfe(1, 31, AndOp1); - break; - case 8: - AndOp1 = _Bfe(1, 63, AndOp1); - break; - default: LOGMAN_MSG_A("Unknown BFESize: %d", GetSrcSize(Op)); break; - } - SetRFLAG(AndOp1); - } -} - -void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *High) { - // PF/AF/ZF/SF - // Undefined - { - SetRFLAG(_Constant(0)); - SetRFLAG(_Constant(0)); - SetRFLAG(_Constant(0)); - SetRFLAG(_Constant(0)); - } - - // CF/OF - { - // CF and OF are set if the result of the operation can't be fit in to the destination register - // If the value can fit then the top bits will be zero - - auto SignBit = _Sbfe(1, GetSrcSize(Op) * 8 - 1, Res); - - auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, SignBit, _Constant(0), _Constant(1)); - - SetRFLAG(SelectOp); - SetRFLAG(SelectOp); - } -} - -void OpDispatchBuilder::GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *High) { - // AF/SF/PF/ZF - // Undefined - { - SetRFLAG(_Constant(0)); - SetRFLAG(_Constant(0)); - SetRFLAG(_Constant(0)); - SetRFLAG(_Constant(0)); - } - - // CF/OF - { - // CF and OF are set if the result of the operation can't be fit in to the destination register - // The result register will be all zero if it can't fit due to how multiplication behaves - - auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, _Constant(0), _Constant(0), _Constant(1)); - - SetRFLAG(SelectOp); - SetRFLAG(SelectOp); - } -} - -void OpDispatchBuilder::GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - // AF - { - // Undefined - // Set to zero anyway - SetRFLAG(_Constant(0)); - } - - // SF - { - auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); - - auto LshrOp = _Lshr(Res, SignBitConst); - SetRFLAG(LshrOp); - } - - // PF - if (!CTX->Config.ABINoPF) { - auto EightBitMask = _Constant(0xFF); - auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, _Constant(1)); - SetRFLAG(XorOp); - } else { - _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); - } - - // ZF - { - auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, _Constant(0), _Constant(1), _Constant(0)); - SetRFLAG(SelectOp); - } - - // CF/OF - { - SetRFLAG(_Constant(0)); - SetRFLAG(_Constant(0)); - } -} - -#define COND_FLAG_SET(cond, flag, newflag) \ -auto oldflag = GetRFLAG(FEXCore::X86State::flag);\ -auto newval = _Select(FEXCore::IR::COND_EQ, cond, _Constant(0), oldflag, newflag);\ -SetRFLAG(newval); - -void OpDispatchBuilder::GenerateFlags_ShiftLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - // CF - { - // Extract the last bit shifted in to CF - auto Size = _Constant(GetSrcSize(Op) * 8); - auto ShiftAmt = _Sub(Size, Src2); - auto LastBit = _And(_Lshr(Src1, ShiftAmt), _Constant(1)); - COND_FLAG_SET(Src2, RFLAG_CF_LOC, LastBit); - } - - // PF - if (!CTX->Config.ABINoPF) { - auto EightBitMask = _Constant(0xFF); - auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, _Constant(1)); - COND_FLAG_SET(Src2, RFLAG_PF_LOC, XorOp); - } else { - _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); - } - - // AF - { - // Undefined - // Set to zero anyway - COND_FLAG_SET(Src2, RFLAG_AF_LOC, _Constant(0)); - } - - // ZF - { - auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, _Constant(0), _Constant(1), _Constant(0)); - COND_FLAG_SET(Src2, RFLAG_ZF_LOC, SelectOp); - } - - // SF - { - auto val = _Bfe(1, GetSrcSize(Op) * 8 - 1, Res); - COND_FLAG_SET(Src2, RFLAG_SF_LOC, val); - } - - // OF - { - // In the case of left shift. OF is only set from the result of XOR - // When Shift > 1 then OF is undefined - auto val = _Bfe(1, GetSrcSize(Op) * 8 - 1, _Xor(Src1, Res)); - COND_FLAG_SET(Src2, RFLAG_OF_LOC, val); - } -} - -void OpDispatchBuilder::GenerateFlags_ShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - // CF - { - // Extract the last bit shifted in to CF - auto ShiftAmt = _Sub(Src2, _Constant(1)); - auto LastBit = _And(_Lshr(Src1, ShiftAmt), _Constant(1)); - COND_FLAG_SET(Src2, RFLAG_CF_LOC, LastBit); - } - - // PF - if (!CTX->Config.ABINoPF) { - auto EightBitMask = _Constant(0xFF); - auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, _Constant(1)); - COND_FLAG_SET(Src2, RFLAG_PF_LOC, XorOp); - } else { - _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); - } - - // AF - { - // Undefined - // Set to zero anyway - COND_FLAG_SET(Src2, RFLAG_AF_LOC, _Constant(0)); - } - - // ZF - { - auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, _Constant(0), _Constant(1), _Constant(0)); - COND_FLAG_SET(Src2, RFLAG_ZF_LOC, SelectOp); - } - - // SF - { - auto val =_Bfe(1, GetSrcSize(Op) * 8 - 1, Res); - COND_FLAG_SET(Src2, RFLAG_SF_LOC, val); - } - - // OF - { - // Only defined when Shift is 1 else undefined - // OF flag is set if a sign change occurred - auto val = _Bfe(1, GetSrcSize(Op) * 8 - 1, _Xor(Src1, Res)); - COND_FLAG_SET(Src2, RFLAG_OF_LOC, val); - } -} - -void OpDispatchBuilder::GenerateFlags_SignShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - // CF - { - // Extract the last bit shifted in to CF - auto ShiftAmt = _Sub(Src2, _Constant(1)); - auto LastBit = _And(_Lshr(Src1, ShiftAmt), _Constant(1)); - COND_FLAG_SET(Src2, RFLAG_CF_LOC, LastBit); - } - - // PF - if (!CTX->Config.ABINoPF) { - auto EightBitMask = _Constant(0xFF); - auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, _Constant(1)); - COND_FLAG_SET(Src2, RFLAG_PF_LOC, XorOp); - } else { - _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); - } - - // AF - { - // Undefined - // Set to zero anyway - COND_FLAG_SET(Src2, RFLAG_AF_LOC, _Constant(0)); - } - - // ZF - { - auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, _Constant(0), _Constant(1), _Constant(0)); - COND_FLAG_SET(Src2, RFLAG_ZF_LOC, SelectOp); - } - - // SF - { - auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); - - auto LshrOp = _Lshr(Res, SignBitConst); - COND_FLAG_SET(Src2, RFLAG_SF_LOC, LshrOp); - } - - // OF - { - COND_FLAG_SET(Src2, RFLAG_OF_LOC, _Constant(0)); - } -} - -void OpDispatchBuilder::GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) { - // No flags changed if shift is zero - if (Shift == 0) return; - - // CF - { - // Extract the last bit shifted in to CF - SetRFLAG(_Bfe(1, GetSrcSize(Op) * 8 - Shift, Src1)); - } - - // PF - if (!CTX->Config.ABINoPF) { - auto EightBitMask = _Constant(0xFF); - auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, _Constant(1)); - SetRFLAG(XorOp); - } else { - _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); - } - - // AF - { - // Undefined - // Set to zero anyway - SetRFLAG(_Constant(0)); - } - - // ZF - { - auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, _Constant(0), _Constant(1), _Constant(0)); - SetRFLAG(SelectOp); - } - - // SF - { - auto LshrOp = _Bfe(1, GetSrcSize(Op) * 8 - 1, Res); - - SetRFLAG(LshrOp); - - // OF - // In the case of left shift. OF is only set from the result of XOR - if (Shift == 1) { - auto SourceBit = _Bfe(1, GetSrcSize(Op) * 8 - 1, Src1); - SetRFLAG(_Xor(SourceBit, LshrOp)); - } - } -} - -void OpDispatchBuilder::GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) { - // No flags changed if shift is zero - if (Shift == 0) return; - - // CF - { - // Extract the last bit shifted in to CF - SetRFLAG(_Bfe(1, Shift-1, Src1)); - } - - // PF - if (!CTX->Config.ABINoPF) { - auto EightBitMask = _Constant(0xFF); - auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, _Constant(1)); - SetRFLAG(XorOp); - } else { - _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); - } - - // AF - { - // Undefined - // Set to zero anyway - SetRFLAG(_Constant(0)); - } - - // ZF - { - auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, _Constant(0), _Constant(1), _Constant(0)); - SetRFLAG(SelectOp); - } - - // SF - { - auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); - - auto LshrOp = _Lshr(Res, SignBitConst); - SetRFLAG(LshrOp); - - // OF - // Only defined when Shift is 1 else undefined - // Only is set if the top bit was set to 1 when shifted - // So it is set to same value as SF - if (Shift == 1) { - SetRFLAG(_Constant(0)); - } - } -} - -void OpDispatchBuilder::GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) { - // No flags changed if shift is zero - if (Shift == 0) return; - - // CF - { - // Extract the last bit shifted in to CF - SetRFLAG(_Bfe(1, Shift-1, Src1)); - } - - // PF - if (!CTX->Config.ABINoPF) { - auto EightBitMask = _Constant(0xFF); - auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, _Constant(1)); - SetRFLAG(XorOp); - } else { - _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); - } - - // AF - { - // Undefined - // Set to zero anyway - SetRFLAG(_Constant(0)); - } - - // ZF - { - auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, _Constant(0), _Constant(1), _Constant(0)); - SetRFLAG(SelectOp); - } - - // SF - { - auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); - - auto LshrOp = _Lshr(Res, SignBitConst); - SetRFLAG(LshrOp); - } - - // OF - { - // Only defined when Shift is 1 else undefined - // Is set to the MSB of the original value - if (Shift == 1) { - SetRFLAG(_Bfe(1, GetSrcSize(Op) * 8 - 1, Src1)); - } - } -} - -void OpDispatchBuilder::GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - auto OpSize = GetSrcSize(Op) * 8; - - // Extract the last bit shifted in to CF - auto NewCF = _Bfe(1, OpSize - 1, Res); - - // CF - { - auto OldCF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC); - auto CF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldCF, NewCF); - - // Extract the last bit shifted in to CF - SetRFLAG(CF); - } - - // OF - { - auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC); - - // OF is set to the XOR of the new CF bit and the most significant bit of the result - auto NewOF = _Xor(_Bfe(1, OpSize - 2, Res), NewCF); - - // If shift == 0, don't update flags - auto OF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldOF, NewOF); - - SetRFLAG(OF); - } -} - -void OpDispatchBuilder::GenerateFlags_RotateLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - auto OpSize = GetSrcSize(Op) * 8; - - // Extract the last bit shifted in to CF - //auto Size = _Constant(GetSrcSize(Res) * 8); - //auto ShiftAmt = _Sub(Size, Src2); - auto NewCF = _Bfe(1, 0, Res); - - // CF - { - auto OldCF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC); - auto CF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldCF, NewCF); - - // Extract the last bit shifted in to CF - SetRFLAG(CF); - } - - // OF - { - auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC); - // OF is set to the XOR of the new CF bit and the most significant bit of the result - auto NewOF = _Xor(_Bfe(1, OpSize - 1, Res), NewCF); - - auto OF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldOF, NewOF); - - // If shift == 0, don't update flags - SetRFLAG(OF); - } -} - -void OpDispatchBuilder::GenerateFlags_RotateRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) { - if (Shift == 0) return; - - auto OpSize = GetSrcSize(Op) * 8; - - auto NewCF = _Bfe(1, OpSize - Shift, Src1); - - // CF - { - // Extract the last bit shifted in to CF - SetRFLAG(NewCF); - } - - // OF - { - if (Shift == 1) { - // OF is set to the XOR of the new CF bit and the most significant bit of the result - SetRFLAG(_Xor(_Bfe(1, OpSize - 1, Res), NewCF)); - } - } -} - -void OpDispatchBuilder::GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) { - if (Shift == 0) return; - - auto OpSize = GetSrcSize(Op) * 8; - - // CF - { - // Extract the last bit shifted in to CF - SetRFLAG(_Bfe(1, Shift, Src1)); - } - - // OF - { - if (Shift == 1) { - // OF is the top two MSBs XOR'd together - SetRFLAG(_Xor(_Bfe(1, OpSize - 1, Src1), _Bfe(1, OpSize - 2, Src1))); - } - } -} - void OpDispatchBuilder::UnhandledOp(OpcodeArgs) { DecodeFailure = true; } @@ -5838,11 +4528,6 @@ void OpDispatchBuilder::MOVGPROp(OpcodeArgs) { StoreResult(GPRClass, Op, Src, 1); } -void OpDispatchBuilder::MOVVectorOp(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 1); - StoreResult(FPRClass, Op, Src, 1); -} - void OpDispatchBuilder::ALUOp(OpcodeArgs) { bool RequiresMask = false; FEXCore::IR::IROps IROp; @@ -6037,359 +4722,6 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) { } } -template -void OpDispatchBuilder::PSRLDOp(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[SrcIndex], Op->Flags, -1); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - auto Size = GetSrcSize(Op); - - OrderedNode *Result{}; - - if constexpr (Scalar) { - // Incoming element size for the shift source is always 8 - auto MaxShift = _VectorImm(ElementSize * 8, 8, 8); - Src = _VUMin(8, 8, MaxShift, Src); - Result = _VUShrS(Size, ElementSize, Dest, Src); - } - else { - Result = _VUShr(Size, ElementSize, Dest, Src); - } - - StoreResult(FPRClass, Op, Result, -1); -} - -template -void OpDispatchBuilder::PSRLI(OpcodeArgs) { - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint64_t ShiftConstant = Op->Src[1].Data.Literal.Value; - - auto Size = GetSrcSize(Op); - - auto Shift = _VUShrI(Size, ElementSize, Dest, ShiftConstant); - StoreResult(FPRClass, Op, Shift, -1); -} - -template -void OpDispatchBuilder::PSLLI(OpcodeArgs) { - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint64_t ShiftConstant = Op->Src[1].Data.Literal.Value; - - auto Size = GetSrcSize(Op); - - auto Shift = _VShlI(Size, ElementSize, Dest, ShiftConstant); - StoreResult(FPRClass, Op, Shift, -1); -} - -template -void OpDispatchBuilder::PSLL(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[SrcIndex], Op->Flags, -1); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - auto Size = GetDstSize(Op); - - OrderedNode *Result{}; - - if constexpr (Scalar) { - // Incoming element size for the shift source is always 8 - auto MaxShift = _VectorImm(ElementSize * 8, 8, 8); - Src = _VUMin(8, 8, MaxShift, Src); - Result = _VUShlS(Size, ElementSize, Dest, Src); - } - else { - Result = _VUShl(Size, ElementSize, Dest, Src); - } - - StoreResult(FPRClass, Op, Result, -1); -} - -template -void OpDispatchBuilder::PSRAOp(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[SrcIndex], Op->Flags, -1); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - auto Size = GetDstSize(Op); - - OrderedNode *Result{}; - - if constexpr (Scalar) { - // Incoming element size for the shift source is always 8 - auto MaxShift = _VectorImm(ElementSize * 8, 8, 8); - Src = _VUMin(8, 8, MaxShift, Src); - Result = _VSShrS(Size, ElementSize, Dest, Src); - } - else { - Result = _VSShr(Size, ElementSize, Dest, Src); - } - - StoreResult(FPRClass, Op, Result, -1); -} - -void OpDispatchBuilder::PSRLDQ(OpcodeArgs) { - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint64_t Shift = Op->Src[1].Data.Literal.Value; - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - auto Size = GetDstSize(Op); - - auto Result = _VSRI(Size, 16, Dest, Shift); - StoreResult(FPRClass, Op, Result, -1); -} - -void OpDispatchBuilder::PSLLDQ(OpcodeArgs) { - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint64_t Shift = Op->Src[1].Data.Literal.Value; - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - auto Size = GetDstSize(Op); - - auto Result = _VSLI(Size, 16, Dest, Shift); - StoreResult(FPRClass, Op, Result, -1); -} - -template -void OpDispatchBuilder::PSRAIOp(OpcodeArgs) { - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint64_t Shift = Op->Src[1].Data.Literal.Value; - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - auto Size = GetDstSize(Op); - - auto Result = _VSShrI(Size, ElementSize, Dest, Shift); - StoreResult(FPRClass, Op, Result, -1); -} - -template -void OpDispatchBuilder::PAVGOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - - auto Result = _VURAvg(Size, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, Result, -1); -} - -void OpDispatchBuilder::MOVDDUPOp(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - OrderedNode *Res = _SplatVector2(Src); - StoreResult(FPRClass, Op, Res, -1); -} - -template -void OpDispatchBuilder::CVTGPR_To_FPR(OpcodeArgs) { - OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); - - size_t GPRSize = GetSrcSize(Op); - - Src = _Float_FromGPR_S(DstElementSize, GPRSize, Src); - - OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1); - - Src = _VInsScalarElement(16, DstElementSize, 0, Dest, Src); - - StoreResult(FPRClass, Op, Src, -1); -} - -template -void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - // GPR size is determined by REX.W - // Source Element size is determined by instruction - size_t GPRSize = GetDstSize(Op); - - size_t ElementSize = SrcElementSize; - if constexpr (HostRoundingMode) { - Src = _Float_ToGPR_S(Src, ElementSize, GPRSize); - } - else { - Src = _Float_ToGPR_ZS(Src, ElementSize, GPRSize); - } - - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Src, GPRSize, -1); -} - -template -void OpDispatchBuilder::Vector_CVT_Int_To_Float(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - size_t ElementSize = SrcElementSize; - size_t Size = GetDstSize(Op); - if constexpr (Widen) { - Src = _VSXTL(Src, Size, ElementSize); - ElementSize <<= 1; - } - - Src = _Vector_SToF(Src, Size, ElementSize); - - StoreResult(FPRClass, Op, Src, -1); -} - -template -void OpDispatchBuilder::Vector_CVT_Float_To_Int(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - size_t ElementSize = SrcElementSize; - size_t Size = GetDstSize(Op); - - if constexpr (Narrow) { - Src = _Vector_FToF(Size, SrcElementSize >> 1, SrcElementSize, Src); - ElementSize >>= 1; - } - - if constexpr (HostRoundingMode) { - Src = _Vector_FToS(Src, Size, ElementSize); - } - else { - Src = _Vector_FToZS(Src, Size, ElementSize); - } - - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, -1); -} - -template -void OpDispatchBuilder::Scalar_CVT_Float_To_Float(OpcodeArgs) { - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - Src = _Float_FToF(DstElementSize, SrcElementSize, Src); - Src = _VInsScalarElement(16, DstElementSize, 0, Dest, Src); - - StoreResult(FPRClass, Op, Src, -1); -} - -template -void OpDispatchBuilder::Vector_CVT_Float_To_Float(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - size_t Size = GetDstSize(Op); - - if constexpr (DstElementSize > SrcElementSize) { - Src = _Vector_FToF(Size, SrcElementSize << 1, SrcElementSize, Src); - } - else { - Src = _Vector_FToF(Size, SrcElementSize >> 1, SrcElementSize, Src); - } - - StoreResult(FPRClass, Op, Src, -1); -} - -template -void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - size_t ElementSize = SrcElementSize; - size_t DstSize = GetDstSize(Op); - if constexpr (Widen) { - Src = _VSXTL(Src, DstSize, ElementSize); - ElementSize <<= 1; - } - - if constexpr (Signed) { - Src = _Vector_SToF(Src, DstSize, ElementSize); - } - else { - Src = _Vector_UToF(Src, DstSize, ElementSize); - } - - OrderedNode *Dest{}; - if constexpr (Widen) { - Dest = Src; - } - else { - Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1); - // Insert the lower bits - Dest = _VInsElement(GetDstSize(Op), 8, 0, 0, Dest, Src); - } - - StoreResult(FPRClass, Op, Dest, -1); -} - -template -void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - size_t ElementSize = SrcElementSize; - size_t Size = GetDstSize(Op); - - if constexpr (Narrow) { - Src = _Vector_FToF(Size, SrcElementSize >> 1, SrcElementSize, Src); - ElementSize >>= 1; - } - - if constexpr (HostRoundingMode) { - Src = _Vector_FToS(Src, Size, ElementSize); - } - else { - Src = _Vector_FToZS(Src, Size, ElementSize); - } - - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, -1); -} - -void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) { - // Until we get correct PHI nodes this is required to be a loop unroll - const auto GPRSize = CTX->GetGPRSize(); - const auto Size = uint32_t{GetSrcSize(Op)} * 8; - - OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); - OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1); - - OrderedNode *MemDest = _LoadContext(GPRSize, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RDI]), GPRClass); - - const size_t NumElements = Size / 64; - for (size_t Element = 0; Element < NumElements; ++Element) { - // Extract the current element - auto SrcElement = _VExtractToGPR(GetSrcSize(Op), 8, Src, Element); - auto DestElement = _VExtractToGPR(GetSrcSize(Op), 8, Dest, Element); - - constexpr size_t NumSelectBits = 64 / 8; - for (size_t Select = 0; Select < NumSelectBits; ++Select) { - auto SelectMask = _Bfe(1, 8 * Select + 7, SrcElement); - auto CondJump = _CondJump(SelectMask, {COND_EQ}); - auto StoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock()); - SetFalseJumpTarget(CondJump, StoreBlock); - SetCurrentCodeBlock(StoreBlock); - { - auto DestByte = _Bfe(8, 8 * Select, DestElement); - auto MemLocation = _Add(MemDest, _Constant(Element * 8 + Select)); - _StoreMemAutoTSO(GPRClass, 1, MemLocation, DestByte, 1); - } - auto Jump = _Jump(); - auto NextJumpTarget = CreateNewCodeBlockAfter(StoreBlock); - SetJumpTarget(Jump, NextJumpTarget); - SetTrueJumpTarget(CondJump, NextJumpTarget); - SetCurrentCodeBlock(NextJumpTarget); - } - } -} - -void OpDispatchBuilder::MOVBetweenGPR_FPR(OpcodeArgs) { - if (Op->Dest.IsGPR() && - Op->Dest.Data.GPR.GPR >= FEXCore::X86State::REG_XMM_0) { - OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); - // zext to 128bit - auto Converted = _VCastFromGPR(16, GetSrcSize(Op), Src); - StoreResult(FPRClass, Op, Op->Dest, Converted, -1); - } - else { - // Destination is GPR or mem - // Extract from XMM first - auto ElementSize = GetDstSize(Op); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0],Op->Flags, -1); - - Src = _VExtractToGPR(GetSrcSize(Op), ElementSize, Src, 0); - - StoreResult(GPRClass, Op, Op->Dest, Src, -1); - } -} - void OpDispatchBuilder::TZCNT(OpcodeArgs) { OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); @@ -6422,1864 +4754,12 @@ void OpDispatchBuilder::LZCNT(OpcodeArgs) { SetRFLAG(_Bfe(1, GetSrcSize(Op) * 8 - 1, Src)); } -template -void OpDispatchBuilder::VFCMPOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetDstSize(Op), Op->Flags, -1); - OrderedNode *Src2{}; - if constexpr (Scalar) { - Src2 = _VExtractElement(GetDstSize(Op), Size, Dest, 0); - } - else { - Src2 = Dest; - } - uint8_t CompType = Op->Src[1].Data.Literal.Value; - - OrderedNode *Result{}; - // This maps 1:1 to an AArch64 NEON Op - //auto ALUOp = _VCMPGT(Size, ElementSize, Dest, Src); - switch (CompType) { - case 0x00: case 0x08: case 0x10: case 0x18: // EQ - Result = _VFCMPEQ(Size, ElementSize, Src2, Src); - break; - case 0x01: case 0x09: case 0x11: case 0x19: // LT, GT(Swapped operand) - Result = _VFCMPLT(Size, ElementSize, Src2, Src); - break; - case 0x02: case 0x0A: case 0x12: case 0x1A: // LE, GE(Swapped operand) - Result = _VFCMPLE(Size, ElementSize, Src2, Src); - break; - case 0x03: case 0x0B: case 0x13: case 0x1B: // Unordered - Result = _VFCMPUNO(Size, ElementSize, Src2, Src); - break; - case 0x04: case 0x0C: case 0x14: case 0x1C: // NEQ - Result = _VFCMPNEQ(Size, ElementSize, Src2, Src); - break; - case 0x05: case 0x0D: case 0x15: case 0x1D: // NLT, NGT(Swapped operand) - Result = _VFCMPLT(Size, ElementSize, Src2, Src); - Result = _VNot(Size, ElementSize, Result); - break; - case 0x06: case 0x0E: case 0x16: case 0x1E: // NLE, NGE(Swapped operand) - Result = _VFCMPLE(Size, ElementSize, Src2, Src); - Result = _VNot(Size, ElementSize, Result); - break; - case 0x07: case 0x0F: case 0x17: case 0x1F: // Ordered - Result = _VFCMPORD(Size, ElementSize, Src2, Src); - break; - default: LOGMAN_MSG_A("Unknown Comparison type: %d", CompType); - } - - if constexpr (Scalar) { - // Insert the lower bits - Result = _VInsScalarElement(GetDstSize(Op), ElementSize, 0, Dest, Result); - } - - StoreResult(FPRClass, Op, Result, -1); -} - -OrderedNode *OpDispatchBuilder::GetX87Top() { - // Yes, we are storing 3 bits in a single flag register. - // Deal with it - return _LoadContext(1, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC, GPRClass); -} - -void OpDispatchBuilder::SetX87Top(OrderedNode *Value) { - _StoreContext(GPRClass, 1, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC, Value); -} - -template -void OpDispatchBuilder::FLD(OpcodeArgs) { - - // Update TOP - auto orig_top = GetX87Top(); - auto mask = _Constant(7); - - size_t read_width = (width == 80) ? 16 : width / 8; - - OrderedNode *data{}; - - if (!Op->Src[0].IsNone()) { - // Read from memory - data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags, -1); - } - else { - // Implicit arg - auto offset = _Constant(Op->OP & 7); - data = _And(_Add(orig_top, offset), mask); - data = _LoadContextIndexed(data, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - } - OrderedNode *converted = data; - - // Convert to 80bit float - if constexpr (width == 32 || width == 64) { - converted = _F80CVTTo(data, width / 8); - } - - auto top = _And(_Sub(orig_top, _Constant(1)), mask); - SetX87Top(top); - // Write to ST[TOP] - _StoreContextIndexed(converted, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - //_StoreContext(converted, 16, offsetof(FEXCore::Core::CPUState, mm[7][0])); -} - -void OpDispatchBuilder::FBLD(OpcodeArgs) { - - // Update TOP - auto orig_top = GetX87Top(); - auto mask = _Constant(7); - auto top = _And(_Sub(orig_top, _Constant(1)), mask); - SetX87Top(top); - - // Read from memory - OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags, -1); - OrderedNode *converted = _F80BCDLoad(data); - _StoreContextIndexed(converted, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::FBSTP(OpcodeArgs) { - - auto orig_top = GetX87Top(); - auto data = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - OrderedNode *converted = _F80BCDStore(data); - - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, 10, 1); - - auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7)); - SetX87Top(top); -} - -template -void OpDispatchBuilder::FLD_Const(OpcodeArgs) { - // Update TOP - auto orig_top = GetX87Top(); - auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7)); - SetX87Top(top); - - auto low = _Constant(Lower); - auto high = _Constant(Upper); - OrderedNode *data = _VCastFromGPR(16, 8, low); - data = _VInsGPR(16, 8, data, high, 1); - // Write to ST[TOP] - _StoreContextIndexed(data, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::FILD(OpcodeArgs) { - - // Update TOP - auto orig_top = GetX87Top(); - auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7)); - SetX87Top(top); - - size_t read_width = GetSrcSize(Op); - - // Read from memory - auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags, -1); - - auto zero = _Constant(0); - - // Sign extend to 64bits - if (read_width != 8) - data = _Sext(read_width * 8, data); - - // Extract sign and make interger absolute - auto sign = _Select(COND_SLT, data, zero, _Constant(0x8000), zero); - auto absolute = _Select(COND_SLT, data, zero, _Sub(zero, data), data); - - // left justify the absolute interger - auto shift = _Sub(_Constant(63), _FindMSB(absolute)); - auto shifted = _Lshl(absolute, shift); - - auto adjusted_exponent = _Sub(_Constant(0x3fff + 63), shift); - auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent); - auto upper = _Or(sign, zeroed_exponent); - - - OrderedNode *converted = _VCastFromGPR(16, 8, shifted); - converted = _VInsElement(16, 8, 1, 0, converted, _VCastFromGPR(16, 8, upper)); - - // Write to ST[TOP] - _StoreContextIndexed(converted, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -template -void OpDispatchBuilder::FST(OpcodeArgs) { - auto orig_top = GetX87Top(); - auto data = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - if constexpr (width == 80) { - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, data, 10, 1); - } - else if constexpr (width == 32 || width == 64) { - auto result = _F80CVT(data, width / 8); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, result, width / 8, 1); - } - - if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { - auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7)); - SetX87Top(top); - } -} - -template -void OpDispatchBuilder::FIST(OpcodeArgs) { - - auto Size = GetSrcSize(Op); - - auto orig_top = GetX87Top(); - OrderedNode *data = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - data = _F80CVTInt(data, Truncate, Size); - - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, 1); - - if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { - auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7)); - SetX87Top(top); - } -} - -template -void OpDispatchBuilder::FADD(OpcodeArgs) { - - auto top = GetX87Top(); - OrderedNode *StackLocation = top; - - OrderedNode *arg{}; - OrderedNode *b{}; - - auto mask = _Constant(7); - - if (!Op->Src[0].IsNone()) { - // Memory arg - if constexpr (width == 16 || width == 32 || width == 64) { - if constexpr (Integer) { - arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); - b = _F80CVTToInt(arg, width / 8); - } - else { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - b = _F80CVTTo(arg, width / 8); - } - } - } else { - // Implicit arg - auto offset = _Constant(Op->OP & 7); - arg = _And(_Add(top, offset), mask); - if constexpr (ResInST0 == OpResult::RES_STI) { - StackLocation = arg; - } - b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - } - - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - auto result = _F80Add(a, b); - - if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { - top = _And(_Add(top, _Constant(1)), mask); - SetX87Top(top); - } - - // Write to ST[TOP] - _StoreContextIndexed(result, StackLocation, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -template -void OpDispatchBuilder::FMUL(OpcodeArgs) { - - auto top = GetX87Top(); - OrderedNode *StackLocation = top; - OrderedNode *arg{}; - OrderedNode *b{}; - - auto mask = _Constant(7); - - if (!Op->Src[0].IsNone()) { - // Memory arg - - if constexpr (width == 16 || width == 32 || width == 64) { - if constexpr (Integer) { - arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); - b = _F80CVTToInt(arg, width / 8); - } - else { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - b = _F80CVTTo(arg, width / 8); - } - } - } else { - // Implicit arg - auto offset = _Constant(Op->OP & 7); - arg = _And(_Add(top, offset), mask); - if constexpr (ResInST0 == OpResult::RES_STI) { - StackLocation = arg; - } - - b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - } - - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - auto result = _F80Mul(a, b); - - if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { - top = _And(_Add(top, _Constant(1)), mask); - SetX87Top(top); - } - - // Write to ST[TOP] - _StoreContextIndexed(result, StackLocation, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -template -void OpDispatchBuilder::FDIV(OpcodeArgs) { - - auto top = GetX87Top(); - OrderedNode *StackLocation = top; - OrderedNode *arg{}; - OrderedNode *b{}; - - auto mask = _Constant(7); - - if (!Op->Src[0].IsNone()) { - // Memory arg - - if constexpr (width == 16 || width == 32 || width == 64) { - if constexpr (Integer) { - arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); - b = _F80CVTToInt(arg, width / 8); - } - else { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - b = _F80CVTTo(arg, width / 8); - } - } - } else { - // Implicit arg - auto offset = _Constant(Op->OP & 7); - arg = _And(_Add(top, offset), mask); - if constexpr (ResInST0 == OpResult::RES_STI) { - StackLocation = arg; - } - - b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - } - - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - OrderedNode *result{}; - if constexpr (reverse) { - result = _F80Div(b, a); - } - else { - result = _F80Div(a, b); - } - - if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { - top = _And(_Add(top, _Constant(1)), mask); - SetX87Top(top); - } - - // Write to ST[TOP] - _StoreContextIndexed(result, StackLocation, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -template -void OpDispatchBuilder::FSUB(OpcodeArgs) { - auto top = GetX87Top(); - OrderedNode *StackLocation = top; - OrderedNode *arg{}; - OrderedNode *b{}; - - auto mask = _Constant(7); - - if (!Op->Src[0].IsNone()) { - // Memory arg - - if constexpr (width == 16 || width == 32 || width == 64) { - if constexpr (Integer) { - arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); - b = _F80CVTToInt(arg, width / 8); - } - else { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - b = _F80CVTTo(arg, width / 8); - } - } - } else { - // Implicit arg - auto offset = _Constant(Op->OP & 7); - arg = _And(_Add(top, offset), mask); - if constexpr (ResInST0 == OpResult::RES_STI) { - StackLocation = arg; - } - b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - } - - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - OrderedNode *result{}; - if constexpr (reverse) { - result = _F80Sub(b, a); - } - else { - result = _F80Sub(a, b); - } - - if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { - top = _And(_Add(top, _Constant(1)), mask); - SetX87Top(top); - } - - // Write to ST[TOP] - _StoreContextIndexed(result, StackLocation, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::FCHS(OpcodeArgs) { - - auto top = GetX87Top(); - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - auto low = _Constant(0); - auto high = _Constant(0b1'000'0000'0000'0000); - OrderedNode *data = _VCastFromGPR(16, 8, low); - data = _VInsGPR(16, 8, data, high, 1); - - auto result = _VXor(a, data, 16, 1); - - // Write to ST[TOP] - _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::FABS(OpcodeArgs) { - - auto top = GetX87Top(); - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - auto low = _Constant(~0ULL); - auto high = _Constant(0b0'111'1111'1111'1111); - OrderedNode *data = _VCastFromGPR(16, 8, low); - data = _VInsGPR(16, 8, data, high, 1); - - auto result = _VAnd(a, data, 16, 1); - - // Write to ST[TOP] - _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::FTST(OpcodeArgs) { - - auto top = GetX87Top(); - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - auto low = _Constant(0); - OrderedNode *data = _VCastFromGPR(16, 8, low); - - OrderedNode *Res = _F80Cmp(a, data, - (1 << FCMP_FLAG_EQ) | - (1 << FCMP_FLAG_LT) | - (1 << FCMP_FLAG_UNORDERED)); - - OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT); - OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ); - OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED); - HostFlag_CF = _Or(HostFlag_CF, HostFlag_Unordered); - HostFlag_ZF = _Or(HostFlag_ZF, HostFlag_Unordered); - - SetRFLAG(HostFlag_CF); - SetRFLAG(_Constant(0)); - SetRFLAG(HostFlag_Unordered); - SetRFLAG(HostFlag_ZF); -} - -void OpDispatchBuilder::FRNDINT(OpcodeArgs) { - - auto top = GetX87Top(); - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - auto result = _F80Round(a); - - // Write to ST[TOP] - _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::FXTRACT(OpcodeArgs) { - - auto orig_top = GetX87Top(); - auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7)); - SetX87Top(top); - - auto a = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - auto exp = _F80XTRACT_EXP(a); - auto sig = _F80XTRACT_SIG(a); - - // Write to ST[TOP] - _StoreContextIndexed(exp, orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - _StoreContextIndexed(sig, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::FNINIT(OpcodeArgs) { - - // Init FCW to 0x037 - auto NewFCW = _Constant(16, 0x037); - _F80LoadFCW(NewFCW); - _StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW); - - // Init FSW to 0 - SetX87Top(_Constant(0)); - - SetRFLAG(_Constant(0)); - SetRFLAG(_Constant(0)); - SetRFLAG(_Constant(0)); - SetRFLAG(_Constant(0)); - - // XXX: Add FTW support -} - -template -void OpDispatchBuilder::FCOMI(OpcodeArgs) { - - auto top = GetX87Top(); - auto mask = _Constant(7); - - OrderedNode *arg{}; - OrderedNode *b{}; - - if (!Op->Src[0].IsNone()) { - // Memory arg - if constexpr (width == 16 || width == 32 || width == 64) { - if constexpr (Integer) { - arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); - b = _F80CVTToInt(arg, width / 8); - } - else { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - b = _F80CVTTo(arg, width / 8); - } - } - } else { - // Implicit arg - auto offset = _Constant(Op->OP & 7); - arg = _And(_Add(top, offset), mask); - b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - } - - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - OrderedNode *Res = _F80Cmp(a, b, - (1 << FCMP_FLAG_EQ) | - (1 << FCMP_FLAG_LT) | - (1 << FCMP_FLAG_UNORDERED)); - - OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT); - OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ); - OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED); - HostFlag_CF = _Or(HostFlag_CF, HostFlag_Unordered); - HostFlag_ZF = _Or(HostFlag_ZF, HostFlag_Unordered); - - if constexpr (whichflags == FCOMIFlags::FLAGS_X87) { - SetRFLAG(HostFlag_CF); - SetRFLAG(_Constant(0)); - SetRFLAG(HostFlag_Unordered); - SetRFLAG(HostFlag_ZF); - } - else { - SetRFLAG(HostFlag_CF); - SetRFLAG(HostFlag_ZF); - SetRFLAG(HostFlag_Unordered); - } - - - if constexpr (poptwice) { - top = _And(_Add(top, _Constant(2)), mask); - SetX87Top(top); - } - else if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { - top = _And(_Add(top, _Constant(1)), mask); - SetX87Top(top); - } -} - -void OpDispatchBuilder::FXCH(OpcodeArgs) { - - auto top = GetX87Top(); - OrderedNode* arg; - - auto mask = _Constant(7); - - // Implicit arg - auto offset = _Constant(Op->OP & 7); - arg = _And(_Add(top, offset), mask); - - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - auto b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - // Write to ST[TOP] - _StoreContextIndexed(b, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - _StoreContextIndexed(a, arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::FST(OpcodeArgs) { - - auto top = GetX87Top(); - OrderedNode* arg; - - auto mask = _Constant(7); - - // Implicit arg - auto offset = _Constant(Op->OP & 7); - arg = _And(_Add(top, offset), mask); - - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - // Write to ST[TOP] - _StoreContextIndexed(a, arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { - top = _And(_Add(top, _Constant(1)), _Constant(7)); - SetX87Top(top); - } -} - -template -void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) { - - auto top = GetX87Top(); - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - auto result = _F80Round(a); - // Overwrite the op - result.first->Header.Op = IROp; - - // Write to ST[TOP] - _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -template -void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) { - auto top = GetX87Top(); - - auto mask = _Constant(7); - OrderedNode *st1 = _And(_Add(top, _Constant(1)), mask); - - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - st1 = _LoadContextIndexed(st1, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - auto result = _F80Add(a, st1); - // Overwrite the op - result.first->Header.Op = IROp; - - if constexpr (IROp == IR::OP_F80FPREM) { - //TODO: Set C0 to Q2, C3 to Q1, C1 to Q0 - SetRFLAG(_Constant(0)); - } - - // Write to ST[TOP] - _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -template -void OpDispatchBuilder::X87ModifySTP(OpcodeArgs) { - auto orig_top = GetX87Top(); - if (Inc) { - auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7)); - SetX87Top(top); - } - else { - auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7)); - SetX87Top(top); - } -} - -void OpDispatchBuilder::X87SinCos(OpcodeArgs) { - - auto orig_top = GetX87Top(); - auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7)); - SetX87Top(top); - - auto a = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - auto sin = _F80SIN(a); - auto cos = _F80COS(a); - - // Write to ST[TOP] - _StoreContextIndexed(sin, orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - _StoreContextIndexed(cos, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::X87FYL2X(OpcodeArgs) { - bool Plus1 = Op->OP == 0x01F9; // FYL2XP - - auto orig_top = GetX87Top(); - auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7)); - SetX87Top(top); - - OrderedNode *st0 = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - OrderedNode *st1 = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - if (Plus1) { - auto low = _Constant(0x8000'0000'0000'0000); - auto high = _Constant(0b0'011'1111'1111'1111); - OrderedNode *data = _VCastFromGPR(16, 8, low); - data = _VInsGPR(16, 8, data, high, 1); - st0 = _F80Add(st0, data); - } - - auto result = _F80FYL2X(st0, st1); - - // Write to ST[TOP] - _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::X87TAN(OpcodeArgs) { - - auto orig_top = GetX87Top(); - auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7)); - SetX87Top(top); - - auto a = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - auto result = _F80TAN(a); - - auto low = _Constant(0x8000'0000'0000'0000); - auto high = _Constant(0b0'011'1111'1111'1111); - OrderedNode *data = _VCastFromGPR(16, 8, low); - data = _VInsGPR(16, 8, data, high, 1); - - // Write to ST[TOP] - _StoreContextIndexed(result, orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - _StoreContextIndexed(data, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::X87ATAN(OpcodeArgs) { - - auto orig_top = GetX87Top(); - auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7)); - SetX87Top(top); - - auto a = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - OrderedNode *st1 = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - auto result = _F80ATAN(st1, a); - - // Write to ST[TOP] - _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::X87LDENV(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false); - Mem = AppendSegmentOffset(Mem, Op->Flags); - - auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2); - _F80LoadFCW(NewFCW); - _StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW); - - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 1)); - auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size); - - // Strip out the FSW information - auto Top = _Bfe(3, 11, NewFSW); - SetX87Top(Top); - - auto C0 = _Bfe(1, 8, NewFSW); - auto C1 = _Bfe(1, 9, NewFSW); - auto C2 = _Bfe(1, 10, NewFSW); - auto C3 = _Bfe(1, 14, NewFSW); - - SetRFLAG(C0); - SetRFLAG(C1); - SetRFLAG(C2); - SetRFLAG(C3); -} - -void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) { - // 14 bytes for 16bit - // 2 Bytes : FCW - // 2 Bytes : FSW - // 2 bytes : FTW - // 2 bytes : Instruction offset - // 2 bytes : Instruction CS selector - // 2 bytes : Data offset - // 2 bytes : Data selector - - // 28 bytes for 32bit - // 4 bytes : FCW - // 4 bytes : FSW - // 4 bytes : FTW - // 4 bytes : Instruction pointer - // 2 bytes : instruction pointer selector - // 2 bytes : Opcode - // 4 bytes : data pointer offset - // 4 bytes : data pointer selector - - auto Size = GetDstSize(Op); - OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false); - Mem = AppendSegmentOffset(Mem, Op->Flags); - - { - auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass); - _StoreMem(GPRClass, Size, Mem, FCW, Size); - } - - { - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 1)); - // We must construct the FSW from our various bits - OrderedNode *FSW = _Constant(0); - auto Top = GetX87Top(); - FSW = _Or(FSW, _Lshl(Top, _Constant(11))); - - auto C0 = GetRFLAG(FEXCore::X86State::X87FLAG_C0_LOC); - auto C1 = GetRFLAG(FEXCore::X86State::X87FLAG_C1_LOC); - auto C2 = GetRFLAG(FEXCore::X86State::X87FLAG_C2_LOC); - auto C3 = GetRFLAG(FEXCore::X86State::X87FLAG_C3_LOC); - - FSW = _Or(FSW, _Lshl(C0, _Constant(8))); - FSW = _Or(FSW, _Lshl(C1, _Constant(9))); - FSW = _Or(FSW, _Lshl(C2, _Constant(10))); - FSW = _Or(FSW, _Lshl(C3, _Constant(14))); - _StoreMem(GPRClass, Size, MemLocation, FSW, Size); - } - - auto ZeroConst = _Constant(0); - - { - // FTW - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 2)); - _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); - } - - { - // Instruction Offset - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 3)); - _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); - } - - { - // Instruction CS selector (+ Opcode) - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 4)); - _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); - } - - { - // Data pointer offset - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 5)); - _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); - } - - { - // Data pointer selector - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 6)); - _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); - } -} - -void OpDispatchBuilder::X87FLDCW(OpcodeArgs) { - OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); - _F80LoadFCW(NewFCW); - _StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW); -} - -void OpDispatchBuilder::X87FSTCW(OpcodeArgs) { - auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass); - - StoreResult(GPRClass, Op, FCW, -1); -} - -void OpDispatchBuilder::X87LDSW(OpcodeArgs) { - OrderedNode *NewFSW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); - // Strip out the FSW information - auto Top = _Bfe(3, 11, NewFSW); - SetX87Top(Top); - - auto C0 = _Bfe(1, 8, NewFSW); - auto C1 = _Bfe(1, 9, NewFSW); - auto C2 = _Bfe(1, 10, NewFSW); - auto C3 = _Bfe(1, 14, NewFSW); - - SetRFLAG(C0); - SetRFLAG(C1); - SetRFLAG(C2); - SetRFLAG(C3); -} - -void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) { - // We must construct the FSW from our various bits - OrderedNode *FSW = _Constant(0); - auto Top = GetX87Top(); - FSW = _Or(FSW, _Lshl(Top, _Constant(11))); - - auto C0 = GetRFLAG(FEXCore::X86State::X87FLAG_C0_LOC); - auto C1 = GetRFLAG(FEXCore::X86State::X87FLAG_C1_LOC); - auto C2 = GetRFLAG(FEXCore::X86State::X87FLAG_C2_LOC); - auto C3 = GetRFLAG(FEXCore::X86State::X87FLAG_C3_LOC); - - FSW = _Or(FSW, _Lshl(C0, _Constant(8))); - FSW = _Or(FSW, _Lshl(C1, _Constant(9))); - FSW = _Or(FSW, _Lshl(C2, _Constant(10))); - FSW = _Or(FSW, _Lshl(C3, _Constant(14))); - - StoreResult(GPRClass, Op, FSW, -1); -} - -void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) { - // 14 bytes for 16bit - // 2 Bytes : FCW - // 2 Bytes : FSW - // 2 bytes : FTW - // 2 bytes : Instruction offset - // 2 bytes : Instruction CS selector - // 2 bytes : Data offset - // 2 bytes : Data selector - - // 28 bytes for 32bit - // 4 bytes : FCW - // 4 bytes : FSW - // 4 bytes : FTW - // 4 bytes : Instruction pointer - // 2 bytes : instruction pointer selector - // 2 bytes : Opcode - // 4 bytes : data pointer offset - // 4 bytes : data pointer selector - - auto Size = GetDstSize(Op); - OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false); - Mem = AppendSegmentOffset(Mem, Op->Flags); - - OrderedNode *Top = GetX87Top(); - { - auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass); - _StoreMem(GPRClass, Size, Mem, FCW, Size); - } - - { - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 1)); - // We must construct the FSW from our various bits - OrderedNode *FSW = _Constant(0); - FSW = _Or(FSW, _Lshl(Top, _Constant(11))); - - auto C0 = GetRFLAG(FEXCore::X86State::X87FLAG_C0_LOC); - auto C1 = GetRFLAG(FEXCore::X86State::X87FLAG_C1_LOC); - auto C2 = GetRFLAG(FEXCore::X86State::X87FLAG_C2_LOC); - auto C3 = GetRFLAG(FEXCore::X86State::X87FLAG_C3_LOC); - - FSW = _Or(FSW, _Lshl(C0, _Constant(8))); - FSW = _Or(FSW, _Lshl(C1, _Constant(9))); - FSW = _Or(FSW, _Lshl(C2, _Constant(10))); - FSW = _Or(FSW, _Lshl(C3, _Constant(14))); - _StoreMem(GPRClass, Size, MemLocation, FSW, Size); - } - - auto ZeroConst = _Constant(0); - - { - // FTW - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 2)); - _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); - } - - { - // Instruction Offset - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 3)); - _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); - } - - { - // Instruction CS selector (+ Opcode) - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 4)); - _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); - } - - { - // Data pointer offset - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 5)); - _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); - } - - { - // Data pointer selector - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 6)); - _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); - } - - OrderedNode *ST0Location = _Add(Mem, _Constant(Size * 7)); - - auto OneConst = _Constant(1); - auto SevenConst = _Constant(7); - auto TenConst = _Constant(10); - for (int i = 0; i < 7; ++i) { - auto data = _LoadContextIndexed(Top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - _StoreMem(FPRClass, 16, ST0Location, data, 1); - ST0Location = _Add(ST0Location, TenConst); - Top = _And(_Add(Top, OneConst), SevenConst); - } - - // The final st(7) needs a bit of special handling here - auto data = _LoadContextIndexed(Top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - // ST7 broken in to two parts - // Lower 64bits [63:0] - // upper 16 bits [79:64] - _StoreMem(FPRClass, 8, ST0Location, data, 1); - ST0Location = _Add(ST0Location, _Constant(8)); - auto topBytes = _VExtractElement(16, 2, data, 4); - _StoreMem(FPRClass, 2, ST0Location, topBytes, 1); - - // reset to default - FNINIT(Op); -} - -void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false); - Mem = AppendSegmentOffset(Mem, Op->Flags); - - auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2); - _F80LoadFCW(NewFCW); - _StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW); - - OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 1)); - auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size); - - // Strip out the FSW information - OrderedNode *Top = _Bfe(3, 11, NewFSW); - SetX87Top(Top); - - auto C0 = _Bfe(1, 8, NewFSW); - auto C1 = _Bfe(1, 9, NewFSW); - auto C2 = _Bfe(1, 10, NewFSW); - auto C3 = _Bfe(1, 14, NewFSW); - - SetRFLAG(C0); - SetRFLAG(C1); - SetRFLAG(C2); - SetRFLAG(C3); - - OrderedNode *ST0Location = _Add(Mem, _Constant(Size * 7)); - - auto OneConst = _Constant(1); - auto SevenConst = _Constant(7); - auto TenConst = _Constant(10); - - auto low = _Constant(~0ULL); - auto high = _Constant(0xFFFF); - OrderedNode *Mask = _VCastFromGPR(16, 8, low); - Mask = _VInsGPR(16, 8, Mask, high, 1); - - for (int i = 0; i < 7; ++i) { - OrderedNode *Reg = _LoadMem(FPRClass, 16, ST0Location, 1); - // Mask off the top bits - Reg = _VAnd(16, 16, Reg, Mask); - - _StoreContextIndexed(Reg, Top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - - ST0Location = _Add(ST0Location, TenConst); - Top = _And(_Add(Top, OneConst), SevenConst); - } - - // The final st(7) needs a bit of special handling here - // ST7 broken in to two parts - // Lower 64bits [63:0] - // upper 16 bits [79:64] - - OrderedNode *Reg = _LoadMem(FPRClass, 8, ST0Location, 1); - ST0Location = _Add(ST0Location, _Constant(8)); - OrderedNode *RegHigh = _LoadMem(FPRClass, 2, ST0Location, 1); - Reg = _VInsElement(16, 2, 4, 0, Reg, RegHigh); - _StoreContextIndexed(Reg, Top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::X87FXAM(OpcodeArgs) { - auto top = GetX87Top(); - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - OrderedNode *Result = _VExtractToGPR(16, 8, a, 1); - - // Extract the sign bit - Result = _Lshr(Result, _Constant(15)); - SetRFLAG(Result); - - // Claim this is a normal number - // We don't support anything else - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - SetRFLAG(ZeroConst); - SetRFLAG(OneConst); - SetRFLAG(ZeroConst); -} - -void OpDispatchBuilder::X87FCMOV(OpcodeArgs) { - enum CompareType { - COMPARE_ZERO, - COMPARE_NOTZERO, - }; - uint32_t FLAGMask{}; - CompareType Type = COMPARE_ZERO; - OrderedNode *SrcCond; - - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - - uint16_t Opcode = Op->OP & 0b1111'1111'1000; - switch (Opcode) { - case 0x3'C0: - FLAGMask = 1 << FEXCore::X86State::RFLAG_CF_LOC; - Type = COMPARE_ZERO; - break; - case 0x2'C0: - FLAGMask = 1 << FEXCore::X86State::RFLAG_CF_LOC; - Type = COMPARE_NOTZERO; - break; - case 0x2'C8: - FLAGMask = 1 << FEXCore::X86State::RFLAG_ZF_LOC; - Type = COMPARE_NOTZERO; - break; - case 0x3'C8: - FLAGMask = 1 << FEXCore::X86State::RFLAG_ZF_LOC; - Type = COMPARE_ZERO; - break; - case 0x2'D0: - FLAGMask = (1 << FEXCore::X86State::RFLAG_ZF_LOC) | (1 << FEXCore::X86State::RFLAG_CF_LOC); - Type = COMPARE_NOTZERO; - break; - case 0x3'D0: - FLAGMask = (1 << FEXCore::X86State::RFLAG_ZF_LOC) | (1 << FEXCore::X86State::RFLAG_CF_LOC); - Type = COMPARE_ZERO; - break; - case 0x2'D8: - FLAGMask = 1 << FEXCore::X86State::RFLAG_PF_LOC; - Type = COMPARE_NOTZERO; - break; - case 0x3'D8: - FLAGMask = 1 << FEXCore::X86State::RFLAG_PF_LOC; - Type = COMPARE_ZERO; - break; - default: - LOGMAN_MSG_A("Unhandled FCMOV op: 0x%x", Opcode); - break; - } - - auto MaskConst = _Constant(FLAGMask); - - auto RFLAG = GetPackedRFLAG(false); - - auto AndOp = _And(RFLAG, MaskConst); - switch (Type) { - case COMPARE_ZERO: { - SrcCond = _Select(FEXCore::IR::COND_EQ, - AndOp, ZeroConst, OneConst, ZeroConst); - break; - } - case COMPARE_NOTZERO: { - SrcCond = _Select(FEXCore::IR::COND_EQ, - AndOp, ZeroConst, ZeroConst, OneConst); - break; - } - } - - SrcCond = _Sbfe(1, 0, SrcCond); - - OrderedNode *VecCond = _VCastFromGPR(16, 8, SrcCond); - VecCond = _VInsGPR(16, 8, VecCond, SrcCond, 1); - - auto top = GetX87Top(); - OrderedNode* arg; - - auto mask = _Constant(7); - - // Implicit arg - auto offset = _Constant(Op->OP & 7); - arg = _And(_Add(top, offset), mask); - - auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - auto b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); - auto Result = _VBSL(VecCond, b, a); - - // Write to ST[TOP] - _StoreContextIndexed(Result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); -} - -void OpDispatchBuilder::FXSaveOp(OpcodeArgs) { - OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false); - Mem = AppendSegmentOffset(Mem, Op->Flags); - - // Saves 512bytes to the memory location provided - // Header changes depending on if REX.W is set or not - if (Op->Flags & X86Tables::DecodeFlags::FLAG_REX_WIDENING) { - // BYTE | 0 1 | 2 3 | 4 | 5 | 6 7 | 8 9 | a b | c d | e f | - // ------------------------------------------ - // 00 | FCW | FSW | FTW | | FOP | FIP | - // 16 | FDP | MXCSR | MXCSR_MASK| - } - else { - // BYTE | 0 1 | 2 3 | 4 | 5 | 6 7 | 8 9 | a b | c d | e f | - // ------------------------------------------ - // 00 | FCW | FSW | FTW | | FOP | FIP[31:0] | FCS | | - // 16 | FDP[31:0] | FDS | | MXCSR | MXCSR_MASK| - } - - { - auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass); - _StoreMem(GPRClass, 2, Mem, FCW, 2); - } - - { - // We must construct the FSW from our various bits - OrderedNode *MemLocation = _Add(Mem, _Constant(2)); - OrderedNode *FSW = _Constant(0); - auto Top = GetX87Top(); - FSW = _Or(FSW, _Lshl(Top, _Constant(11))); - - auto C0 = GetRFLAG(FEXCore::X86State::X87FLAG_C0_LOC); - auto C1 = GetRFLAG(FEXCore::X86State::X87FLAG_C1_LOC); - auto C2 = GetRFLAG(FEXCore::X86State::X87FLAG_C2_LOC); - auto C3 = GetRFLAG(FEXCore::X86State::X87FLAG_C3_LOC); - - FSW = _Or(FSW, _Lshl(C0, _Constant(8))); - FSW = _Or(FSW, _Lshl(C1, _Constant(9))); - FSW = _Or(FSW, _Lshl(C2, _Constant(10))); - FSW = _Or(FSW, _Lshl(C3, _Constant(14))); - _StoreMem(GPRClass, 2, MemLocation, FSW, 2); - } - - // BYTE | 0 1 | 2 3 | 4 | 5 | 6 7 | 8 9 | a b | c d | e f | - // ------------------------------------------ - // 32 | ST0/MM0 | - // 48 | ST1/MM1 | - // 64 | ST2/MM2 | - // 80 | ST3/MM3 | - // 96 | ST4/MM4 | - // 112 | ST5/MM5 | - // 128 | ST6/MM6 | - // 144 | ST7/MM7 | - // 160 | XMM0 - // 173 | XMM1 - // 192 | XMM2 - // 208 | XMM3 - // 224 | XMM4 - // 240 | XMM5 - // 256 | XMM6 - // 272 | XMM7 - // 288 | XMM8 - // 304 | XMM9 - // 320 | XMM10 - // 336 | XMM11 - // 352 | XMM12 - // 368 | XMM13 - // 384 | XMM14 - // 400 | XMM15 - // 416 | - // 432 | - // 448 | - // 464 | Available - // 480 | Available - // 496 | Available - // FCW: x87 FPU control word - // FSW: x87 FPU status word - // FTW: x87 FPU Tag word (Abridged) - // FOP: x87 FPU opcode. Lower 11 bits of the opcode - // FIP: x87 FPU instructyion pointer offset - // FCS: x87 FPU instruction pointer selector. If CPUID_0000_0007_0000_00000:EBX[bit 13] = 1 then this is deprecated and stores as 0 - // FDP: x87 FPU instruction operand (data) pointer offset - // FDS: x87 FPU instruction operand (data) pointer selector. Same deprecation as FCS - // MXCSR: If OSFXSR bit in CR4 is not set then this may not be saved - // MXCSR_MASK: Mask for writes to the MXCSR register - // If OSFXSR bit in CR4 is not set than FXSAVE /may/ not save the XMM registers - // This is implementation dependent - for (unsigned i = 0; i < 8; ++i) { - OrderedNode *MMReg = _LoadContext(16, offsetof(FEXCore::Core::CPUState, mm[i]), FPRClass); - OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32)); - - _StoreMem(FPRClass, 16, MemLocation, MMReg, 16); - } - for (unsigned i = 0; i < 16; ++i) { - OrderedNode *XMMReg = _LoadContext(16, offsetof(FEXCore::Core::CPUState, xmm[i]), FPRClass); - OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160)); - - _StoreMem(FPRClass, 16, MemLocation, XMMReg, 16); - } -} - -void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) { - OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false); - Mem = AppendSegmentOffset(Mem, Op->Flags); - - auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2); - _F80LoadFCW(NewFCW); - _StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW); - - { - OrderedNode *MemLocation = _Add(Mem, _Constant(2)); - auto NewFSW = _LoadMem(GPRClass, 2, MemLocation, 2); - - // Strip out the FSW information - auto Top = _Bfe(3, 11, NewFSW); - SetX87Top(Top); - - auto C0 = _Bfe(1, 8, NewFSW); - auto C1 = _Bfe(1, 9, NewFSW); - auto C2 = _Bfe(1, 10, NewFSW); - auto C3 = _Bfe(1, 14, NewFSW); - - SetRFLAG(C0); - SetRFLAG(C1); - SetRFLAG(C2); - SetRFLAG(C3); - } - - for (unsigned i = 0; i < 8; ++i) { - OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32)); - auto MMReg = _LoadMem(FPRClass, 16, MemLocation, 16); - _StoreContext(FPRClass, 16, offsetof(FEXCore::Core::CPUState, mm[i]), MMReg); - } - for (unsigned i = 0; i < 16; ++i) { - OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160)); - auto XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16); - _StoreContext(FPRClass, 16, offsetof(FEXCore::Core::CPUState, xmm[i]), XMMReg); - } -} - -void OpDispatchBuilder::PAlignrOp(OpcodeArgs) { - OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - auto Size = GetDstSize(Op); - - uint8_t Index = Op->Src[1].Data.Literal.Value; - OrderedNode *Res{}; - if (Index >= (Size * 2)) { - // If the immediate is greater than both vectors combined then it zeroes the vector - Res = _VectorZero(Size); - } - else { - Res = _VExtr(Size, 1, Src1, Src2, Index); - } - StoreResult(FPRClass, Op, Res, -1); -} - -template -void OpDispatchBuilder::UCOMISxOp(OpcodeArgs) { - OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - OrderedNode *Res = _FCmp(Src1, Src2, ElementSize, - (1 << FCMP_FLAG_EQ) | - (1 << FCMP_FLAG_LT) | - (1 << FCMP_FLAG_UNORDERED)); - - OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT); - OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ); - OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED); - - SetRFLAG(HostFlag_CF); - SetRFLAG(HostFlag_ZF); - SetRFLAG(HostFlag_Unordered); - - auto ZeroConst = _Constant(0); - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); - - flagsOp = FLAGS_OP_FCMP; - flagsOpDest = Src1; - flagsOpSrc = Src2; - flagsOpSize = GetSrcSize(Op); -} - -void OpDispatchBuilder::LDMXCSR(OpcodeArgs) { - OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1); - // We only support the rounding mode being set - OrderedNode *RoundingMode = _Bfe(4, 3, 13, Dest); - _SetRoundingMode(RoundingMode); -} - -void OpDispatchBuilder::STMXCSR(OpcodeArgs) { - // Default MXCSR - OrderedNode *MXCSR = _Constant(32, 0x1F80); - OrderedNode *RoundingMode = _GetRoundingMode(); - MXCSR = _Bfi(4, 3, 13, MXCSR, RoundingMode); - - StoreResult(GPRClass, Op, MXCSR, -1); -} - -template -void OpDispatchBuilder::PACKUSOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *Res = _VSQXTUN(Size, ElementSize, Dest); - Res = _VSQXTUN2(Size, ElementSize, Res, Src); - - StoreResult(FPRClass, Op, Res, -1); -} - -template -void OpDispatchBuilder::PACKSSOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *Res = _VSQXTN(Size, ElementSize, Dest); - Res = _VSQXTN2(Size, ElementSize, Res, Src); - - StoreResult(FPRClass, Op, Res, -1); -} - -template -void OpDispatchBuilder::PMULLOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *Res{}; - - if (Size == 8) { - if constexpr (Signed) { - Res = _VSMull(16, ElementSize, Src1, Src2); - } - else { - Res = _VUMull(16, ElementSize, Src1, Src2); - } - } - else { - OrderedNode* Srcs1[2]{}; - OrderedNode* Srcs2[2]{}; - - Srcs1[0] = _VExtr(Size, ElementSize, Src1, Src1, 0); - Srcs1[1] = _VExtr(Size, ElementSize, Src1, Src1, 2); - - Srcs2[0] = _VExtr(Size, ElementSize, Src2, Src2, 0); - Srcs2[1] = _VExtr(Size, ElementSize, Src2, Src2, 2); - - Src1 = _VInsElement(Size, ElementSize, 1, 0, Srcs1[0], Srcs1[1]); - Src2 = _VInsElement(Size, ElementSize, 1, 0, Srcs2[0], Srcs2[1]); - - if constexpr (Signed) { - Res = _VSMull(Size, ElementSize, Src1, Src2); - } - else { - Res = _VUMull(Size, ElementSize, Src1, Src2); - } - } - StoreResult(FPRClass, Op, Res, -1); -} - -template -void OpDispatchBuilder::MOVQ2DQ(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - // This instruction is a bit special in that if the source is MMX then it zexts to 128bit - if constexpr (ToXMM) { - Src = _VMov(Src, 16); - _StoreContext(FPRClass, 16, offsetof(FEXCore::Core::CPUState, xmm[Op->Dest.Data.GPR.GPR - FEXCore::X86State::REG_XMM_0][0]), Src); - } - else { - // This is simple, just store the result - StoreResult(FPRClass, Op, Src, -1); - } -} - -template -void OpDispatchBuilder::PADDSOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *Res{}; - if constexpr (Signed) { - Res = _VSQAdd(Size, ElementSize, Dest, Src); - } - else { - Res = _VUQAdd(Size, ElementSize, Dest, Src); - } - - StoreResult(FPRClass, Op, Res, -1); -} - -template -void OpDispatchBuilder::PSUBSOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *Res{}; - if constexpr (Signed) { - Res = _VSQSub(Size, ElementSize, Dest, Src); - } - else { - Res = _VUQSub(Size, ElementSize, Dest, Src); - } - - StoreResult(FPRClass, Op, Res, -1); -} - -template -void OpDispatchBuilder::ADDSUBPOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *ResAdd{}; - OrderedNode *ResSub{}; - ResAdd = _VFAdd(Size, ElementSize, Dest, Src); - ResSub = _VFSub(Size, ElementSize, Dest, Src); - - // We now need to swizzle results - uint8_t NumElements = Size / ElementSize; - // Even elements are the sub result - // Odd elements are the add results - for (size_t i = 0; i < NumElements; i += 2) { - ResAdd = _VInsElement(Size, ElementSize, i, i, ResAdd, ResSub); - } - StoreResult(FPRClass, Op, ResAdd, -1); -} - -void OpDispatchBuilder::PMADDWD(OpcodeArgs) { - // This is a pretty curious operation - // Does two MADD operations across 4 16bit signed integers and accumulates to 32bit integers in the destination - // - // x86 PMADDWD: xmm1, xmm2 - // xmm1[31:0] = (xmm1[15:0] * xmm2[15:0]) + (xmm1[31:16] * xmm2[31:16]) - // xmm1[63:32] = (xmm1[47:32] * xmm2[47:32]) + (xmm1[63:48] * xmm2[63:48]) - // etc.. for larger registers - - auto Size = GetSrcSize(Op); - - OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - if (Size == 8) { - Size <<= 1; - Src1 = _VBitcast(Size, 2, Src1); - Src2 = _VBitcast(Size, 2, Src2); - } - - auto Src1_L = _VSXTL(Size, 2, Src1); // [15:0 ], [31:16], [32:47 ], [63:48 ] - auto Src1_H = _VSXTL2(Size, 2, Src1); // [79:64], [95:80], [111:96], [127:112] - - auto Src2_L = _VSXTL(Size, 2, Src2); // [15:0 ], [31:16], [32:47 ], [63:48 ] - auto Src2_H = _VSXTL2(Size, 2, Src2); // [79:64], [95:80], [111:96], [127:112] - - auto Res_L = _VSMul(Size, 4, Src1_L, Src2_L); // [15:0 ], [31:16], [32:47 ], [63:48 ] : Original elements - auto Res_H = _VSMul(Size, 4, Src1_H, Src2_H); // [79:64], [95:80], [111:96], [127:112] : Original elements - - // [15:0 ] + [31:16], [32:47 ] + [63:48 ], [79:64] + [95:80], [111:96] + [127:112] - auto Res = _VAddP(Size, 4, Res_L, Res_H); - StoreResult(FPRClass, Op, Res, -1); -} - -void OpDispatchBuilder::PMADDUBSW(OpcodeArgs) { - // This is a pretty curious operation - // Does four MADD operations across 8 8bit signed and unsigned integers and accumulates to 16bit integers in the destination WITH saturation - // - // x86 PMADDUBSW: mm1, mm2 - // mm1[15:0] = SaturateSigned16(((s8)mm2[15:8] * (u8)mm1[15:8]) + ((s8)mm2[7:0] * (u8)mm1[7:0])) - // mm1[31:16] = SaturateSigned16(((s8)mm2[31:24] * (u8)mm1[31:24]) + ((s8)mm2[23:16] * (u8)mm1[23:16])) - // mm1[47:32] = SaturateSigned16(((s8)mm2[47:40] * (u8)mm1[47:40]) + ((s8)mm2[39:32] * (u8)mm1[39:32])) - // mm1[63:48] = SaturateSigned16(((s8)mm2[63:56] * (u8)mm1[63:56]) + ((s8)mm2[55:48] * (u8)mm1[55:48])) - // Extends to larger registers - auto Size = GetSrcSize(Op); - - OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - if (Size == 8) { - // 64bit is more efficient - - // Src1 is unsigned - auto Src1_16b = _VUXTL(Size * 2, 1, Src1); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] - - // Src2 is signed - auto Src2_16b = _VSXTL(Size * 2, 1, Src2); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] - - auto ResMul_L = _VSMull(Size * 2, 2, Src1_16b, Src2_16b); - auto ResMul_H = _VSMull2(Size * 2, 2, Src1_16b, Src2_16b); - - // Now add pairwise across the vector - auto ResAdd = _VAddP(Size * 2, 4, ResMul_L, ResMul_H); - - // Add saturate back down to 16bit - OrderedNode *Res = _VSQXTN(Size * 2, 4, ResAdd); - StoreResult(FPRClass, Op, Res, -1); - } - else { - // Src1 is unsigned - auto Src1_16b_L = _VUXTL(Size, 1, Src1); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] - auto Src1_16b_H = _VUXTL2(Size, 1, Src1); // Offset to +64bits [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] - - // Src2 is signed - auto Src2_16b_L = _VSXTL(Size, 1, Src2); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] - auto Src2_16b_H = _VSXTL2(Size, 1, Src2); // Offset to +64bits [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] - - auto ResMul_L = _VSMull(Size, 2, Src1_16b_L, Src2_16b_L); - auto ResMul_L_H = _VSMull2(Size, 2, Src1_16b_L, Src2_16b_L); - - auto ResMul_H = _VSMull(Size, 2, Src1_16b_H, Src2_16b_H); - auto ResMul_H_H = _VSMull2(Size, 2, Src1_16b_H, Src2_16b_H); - - // Now add pairwise across the vector - auto ResAdd_L = _VAddP(Size, 4, ResMul_L, ResMul_L_H); - auto ResAdd_H = _VAddP(Size, 4, ResMul_H, ResMul_H_H); - - // Add saturate back down to 16bit - OrderedNode *Res = _VSQXTN(Size, 4, ResAdd_L); - Res = _VSQXTN2(Size, 4, Res, ResAdd_H); - - StoreResult(FPRClass, Op, Res, -1); - } -} - -template -void OpDispatchBuilder::PMULHW(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *Res{}; - if (Size == 8) { - Dest = _VBitcast(Size * 2, 2, Dest); - Src = _VBitcast(Size * 2, 2, Src); - - // Implementation is more efficient for 8byte registers - if (Signed) - Res = _VSMull(Size * 2, 2, Dest, Src); - else - Res = _VUMull(Size * 2, 2, Dest, Src); - - Res = _VUShrNI(Size * 2, 4, Res, 16); - } - else { - // 128bit is less efficient - OrderedNode *ResultLow; - OrderedNode *ResultHigh; - if (Signed) { - ResultLow = _VSMull(Size, 2, Dest, Src); - ResultHigh = _VSMull2(Size, 2, Dest, Src); - } - else { - ResultLow = _VUMull(Size, 2, Dest, Src); - ResultHigh = _VUMull2(Size, 2, Dest, Src); - } - - // Combine the results - Res = _VUShrNI(Size, 4, ResultLow, 16); - Res = _VUShrNI2(Size, 4, Res, ResultHigh, 16); - } - - StoreResult(FPRClass, Op, Res, -1); -} - -void OpDispatchBuilder::PMULHRSW(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *Res{}; - if (Size == 8) { - // Implementation is more efficient for 8byte registers - Res = _VSMull(Size * 2, 2, Dest, Src); - Res = _VSShrI(Size * 2, 4, Res, 14); - auto OneVector = _VectorImm(1, Size * 2, 4); - Res = _VAdd(Size * 2, 4, Res, OneVector); - Res = _VUShrNI(Size * 2, 4, Res, 1); - } - else { - // 128bit is less efficient - OrderedNode *ResultLow; - OrderedNode *ResultHigh; - - ResultLow = _VSMull(Size, 2, Dest, Src); - ResultHigh = _VSMull2(Size, 2, Dest, Src); - - ResultLow = _VSShrI(Size, 4, ResultLow, 14); - ResultHigh = _VSShrI(Size, 4, ResultHigh, 14); - auto OneVector = _VectorImm(1, Size, 4); - - ResultLow = _VAdd(Size, 4, ResultLow, OneVector); - ResultHigh = _VAdd(Size, 4, ResultHigh, OneVector); - - // Combine the results - Res = _VUShrNI(Size, 4, ResultLow, 1); - Res = _VUShrNI2(Size, 4, Res, ResultHigh, 1); - } - - StoreResult(FPRClass, Op, Res, -1); -} - void OpDispatchBuilder::MOVBEOp(OpcodeArgs) { OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, 1); Src = _Rev(Src); StoreResult(GPRClass, Op, Src, 1); } -template -void OpDispatchBuilder::HADDP(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *Res = _VFAddP(Size, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, Res, -1); -} - -template -void OpDispatchBuilder::HSUBP(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - // This is a bit complicated since AArch64 doesn't support a pairwise subtract - auto Dest_Neg = _VFNeg(Size, ElementSize, Dest); - auto Src_Neg = _VFNeg(Size, ElementSize, Src); - - // Now we need to swizzle the values - OrderedNode *Swizzle_Dest = Dest; - OrderedNode *Swizzle_Src = Src; - - if constexpr (ElementSize == 4) { - Swizzle_Dest = _VInsElement(Size, ElementSize, 1, 1, Swizzle_Dest, Dest_Neg); - Swizzle_Dest = _VInsElement(Size, ElementSize, 3, 3, Swizzle_Dest, Dest_Neg); - - Swizzle_Src = _VInsElement(Size, ElementSize, 1, 1, Swizzle_Src, Src_Neg); - Swizzle_Src = _VInsElement(Size, ElementSize, 3, 3, Swizzle_Src, Src_Neg); - } - else { - Swizzle_Dest = _VInsElement(Size, ElementSize, 1, 1, Swizzle_Dest, Dest_Neg); - Swizzle_Src = _VInsElement(Size, ElementSize, 1, 1, Swizzle_Src, Src_Neg); - } - - OrderedNode *Res = _VFAddP(Size, ElementSize, Swizzle_Dest, Swizzle_Src); - StoreResult(FPRClass, Op, Res, -1); -} - -template -void OpDispatchBuilder::PHADD(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *Res = _VAddP(Size, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, Res, -1); -} - -template -void OpDispatchBuilder::PHSUB(OpcodeArgs) { - auto Size = GetSrcSize(Op); - uint8_t NumElements = Size / ElementSize; - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - // This is a bit complicated since AArch64 doesn't support a pairwise subtract - auto Dest_Neg = _VNeg(Size, ElementSize, Dest); - auto Src_Neg = _VNeg(Size, ElementSize, Src); - - // Now we need to swizzle the values - OrderedNode *Swizzle_Dest = Dest; - OrderedNode *Swizzle_Src = Src; - - // Odd elements turn in to negated elements - for (size_t i = 1; i < NumElements; i += 2) { - Swizzle_Dest = _VInsElement(Size, ElementSize, i, i, Swizzle_Dest, Dest_Neg); - Swizzle_Src = _VInsElement(Size, ElementSize, i, i, Swizzle_Src, Src_Neg); - } - - OrderedNode *Res = _VAddP(Size, ElementSize, Swizzle_Dest, Swizzle_Src); - StoreResult(FPRClass, Op, Res, -1); -} - -void OpDispatchBuilder::PHADDS(OpcodeArgs) { - auto Size = GetSrcSize(Op); - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - if (Size == 8) { - // Implementation is more efficient for 8byte registers - auto Dest_Larger = _VSXTL(Size * 2, 2, Dest); - auto Src_Larger = _VSXTL(Size * 2, 2, Src); - - OrderedNode *AddRes = _VAddP(Size * 2, 4, Dest_Larger, Src_Larger); - - // Saturate back down to the result - OrderedNode *Res = _VSQXTN(Size * 2, 4, AddRes); - StoreResult(FPRClass, Op, Res, -1); - } - else { - auto Dest_Larger = _VSXTL(Size, 2, Dest); - auto Dest_Larger_H = _VSXTL2(Size, 2, Dest); - - auto Src_Larger = _VSXTL(Size, 2, Src); - auto Src_Larger_H = _VSXTL2(Size, 2, Src); - - OrderedNode *AddRes_L = _VAddP(Size, 4, Dest_Larger, Dest_Larger_H); - OrderedNode *AddRes_H = _VAddP(Size, 4, Src_Larger, Src_Larger_H); - - // Saturate back down to the result - OrderedNode *Res = _VSQXTN(Size, 4, AddRes_L); - Res = _VSQXTN2(Size, 4, Res, AddRes_H); - - StoreResult(FPRClass, Op, Res, -1); - } -} - -void OpDispatchBuilder::PHSUBS(OpcodeArgs) { - auto Size = GetSrcSize(Op); - uint8_t ElementSize = 2; - uint8_t NumElements = Size / ElementSize; - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - // This is a bit complicated since AArch64 doesn't support a pairwise subtract - auto Dest_Neg = _VNeg(Size, ElementSize, Dest); - auto Src_Neg = _VNeg(Size, ElementSize, Src); - - // Now we need to swizzle the values - OrderedNode *Swizzle_Dest = Dest; - OrderedNode *Swizzle_Src = Src; - - // Odd elements turn in to negated elements - for (size_t i = 1; i < NumElements; i += 2) { - Swizzle_Dest = _VInsElement(Size, ElementSize, i, i, Swizzle_Dest, Dest_Neg); - Swizzle_Src = _VInsElement(Size, ElementSize, i, i, Swizzle_Src, Src_Neg); - } - - Dest = Swizzle_Dest; - Src = Swizzle_Src; - - if (Size == 8) { - // Implementation is more efficient for 8byte registers - auto Dest_Larger = _VSXTL(Size * 2, 2, Dest); - auto Src_Larger = _VSXTL(Size * 2, 2, Src); - - OrderedNode *AddRes = _VAddP(Size * 2, 4, Dest_Larger, Src_Larger); - - // Saturate back down to the result - OrderedNode *Res = _VSQXTN(Size * 2, 4, AddRes); - StoreResult(FPRClass, Op, Res, -1); - } - else { - auto Dest_Larger = _VSXTL(Size, 2, Dest); - auto Dest_Larger_H = _VSXTL2(Size, 2, Dest); - - auto Src_Larger = _VSXTL(Size, 2, Src); - auto Src_Larger_H = _VSXTL2(Size, 2, Src); - - OrderedNode *AddRes_L = _VAddP(Size, 4, Dest_Larger, Dest_Larger_H); - OrderedNode *AddRes_H = _VAddP(Size, 4, Src_Larger, Src_Larger_H); - - // Saturate back down to the result - OrderedNode *Res = _VSQXTN(Size, 4, AddRes_L); - Res = _VSQXTN2(Size, 4, Res, AddRes_H); - - StoreResult(FPRClass, Op, Res, -1); - } -} - template void OpDispatchBuilder::FenceOp(OpcodeArgs) { _Fence({FenceType}); @@ -8298,389 +4778,6 @@ void OpDispatchBuilder::StoreFenceOrCLFlush(OpcodeArgs) { } } -void OpDispatchBuilder::PSADBW(OpcodeArgs) { - // The documentation is actually incorrect in how this instruction operates - // It strongly implies that the `abs(dest[i] - src[i])` operates in 8bit space - // but it actually operates in more than 8bit space - // This can be seen with `abs(0 - 0xFF)` returning a different result depending - // on bit length - auto Size = GetSrcSize(Op); - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *Result{}; - - if (Size == 8) { - Dest = _VUXTL(Size*2, 1, Dest); - Src = _VUXTL(Size*2, 1, Src); - - OrderedNode *SubResult = _VSub(Size*2, 2, Dest, Src); - OrderedNode *AbsResult = _VAbs(Size*2, 2, SubResult); - - // Now vector-wide add the results for each - Result = _VAddV(Size * 2, 2, AbsResult); - } - else { - OrderedNode *Dest_Low = _VUXTL(Size, 1, Dest); - OrderedNode *Dest_High = _VUXTL2(Size, 1, Dest); - - OrderedNode *Src_Low = _VUXTL(Size, 1, Src); - OrderedNode *Src_High = _VUXTL2(Size, 1, Src); - - OrderedNode *SubResult_Low = _VSub(Size, 2, Dest_Low, Src_Low); - OrderedNode *SubResult_High = _VSub(Size, 2, Dest_High, Src_High); - - OrderedNode *AbsResult_Low = _VAbs(Size, 2, SubResult_Low); - OrderedNode *AbsResult_High = _VAbs(Size, 2, SubResult_High); - - // Now vector pairwise add all four of these - OrderedNode * Result_Low = _VAddV(Size, 2, AbsResult_Low); - OrderedNode * Result_High = _VAddV(Size, 2, AbsResult_High); - - Result = _VInsElement(Size, 8, 1, 0, Result_Low, Result_High); - } - - StoreResult(FPRClass, Op, Result, -1); -} - -void OpDispatchBuilder::AESImcOp(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - auto Res = _VAESImc(Src); - StoreResult(FPRClass, Op, Res, -1); -} - -void OpDispatchBuilder::AESEncOp(OpcodeArgs) { - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - auto Res = _VAESEnc(Dest, Src); - StoreResult(FPRClass, Op, Res, -1); -} - -void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) { - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - auto Res = _VAESEncLast(Dest, Src); - StoreResult(FPRClass, Op, Res, -1); -} - -void OpDispatchBuilder::AESDecOp(OpcodeArgs) { - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - auto Res = _VAESDec(Dest, Src); - StoreResult(FPRClass, Op, Res, -1); -} - -void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) { - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - auto Res = _VAESDecLast(Dest, Src); - StoreResult(FPRClass, Op, Res, -1); -} - -void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) { - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint64_t RCON = Op->Src[1].Data.Literal.Value; - - auto Res = _VAESKeyGenAssist(Src, RCON); - StoreResult(FPRClass, Op, Res, -1); -} - -template -void OpDispatchBuilder::ExtendVectorElements(OpcodeArgs) { - auto Size = GetDstSize(Op); - - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *Result {Src}; - - for (size_t CurrentElementSize = ElementSize; - CurrentElementSize != DstElementSize; - CurrentElementSize <<= 1) { - if constexpr (Signed) { - Result = _VSXTL(Result, Size, CurrentElementSize); - } - else { - Result = _VUXTL(Result, Size, CurrentElementSize); - } - } - StoreResult(FPRClass, Op, Result, -1); -} - -template -void OpDispatchBuilder::VectorRound(OpcodeArgs) { - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint64_t Mode = Op->Src[1].Data.Literal.Value; - uint64_t RoundControlSource = (Mode >> 2) & 1; - uint64_t RoundControl = Mode & 0b11; - - auto Size = GetSrcSize(Op); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - if (RoundControlSource) { - RoundControl = 0; // MXCSR - } - - std::array SourceModes = { - FEXCore::IR::Round_Nearest, - FEXCore::IR::Round_Negative_Infinity, - FEXCore::IR::Round_Positive_Infinity, - FEXCore::IR::Round_Towards_Zero, - FEXCore::IR::Round_Host, - }; - - Src = _Vector_FToI(Src, SourceModes[(RoundControlSource << 2) | RoundControl], Size, ElementSize); - - if constexpr (Scalar) { - // Insert the lower bits - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - auto Result = _VInsScalarElement(GetDstSize(Op), ElementSize, 0, Dest, Src); - StoreResult(FPRClass, Op, Result, -1); - } - else { - StoreResult(FPRClass, Op, Src, -1); - } -} - -template -void OpDispatchBuilder::VectorBlend(OpcodeArgs) { - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint8_t Select = Op->Src[1].Data.Literal.Value; - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - for (size_t i = 0; i < (16 / ElementSize); ++i) { - if (Select & (1 << i)) { - // This could be optimized if it becomes costly - Dest = _VInsElement(16, ElementSize, i, i, Dest, Src); - } - } - StoreResult(FPRClass, Op, Dest, -1); -} - -template -void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - // The mask is hardcoded to be xmm0 in this instruction - OrderedNode *Mask = _LoadContext(16, offsetof(FEXCore::Core::CPUState, xmm[0]), FPRClass); - // Each element is selected by the high bit of that element size - // Dest[ElementIdx] = Xmm0[ElementIndex][HighBit] ? Src : Dest; - // - // To emulate this on AArch64 - // Arithmetic shift right by the element size, then use BSL to select the registers - Mask = _VSShrI(Size, ElementSize, Mask, (ElementSize * 8) - 1); - auto Result = _VBSL(Mask, Src, Dest); - - StoreResult(FPRClass, Op, Result, -1); -} - -void OpDispatchBuilder::PTestOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - OrderedNode *Test1 = _VAnd(Dest, Src, Size, 1); - OrderedNode *Test2 = _VBic(Src, Dest, Size, 1); - - Test1 = _VPopcount(Size, 1, Test1); - Test2 = _VPopcount(Size, 1, Test2); - - // Element size doesn't matter here - // x86-64 doesn't support a horizontal byte add though - Test1 = _VAddV(Size, 2, Test1); - Test2 = _VAddV(Size, 2, Test2); - - Test1 = _VExtractToGPR(16, 2, Test1, 0); - Test2 = _VExtractToGPR(16, 2, Test2, 0); - - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - - Test1 = _Select(FEXCore::IR::COND_EQ, - Test1, ZeroConst, OneConst, ZeroConst); - - Test2 = _Select(FEXCore::IR::COND_EQ, - Test2, ZeroConst, OneConst, ZeroConst); - - SetRFLAG(Test1); - SetRFLAG(Test2); -} - -void OpDispatchBuilder::PHMINPOSUWOp(OpcodeArgs) { - auto Size = GetSrcSize(Op); - - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - auto Min = _VUMinV(Size, 2, Src); - - std::array Indexes { - _Constant(0), - _Constant(1), - _Constant(2), - _Constant(3), - _Constant(4), - _Constant(5), - _Constant(6), - _Constant(7), - }; - - auto Pos = Indexes[7]; - auto MinGPR = _VExtractToGPR(16, 2, Min, 0); - - // Calculate position - // This doesn't match with ARM behaviour at all - // Instruction returns the minimum matching index - for (size_t i = 8; i > 0; --i) { - auto Element = _VExtractToGPR(16, 2, Src, i - 1); - Pos = _Select(FEXCore::IR::COND_EQ, - Element, MinGPR, Indexes[i - 1], Pos); - } - - // Insert the minimum in to bits [15:0] - OrderedNode *Result = _VMov(Min, 2); - - // Insert position in to bits [18:16] - Result = _VInsGPR(16, 2, Result, Pos, 1); - - StoreResult(FPRClass, Op, Result, -1); -} - -template -void OpDispatchBuilder::DPPOp(OpcodeArgs) { - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint8_t Mask = Op->Src[1].Data.Literal.Value; - uint8_t SrcMask = Mask >> 4; - uint8_t DstMask = Mask & 0xF; - - OrderedNode *ZeroVec = _VectorZero(16); - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - // First step is to do an FMUL - OrderedNode *Temp = _VFMul(16, ElementSize, Dest, Src); - - // Now we zero out elements based on src mask - for (size_t i = 0; i < (16 / ElementSize); ++i) { - if ((SrcMask & (1 << i)) == 0) { - Temp = _VInsElement(16, ElementSize, i, 0, Temp, ZeroVec); - } - } - - // Now we need to do a horizontal add of the elements - // We only have pairwise float add so this needs to be done in steps - Temp = _VFAddP(16, ElementSize, Temp, ZeroVec); - - if constexpr (ElementSize == 4) { - // For 32-bit float we need one more step to add all four results together - Temp = _VFAddP(16, ElementSize, Temp, ZeroVec); - } - - // Now using the destination mask we choose where the result ends up - // It can duplicate and zero results - auto Result = ZeroVec; - - for (size_t i = 0; i < (16 / ElementSize); ++i) { - if (DstMask & (1 << i)) { - Result = _VInsElement(16, ElementSize, i, 0, Result, Temp); - } - } - - StoreResult(FPRClass, Op, Result, -1); -} - -void OpDispatchBuilder::MPSADBWOp(OpcodeArgs) { - LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); - uint8_t Select = Op->Src[1].Data.Literal.Value; - - // Src1 needs to be in byte offset - uint8_t Select_Dest = ((Select & 0b100) >> 2) * 32 / 8; - uint8_t Select_Src2 = Select & 0b11; - - OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); - OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); - - // Src2 will grab a 32bit element and duplicate it across the 128bits - OrderedNode *DupSrc = _VDupElement(16, 4, Src, Select_Src2); - - // Src1/Dest needs a bunch of magic - - // Shift right by selected bytes - // This will give us Dest[15:0], and Dest[79:64] - OrderedNode *Dest1 = _VExtr(16, 1, Dest, Dest, Select_Dest + 0); - // This will give us Dest[31:16], and Dest[95:80] - OrderedNode *Dest2 = _VExtr(16, 1, Dest, Dest, Select_Dest + 1); - // This will give us Dest[47:32], and Dest[111:96] - OrderedNode *Dest3 = _VExtr(16, 1, Dest, Dest, Select_Dest + 2); - // This will give us Dest[63:48], and Dest[127:112] - OrderedNode *Dest4 = _VExtr(16, 1, Dest, Dest, Select_Dest + 3); - - // For each shifted section, we now have two 32-bit values per vector that can be used - // Dest1.S[0] and Dest1.S[1] = Bytes - 0,1,2,3:4,5,6,7 - // Dest2.S[0] and Dest2.S[1] = Bytes - 1,2,3,4:5,6,7,8 - // Dest3.S[0] and Dest3.S[1] = Bytes - 2,3,4,5:6,7,8,9 - // Dest4.S[0] and Dest4.S[1] = Bytes - 3,4,5,6:7,8,9,10 - Dest1 = _VUABDL(16, 1, Dest1, DupSrc); - Dest2 = _VUABDL(16, 1, Dest2, DupSrc); - Dest3 = _VUABDL(16, 1, Dest3, DupSrc); - Dest4 = _VUABDL(16, 1, Dest4, DupSrc); - - // Dest[1,2,3,4] Now contains the data prior to combining - // Temp[0,1,2,3] for each step - - // Each destination now has 16bit x 8 elements in it that were the absolute difference for each byte - // Needs each to be 16bit to store the next step - // Next stage is to sum pairwise - // Dest1: - // ADDP Dest2, Dest1: TmpCombine1 - // ADDP Dest4, Dest3: TmpCombine2 - // TmpCombine1.8H[0] = Dest1.8H[0] + Dest1.8H[1]; - // TmpCombine1.8H[1] = Dest1.8H[2] + Dest1.8H[3]; - // TmpCombine1.8H[2] = Dest1.8H[4] + Dest1.8H[5]; - // TmpCombine1.8H[3] = Dest1.8H[6] + Dest1.8H[7]; - // TmpCombine1.8H[4] = Dest2.8H[0] + Dest2.8H[1]; - // TmpCombine1.8H[5] = Dest2.8H[2] + Dest2.8H[3]; - // TmpCombine1.8H[6] = Dest2.8H[4] + Dest2.8H[5]; - // TmpCombine1.8H[7] = Dest2.8H[6] + Dest2.8H[7]; - // - // ADDP TmpCombine2, TmpCombine1: FinalCombine - // FinalCombine.8H[0] = TmpCombine1.8H[0] + TmpCombine1.8H[1] - // FinalCombine.8H[1] = TmpCombine1.8H[2] + TmpCombine1.8H[3] - // FinalCombine.8H[2] = TmpCombine1.8H[4] + TmpCombine1.8H[5] - // FinalCombine.8H[3] = TmpCombine1.8H[6] + TmpCombine1.8H[7] - // FinalCombine.8H[4] = TmpCombine2.8H[0] + TmpCombine2.8H[1] - // FinalCombine.8H[5] = TmpCombine2.8H[2] + TmpCombine2.8H[3] - // FinalCombine.8H[6] = TmpCombine2.8H[4] + TmpCombine2.8H[5] - // FinalCombine.8H[7] = TmpCombine2.8H[6] + TmpCombine2.8H[7] - - auto TmpCombine1 = _VAddP(16, 2, Dest1, Dest2); - auto TmpCombine2 = _VAddP(16, 2, Dest3, Dest4); - - auto FinalCombine = _VAddP(16, 2, TmpCombine1, TmpCombine2); - - // This now contains our results but they are in the wrong order. - // We need to swizzle the results in to the correct ordering - // Result.8H[0] = FinalCombine.8H[0] - // Result.8H[1] = FinalCombine.8H[2] - // Result.8H[2] = FinalCombine.8H[4] - // Result.8H[3] = FinalCombine.8H[6] - // Result.8H[4] = FinalCombine.8H[1] - // Result.8H[5] = FinalCombine.8H[3] - // Result.8H[6] = FinalCombine.8H[5] - // Result.8H[7] = FinalCombine.8H[7] - - auto Even = _VUnZip(16, 2, FinalCombine, FinalCombine); - auto Odd = _VUnZip2(16, 2, FinalCombine, FinalCombine); - auto Result = _VInsElement(16, 8, 1, 0, Even, Odd); - - StoreResult(FPRClass, Op, Result, -1); -} - void OpDispatchBuilder::UnimplementedOp(OpcodeArgs) { const uint8_t GPRSize = CTX->GetGPRSize(); @@ -8836,7 +4933,7 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) { {0x16, 1, &OpDispatchBuilder::MOVLHPSOp}, {0x17, 1, &OpDispatchBuilder::MOVUPSOp}, {0x28, 2, &OpDispatchBuilder::MOVUPSOp}, - {0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, true, false>}, + {0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, false>}, {0x2B, 1, &OpDispatchBuilder::MOVAPSOp}, {0x2C, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>}, {0x2D, 1, &OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, true>}, @@ -8879,9 +4976,9 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) { {0xC2, 1, &OpDispatchBuilder::VFCMPOp<4, false>}, {0xC6, 1, &OpDispatchBuilder::SHUFOp<4>}, - {0xD1, 1, &OpDispatchBuilder::PSRLDOp<2, true, 0>}, - {0xD2, 1, &OpDispatchBuilder::PSRLDOp<4, true, 0>}, - {0xD3, 1, &OpDispatchBuilder::PSRLDOp<8, true, 0>}, + {0xD1, 1, &OpDispatchBuilder::PSRLDOp<2>}, + {0xD2, 1, &OpDispatchBuilder::PSRLDOp<4>}, + {0xD3, 1, &OpDispatchBuilder::PSRLDOp<8>}, {0xD4, 1, &OpDispatchBuilder::PADDQOp<8>}, {0xD5, 1, &OpDispatchBuilder::VectorALUOp}, {0xD7, 1, &OpDispatchBuilder::MOVMSKOpOne}, // PMOVMSKB @@ -8894,8 +4991,8 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) { {0xDE, 1, &OpDispatchBuilder::VectorALUOp}, {0xDF, 1, &OpDispatchBuilder::ANDNOp}, {0xE0, 1, &OpDispatchBuilder::PAVGOp<1>}, - {0xE1, 1, &OpDispatchBuilder::PSRAOp<2, true, 0>}, - {0xE2, 1, &OpDispatchBuilder::PSRAOp<4, true, 0>}, + {0xE1, 1, &OpDispatchBuilder::PSRAOp<2>}, + {0xE2, 1, &OpDispatchBuilder::PSRAOp<4>}, {0xE3, 1, &OpDispatchBuilder::PAVGOp<2>}, {0xE4, 1, &OpDispatchBuilder::PMULHW}, {0xE5, 1, &OpDispatchBuilder::PMULHW}, @@ -8909,9 +5006,9 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) { {0xEE, 1, &OpDispatchBuilder::VectorALUOp}, {0xEF, 1, &OpDispatchBuilder::VectorALUOp}, - {0xF1, 1, &OpDispatchBuilder::PSLL<2, true, 0>}, - {0xF2, 1, &OpDispatchBuilder::PSLL<4, true, 0>}, - {0xF3, 1, &OpDispatchBuilder::PSLL<8, true, 0>}, + {0xF1, 1, &OpDispatchBuilder::PSLL<2>}, + {0xF2, 1, &OpDispatchBuilder::PSLL<4>}, + {0xF3, 1, &OpDispatchBuilder::PSLL<8>}, {0xF4, 1, &OpDispatchBuilder::PMULLOp<4, false>}, {0xF5, 1, &OpDispatchBuilder::PMADDWD}, {0xF6, 1, &OpDispatchBuilder::PSADBW}, @@ -9126,10 +5223,10 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) { {0x16, 2, &OpDispatchBuilder::MOVHPDOp}, {0x19, 7, &OpDispatchBuilder::NOPOp}, {0x28, 2, &OpDispatchBuilder::MOVAPSOp}, - {0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, true, true>}, + {0x2A, 1, &OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, true>}, {0x2B, 1, &OpDispatchBuilder::MOVAPSOp}, - {0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, false>}, - {0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true, true>}, + {0x2C, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, false>}, + {0x2D, 1, &OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true>}, {0x2E, 2, &OpDispatchBuilder::UCOMISxOp<8>}, {0x40, 16, &OpDispatchBuilder::CMOVOp}, @@ -9179,9 +5276,9 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) { {0xC6, 1, &OpDispatchBuilder::SHUFOp<8>}, {0xD0, 1, &OpDispatchBuilder::ADDSUBPOp<8>}, - {0xD1, 1, &OpDispatchBuilder::PSRLDOp<2, true, 0>}, - {0xD2, 1, &OpDispatchBuilder::PSRLDOp<4, true, 0>}, - {0xD3, 1, &OpDispatchBuilder::PSRLDOp<8, true, 0>}, + {0xD1, 1, &OpDispatchBuilder::PSRLDOp<2>}, + {0xD2, 1, &OpDispatchBuilder::PSRLDOp<4>}, + {0xD3, 1, &OpDispatchBuilder::PSRLDOp<8>}, {0xD4, 1, &OpDispatchBuilder::PADDQOp<8>}, {0xD5, 1, &OpDispatchBuilder::VectorALUOp}, {0xD6, 1, &OpDispatchBuilder::MOVQOp}, @@ -9195,8 +5292,8 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) { {0xDE, 1, &OpDispatchBuilder::VectorALUOp}, {0xDF, 1, &OpDispatchBuilder::ANDNOp}, {0xE0, 1, &OpDispatchBuilder::PAVGOp<1>}, - {0xE1, 1, &OpDispatchBuilder::PSRAOp<2, true, 0>}, - {0xE2, 1, &OpDispatchBuilder::PSRAOp<4, true, 0>}, + {0xE1, 1, &OpDispatchBuilder::PSRAOp<2>}, + {0xE2, 1, &OpDispatchBuilder::PSRAOp<4>}, {0xE3, 1, &OpDispatchBuilder::PAVGOp<2>}, {0xE4, 1, &OpDispatchBuilder::PMULHW}, {0xE5, 1, &OpDispatchBuilder::PMULHW}, @@ -9211,9 +5308,9 @@ void InstallOpcodeHandlers(Context::OperatingMode Mode) { {0xEE, 1, &OpDispatchBuilder::VectorALUOp}, {0xEF, 1, &OpDispatchBuilder::VectorALUOp}, - {0xF1, 1, &OpDispatchBuilder::PSLL<2, true, 0>}, - {0xF2, 1, &OpDispatchBuilder::PSLL<4, true, 0>}, - {0xF3, 1, &OpDispatchBuilder::PSLL<8, true, 0>}, + {0xF1, 1, &OpDispatchBuilder::PSLL<2>}, + {0xF2, 1, &OpDispatchBuilder::PSLL<4>}, + {0xF3, 1, &OpDispatchBuilder::PSLL<8>}, {0xF4, 1, &OpDispatchBuilder::PMULLOp<4, false>}, {0xF5, 1, &OpDispatchBuilder::PMADDWD}, {0xF6, 1, &OpDispatchBuilder::PSADBW}, diff --git a/External/FEXCore/Source/Interface/Core/OpcodeDispatcher.h b/External/FEXCore/Source/Interface/Core/OpcodeDispatcher.h index b1b8d4a62..e740fca11 100644 --- a/External/FEXCore/Source/Interface/Core/OpcodeDispatcher.h +++ b/External/FEXCore/Source/Interface/Core/OpcodeDispatcher.h @@ -270,15 +270,15 @@ public: template void PSHUFDOp(OpcodeArgs); void MOVDOp(OpcodeArgs); - template + template void PSRLDOp(OpcodeArgs); template void PSRLI(OpcodeArgs); template void PSLLI(OpcodeArgs); - template + template void PSLL(OpcodeArgs); - template + template void PSRAOp(OpcodeArgs); void PSRLDQ(OpcodeArgs); void PSLLDQ(OpcodeArgs); @@ -299,9 +299,9 @@ public: void Vector_CVT_Float_To_Float(OpcodeArgs); template void Vector_CVT_Float_To_Int(OpcodeArgs); - template + template void MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs); - template + template void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs); void MASKMOVOp(OpcodeArgs); void MOVBetweenGPR_FPR(OpcodeArgs); @@ -501,9 +501,19 @@ private: uint8_t GetSrcSize(FEXCore::X86Tables::DecodedOp Op) const; template - void SetRFLAG(OrderedNode *Value); - void SetRFLAG(OrderedNode *Value, unsigned BitOffset); - OrderedNode *GetRFLAG(unsigned BitOffset); + void SetRFLAG(OrderedNode *Value) { + flagsOp = FLAGS_OP_NONE; + _StoreFlag(_Bfe(1, 0, Value), BitOffset); + } + + void SetRFLAG(OrderedNode *Value, unsigned BitOffset) { + flagsOp = FLAGS_OP_NONE; + _StoreFlag(_Bfe(1, 0, Value), BitOffset); + } + + OrderedNode *GetRFLAG(unsigned BitOffset) { + return _LoadFlag(BitOffset); + } OrderedNode *SelectCC(uint8_t OP, OrderedNode *TrueValue, OrderedNode *FalseValue); diff --git a/External/FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp b/External/FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp new file mode 100644 index 000000000..28432f63c --- /dev/null +++ b/External/FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp @@ -0,0 +1,58 @@ +/* +$info$ +tags: frontend|x86-to-ir, opcodes|dispatcher-implementations +desc: Handles x86/64 Crypto instructions to IR +$end_info$ +*/ + +#include "Interface/Core/OpcodeDispatcher.h" + +#include + +namespace FEXCore::IR { +#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op + +void OpDispatchBuilder::AESImcOp(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + auto Res = _VAESImc(Src); + StoreResult(FPRClass, Op, Res, -1); +} + +void OpDispatchBuilder::AESEncOp(OpcodeArgs) { + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + auto Res = _VAESEnc(Dest, Src); + StoreResult(FPRClass, Op, Res, -1); +} + +void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) { + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + auto Res = _VAESEncLast(Dest, Src); + StoreResult(FPRClass, Op, Res, -1); +} + +void OpDispatchBuilder::AESDecOp(OpcodeArgs) { + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + auto Res = _VAESDec(Dest, Src); + StoreResult(FPRClass, Op, Res, -1); +} + +void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) { + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + auto Res = _VAESDecLast(Dest, Src); + StoreResult(FPRClass, Op, Res, -1); +} + +void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint64_t RCON = Op->Src[1].Data.Literal.Value; + + auto Res = _VAESKeyGenAssist(Src, RCON); + StoreResult(FPRClass, Op, Res, -1); +} + +} diff --git a/External/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp b/External/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp new file mode 100644 index 000000000..f294b5cf6 --- /dev/null +++ b/External/FEXCore/Source/Interface/Core/OpcodeDispatcher/Flags.cpp @@ -0,0 +1,810 @@ +/* +$info$ +tags: frontend|x86-to-ir, opcodes|dispatcher-implementations +desc: Handles x86/64 flag generation +$end_info$ +*/ + +#include "Interface/Core/OpcodeDispatcher.h" + +#include + +namespace FEXCore::IR { +constexpr std::array FlagOffsets = { + FEXCore::X86State::RFLAG_CF_LOC, + FEXCore::X86State::RFLAG_PF_LOC, + FEXCore::X86State::RFLAG_AF_LOC, + FEXCore::X86State::RFLAG_ZF_LOC, + FEXCore::X86State::RFLAG_SF_LOC, + FEXCore::X86State::RFLAG_TF_LOC, + FEXCore::X86State::RFLAG_IF_LOC, + FEXCore::X86State::RFLAG_DF_LOC, + FEXCore::X86State::RFLAG_OF_LOC, + FEXCore::X86State::RFLAG_IOPL_LOC, + FEXCore::X86State::RFLAG_NT_LOC, + FEXCore::X86State::RFLAG_RF_LOC, + FEXCore::X86State::RFLAG_VM_LOC, + FEXCore::X86State::RFLAG_AC_LOC, + FEXCore::X86State::RFLAG_VIF_LOC, + FEXCore::X86State::RFLAG_VIP_LOC, + FEXCore::X86State::RFLAG_ID_LOC, +}; + +void OpDispatchBuilder::SetPackedRFLAG(bool Lower8, OrderedNode *Src) { + uint8_t NumFlags = FlagOffsets.size(); + if (Lower8) { + NumFlags = 5; + } + auto OneConst = _Constant(1); + for (int i = 0; i < NumFlags; ++i) { + auto Tmp = _And(_Lshr(Src, _Constant(FlagOffsets[i])), OneConst); + SetRFLAG(Tmp, FlagOffsets[i]); + } +} + +OrderedNode *OpDispatchBuilder::GetPackedRFLAG(bool Lower8) { + OrderedNode *Original = _Constant(2); + uint8_t NumFlags = FlagOffsets.size(); + if (Lower8) { + NumFlags = 5; + } + + for (int i = 0; i < NumFlags; ++i) { + OrderedNode *Flag = _LoadFlag(FlagOffsets[i]); + Flag = _Bfe(4, 32, 0, Flag); + Flag = _Lshl(Flag, _Constant(FlagOffsets[i])); + Original = _Or(Original, Flag); + } + return Original; +} + +void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) { + auto Size = GetSrcSize(Op) * 8; + // AF + { + OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); + AFRes = _Bfe(1, 4, AFRes); + SetRFLAG(AFRes); + } + + // SF + { + auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); + + auto LshrOp = _Lshr(Res, SignBitConst); + SetRFLAG(LshrOp); + } + + // PF + if (!CTX->Config.ABINoPF) { + auto PopCountOp = _Popcount(_And(Res, _Constant(0xFF))); + + auto XorOp = _Xor(PopCountOp, _Constant(1)); + SetRFLAG(XorOp); + } else { + _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); + } + + // ZF + { + auto SelectOp = _Select(FEXCore::IR::COND_EQ, + Res, _Constant(0), _Constant(1), _Constant(0)); + SetRFLAG(SelectOp); + } + + // CF + // Unsigned + { + auto SelectOpLT = _Select(FEXCore::IR::COND_ULT, Res, Src2, _Constant(1), _Constant(0)); + auto SelectOpLE = _Select(FEXCore::IR::COND_ULE, Res, Src2, _Constant(1), _Constant(0)); + auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, _Constant(1), SelectOpLE, SelectOpLT); + SetRFLAG(SelectCF); + } + + // OF + // Signed + { + auto NegOne = _Constant(~0ULL); + auto XorOp1 = _Xor(_Xor(Src1, Src2), NegOne); + auto XorOp2 = _Xor(Res, Src1); + OrderedNode *AndOp1 = _And(XorOp1, XorOp2); + + switch (Size) { + case 8: + AndOp1 = _Bfe(1, 7, AndOp1); + break; + case 16: + AndOp1 = _Bfe(1, 15, AndOp1); + break; + case 32: + AndOp1 = _Bfe(1, 31, AndOp1); + break; + case 64: + AndOp1 = _Bfe(1, 63, AndOp1); + break; + default: LOGMAN_MSG_A("Unknown BFESize: %d", Size); break; + } + SetRFLAG(AndOp1); + } +} + +void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) { + // AF + { + OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); + AFRes = _Bfe(1, 4, AFRes); + SetRFLAG(AFRes); + } + + // SF + { + auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); + + auto LshrOp = _Lshr(Res, SignBitConst); + SetRFLAG(LshrOp); + } + + // PF + if (!CTX->Config.ABINoPF) { + auto PopCountOp = _Popcount(_And(Res, _Constant(0xFF))); + + auto XorOp = _Xor(PopCountOp, _Constant(1)); + SetRFLAG(XorOp); + } else { + _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); + } + + // ZF + { + auto SelectOp = _Select(FEXCore::IR::COND_EQ, + Res, _Constant(0), _Constant(1), _Constant(0)); + SetRFLAG(SelectOp); + } + + // CF + // Unsigned + { + auto SelectOpLT = _Select(FEXCore::IR::COND_UGT, Res, Src1, _Constant(1), _Constant(0)); + auto SelectOpLE = _Select(FEXCore::IR::COND_UGE, Res, Src1, _Constant(1), _Constant(0)); + auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, _Constant(1), SelectOpLE, SelectOpLT); + SetRFLAG(SelectCF); + } + + // OF + // Signed + { + auto XorOp1 = _Xor(Src1, Src2); + auto XorOp2 = _Xor(Res, Src1); + OrderedNode *AndOp1 = _And(XorOp1, XorOp2); + + switch (GetSrcSize(Op)) { + case 1: + AndOp1 = _Bfe(1, 7, AndOp1); + break; + case 2: + AndOp1 = _Bfe(1, 15, AndOp1); + break; + case 4: + AndOp1 = _Bfe(1, 31, AndOp1); + break; + case 8: + AndOp1 = _Bfe(1, 63, AndOp1); + break; + default: LOGMAN_MSG_A("Unknown BFESize: %d", GetSrcSize(Op)); break; + } + SetRFLAG(AndOp1); + } +} + +void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) { + // AF + { + OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); + AFRes = _Bfe(1, 4, AFRes); + SetRFLAG(AFRes); + } + + // SF + { + auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); + + auto LshrOp = _Lshr(Res, SignBitConst); + SetRFLAG(LshrOp); + } + + // PF + if (!CTX->Config.ABINoPF) { + auto EightBitMask = _Constant(0xFF); + auto PopCountOp = _Popcount(_And(Res, EightBitMask)); + auto XorOp = _Xor(PopCountOp, _Constant(1)); + SetRFLAG(XorOp); + } else { + _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); + } + + // ZF + { + auto ZeroConst = _Constant(0); + auto OneConst = _Constant(1); + auto SelectOp = _Select(FEXCore::IR::COND_EQ, + Res, ZeroConst, OneConst, ZeroConst); + SetRFLAG(SelectOp); + } + + // CF + if (UpdateCF) { + auto ZeroConst = _Constant(0); + auto OneConst = _Constant(1); + + auto SelectOp = _Select(FEXCore::IR::COND_ULT, + Src1, Src2, OneConst, ZeroConst); + + SetRFLAG(SelectOp); + } + // OF + { + auto XorOp1 = _Xor(Src1, Src2); + auto XorOp2 = _Xor(Res, Src1); + OrderedNode *FinalAnd = _And(XorOp1, XorOp2); + + FinalAnd = _Bfe(1, GetSrcSize(Op) * 8 - 1, FinalAnd); + + SetRFLAG(FinalAnd); + } +} + +void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, bool UpdateCF) { + // AF + { + OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); + AFRes = _Bfe(1, 4, AFRes); + SetRFLAG(AFRes); + } + + // SF + { + auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); + + auto LshrOp = _Lshr(Res, SignBitConst); + SetRFLAG(LshrOp); + } + + // PF + if (!CTX->Config.ABINoPF) { + auto EightBitMask = _Constant(0xFF); + auto PopCountOp = _Popcount(_And(Res, EightBitMask)); + auto XorOp = _Xor(PopCountOp, _Constant(1)); + SetRFLAG(XorOp); + } else { + _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); + } + + // ZF + { + auto SelectOp = _Select(FEXCore::IR::COND_EQ, + Res, _Constant(0), _Constant(1), _Constant(0)); + SetRFLAG(SelectOp); + } + // CF + if (UpdateCF) { + auto SelectOp = _Select(FEXCore::IR::COND_ULT, Res, Src2, _Constant(1), _Constant(0)); + + SetRFLAG(SelectOp); + } + + // OF + { + auto NegOne = _Constant(~0ULL); + auto XorOp1 = _Xor(_Xor(Src1, Src2), NegOne); + auto XorOp2 = _Xor(Res, Src1); + + OrderedNode *AndOp1 = _And(XorOp1, XorOp2); + + switch (GetSrcSize(Op)) { + case 1: + AndOp1 = _Bfe(1, 7, AndOp1); + break; + case 2: + AndOp1 = _Bfe(1, 15, AndOp1); + break; + case 4: + AndOp1 = _Bfe(1, 31, AndOp1); + break; + case 8: + AndOp1 = _Bfe(1, 63, AndOp1); + break; + default: LOGMAN_MSG_A("Unknown BFESize: %d", GetSrcSize(Op)); break; + } + SetRFLAG(AndOp1); + } +} + +void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *High) { + // PF/AF/ZF/SF + // Undefined + { + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + } + + // CF/OF + { + // CF and OF are set if the result of the operation can't be fit in to the destination register + // If the value can fit then the top bits will be zero + + auto SignBit = _Sbfe(1, GetSrcSize(Op) * 8 - 1, Res); + + auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, SignBit, _Constant(0), _Constant(1)); + + SetRFLAG(SelectOp); + SetRFLAG(SelectOp); + } +} + +void OpDispatchBuilder::GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *High) { + // AF/SF/PF/ZF + // Undefined + { + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + } + + // CF/OF + { + // CF and OF are set if the result of the operation can't be fit in to the destination register + // The result register will be all zero if it can't fit due to how multiplication behaves + + auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, _Constant(0), _Constant(0), _Constant(1)); + + SetRFLAG(SelectOp); + SetRFLAG(SelectOp); + } +} + +void OpDispatchBuilder::GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { + // AF + { + // Undefined + // Set to zero anyway + SetRFLAG(_Constant(0)); + } + + // SF + { + auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); + + auto LshrOp = _Lshr(Res, SignBitConst); + SetRFLAG(LshrOp); + } + + // PF + if (!CTX->Config.ABINoPF) { + auto EightBitMask = _Constant(0xFF); + auto PopCountOp = _Popcount(_And(Res, EightBitMask)); + auto XorOp = _Xor(PopCountOp, _Constant(1)); + SetRFLAG(XorOp); + } else { + _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); + } + + // ZF + { + auto SelectOp = _Select(FEXCore::IR::COND_EQ, + Res, _Constant(0), _Constant(1), _Constant(0)); + SetRFLAG(SelectOp); + } + + // CF/OF + { + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + } +} + +#define COND_FLAG_SET(cond, flag, newflag) \ +auto oldflag = GetRFLAG(FEXCore::X86State::flag);\ +auto newval = _Select(FEXCore::IR::COND_EQ, cond, _Constant(0), oldflag, newflag);\ +SetRFLAG(newval); + +void OpDispatchBuilder::GenerateFlags_ShiftLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { + // CF + { + // Extract the last bit shifted in to CF + auto Size = _Constant(GetSrcSize(Op) * 8); + auto ShiftAmt = _Sub(Size, Src2); + auto LastBit = _And(_Lshr(Src1, ShiftAmt), _Constant(1)); + COND_FLAG_SET(Src2, RFLAG_CF_LOC, LastBit); + } + + // PF + if (!CTX->Config.ABINoPF) { + auto EightBitMask = _Constant(0xFF); + auto PopCountOp = _Popcount(_And(Res, EightBitMask)); + auto XorOp = _Xor(PopCountOp, _Constant(1)); + COND_FLAG_SET(Src2, RFLAG_PF_LOC, XorOp); + } else { + _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); + } + + // AF + { + // Undefined + // Set to zero anyway + COND_FLAG_SET(Src2, RFLAG_AF_LOC, _Constant(0)); + } + + // ZF + { + auto SelectOp = _Select(FEXCore::IR::COND_EQ, + Res, _Constant(0), _Constant(1), _Constant(0)); + COND_FLAG_SET(Src2, RFLAG_ZF_LOC, SelectOp); + } + + // SF + { + auto val = _Bfe(1, GetSrcSize(Op) * 8 - 1, Res); + COND_FLAG_SET(Src2, RFLAG_SF_LOC, val); + } + + // OF + { + // In the case of left shift. OF is only set from the result of XOR + // When Shift > 1 then OF is undefined + auto val = _Bfe(1, GetSrcSize(Op) * 8 - 1, _Xor(Src1, Res)); + COND_FLAG_SET(Src2, RFLAG_OF_LOC, val); + } +} + +void OpDispatchBuilder::GenerateFlags_ShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { + // CF + { + // Extract the last bit shifted in to CF + auto ShiftAmt = _Sub(Src2, _Constant(1)); + auto LastBit = _And(_Lshr(Src1, ShiftAmt), _Constant(1)); + COND_FLAG_SET(Src2, RFLAG_CF_LOC, LastBit); + } + + // PF + if (!CTX->Config.ABINoPF) { + auto EightBitMask = _Constant(0xFF); + auto PopCountOp = _Popcount(_And(Res, EightBitMask)); + auto XorOp = _Xor(PopCountOp, _Constant(1)); + COND_FLAG_SET(Src2, RFLAG_PF_LOC, XorOp); + } else { + _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); + } + + // AF + { + // Undefined + // Set to zero anyway + COND_FLAG_SET(Src2, RFLAG_AF_LOC, _Constant(0)); + } + + // ZF + { + auto SelectOp = _Select(FEXCore::IR::COND_EQ, + Res, _Constant(0), _Constant(1), _Constant(0)); + COND_FLAG_SET(Src2, RFLAG_ZF_LOC, SelectOp); + } + + // SF + { + auto val =_Bfe(1, GetSrcSize(Op) * 8 - 1, Res); + COND_FLAG_SET(Src2, RFLAG_SF_LOC, val); + } + + // OF + { + // Only defined when Shift is 1 else undefined + // OF flag is set if a sign change occurred + auto val = _Bfe(1, GetSrcSize(Op) * 8 - 1, _Xor(Src1, Res)); + COND_FLAG_SET(Src2, RFLAG_OF_LOC, val); + } +} + +void OpDispatchBuilder::GenerateFlags_SignShiftRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { + // CF + { + // Extract the last bit shifted in to CF + auto ShiftAmt = _Sub(Src2, _Constant(1)); + auto LastBit = _And(_Lshr(Src1, ShiftAmt), _Constant(1)); + COND_FLAG_SET(Src2, RFLAG_CF_LOC, LastBit); + } + + // PF + if (!CTX->Config.ABINoPF) { + auto EightBitMask = _Constant(0xFF); + auto PopCountOp = _Popcount(_And(Res, EightBitMask)); + auto XorOp = _Xor(PopCountOp, _Constant(1)); + COND_FLAG_SET(Src2, RFLAG_PF_LOC, XorOp); + } else { + _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); + } + + // AF + { + // Undefined + // Set to zero anyway + COND_FLAG_SET(Src2, RFLAG_AF_LOC, _Constant(0)); + } + + // ZF + { + auto SelectOp = _Select(FEXCore::IR::COND_EQ, + Res, _Constant(0), _Constant(1), _Constant(0)); + COND_FLAG_SET(Src2, RFLAG_ZF_LOC, SelectOp); + } + + // SF + { + auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); + + auto LshrOp = _Lshr(Res, SignBitConst); + COND_FLAG_SET(Src2, RFLAG_SF_LOC, LshrOp); + } + + // OF + { + COND_FLAG_SET(Src2, RFLAG_OF_LOC, _Constant(0)); + } +} + +void OpDispatchBuilder::GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) { + // No flags changed if shift is zero + if (Shift == 0) return; + + // CF + { + // Extract the last bit shifted in to CF + SetRFLAG(_Bfe(1, GetSrcSize(Op) * 8 - Shift, Src1)); + } + + // PF + if (!CTX->Config.ABINoPF) { + auto EightBitMask = _Constant(0xFF); + auto PopCountOp = _Popcount(_And(Res, EightBitMask)); + auto XorOp = _Xor(PopCountOp, _Constant(1)); + SetRFLAG(XorOp); + } else { + _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); + } + + // AF + { + // Undefined + // Set to zero anyway + SetRFLAG(_Constant(0)); + } + + // ZF + { + auto SelectOp = _Select(FEXCore::IR::COND_EQ, + Res, _Constant(0), _Constant(1), _Constant(0)); + SetRFLAG(SelectOp); + } + + // SF + { + auto LshrOp = _Bfe(1, GetSrcSize(Op) * 8 - 1, Res); + + SetRFLAG(LshrOp); + + // OF + // In the case of left shift. OF is only set from the result of XOR + if (Shift == 1) { + auto SourceBit = _Bfe(1, GetSrcSize(Op) * 8 - 1, Src1); + SetRFLAG(_Xor(SourceBit, LshrOp)); + } + } +} + +void OpDispatchBuilder::GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) { + // No flags changed if shift is zero + if (Shift == 0) return; + + // CF + { + // Extract the last bit shifted in to CF + SetRFLAG(_Bfe(1, Shift-1, Src1)); + } + + // PF + if (!CTX->Config.ABINoPF) { + auto EightBitMask = _Constant(0xFF); + auto PopCountOp = _Popcount(_And(Res, EightBitMask)); + auto XorOp = _Xor(PopCountOp, _Constant(1)); + SetRFLAG(XorOp); + } else { + _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); + } + + // AF + { + // Undefined + // Set to zero anyway + SetRFLAG(_Constant(0)); + } + + // ZF + { + auto SelectOp = _Select(FEXCore::IR::COND_EQ, + Res, _Constant(0), _Constant(1), _Constant(0)); + SetRFLAG(SelectOp); + } + + // SF + { + auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); + + auto LshrOp = _Lshr(Res, SignBitConst); + SetRFLAG(LshrOp); + + // OF + // Only defined when Shift is 1 else undefined + // Only is set if the top bit was set to 1 when shifted + // So it is set to same value as SF + if (Shift == 1) { + SetRFLAG(_Constant(0)); + } + } +} + +void OpDispatchBuilder::GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) { + // No flags changed if shift is zero + if (Shift == 0) return; + + // CF + { + // Extract the last bit shifted in to CF + SetRFLAG(_Bfe(1, Shift-1, Src1)); + } + + // PF + if (!CTX->Config.ABINoPF) { + auto EightBitMask = _Constant(0xFF); + auto PopCountOp = _Popcount(_And(Res, EightBitMask)); + auto XorOp = _Xor(PopCountOp, _Constant(1)); + SetRFLAG(XorOp); + } else { + _InvalidateFlags(1UL << FEXCore::X86State::RFLAG_PF_LOC); + } + + // AF + { + // Undefined + // Set to zero anyway + SetRFLAG(_Constant(0)); + } + + // ZF + { + auto SelectOp = _Select(FEXCore::IR::COND_EQ, + Res, _Constant(0), _Constant(1), _Constant(0)); + SetRFLAG(SelectOp); + } + + // SF + { + auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); + + auto LshrOp = _Lshr(Res, SignBitConst); + SetRFLAG(LshrOp); + } + + // OF + { + // Only defined when Shift is 1 else undefined + // Is set to the MSB of the original value + if (Shift == 1) { + SetRFLAG(_Bfe(1, GetSrcSize(Op) * 8 - 1, Src1)); + } + } +} + +void OpDispatchBuilder::GenerateFlags_RotateRight(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { + auto OpSize = GetSrcSize(Op) * 8; + + // Extract the last bit shifted in to CF + auto NewCF = _Bfe(1, OpSize - 1, Res); + + // CF + { + auto OldCF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC); + auto CF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldCF, NewCF); + + // Extract the last bit shifted in to CF + SetRFLAG(CF); + } + + // OF + { + auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC); + + // OF is set to the XOR of the new CF bit and the most significant bit of the result + auto NewOF = _Xor(_Bfe(1, OpSize - 2, Res), NewCF); + + // If shift == 0, don't update flags + auto OF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldOF, NewOF); + + SetRFLAG(OF); + } +} + +void OpDispatchBuilder::GenerateFlags_RotateLeft(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { + auto OpSize = GetSrcSize(Op) * 8; + + // Extract the last bit shifted in to CF + //auto Size = _Constant(GetSrcSize(Res) * 8); + //auto ShiftAmt = _Sub(Size, Src2); + auto NewCF = _Bfe(1, 0, Res); + + // CF + { + auto OldCF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC); + auto CF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldCF, NewCF); + + // Extract the last bit shifted in to CF + SetRFLAG(CF); + } + + // OF + { + auto OldOF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC); + // OF is set to the XOR of the new CF bit and the most significant bit of the result + auto NewOF = _Xor(_Bfe(1, OpSize - 1, Res), NewCF); + + auto OF = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), OldOF, NewOF); + + // If shift == 0, don't update flags + SetRFLAG(OF); + } +} + +void OpDispatchBuilder::GenerateFlags_RotateRightImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) { + if (Shift == 0) return; + + auto OpSize = GetSrcSize(Op) * 8; + + auto NewCF = _Bfe(1, OpSize - Shift, Src1); + + // CF + { + // Extract the last bit shifted in to CF + SetRFLAG(NewCF); + } + + // OF + { + if (Shift == 1) { + // OF is set to the XOR of the new CF bit and the most significant bit of the result + SetRFLAG(_Xor(_Bfe(1, OpSize - 1, Res), NewCF)); + } + } +} + +void OpDispatchBuilder::GenerateFlags_RotateLeftImmediate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, uint64_t Shift) { + if (Shift == 0) return; + + auto OpSize = GetSrcSize(Op) * 8; + + // CF + { + // Extract the last bit shifted in to CF + SetRFLAG(_Bfe(1, Shift, Src1)); + } + + // OF + { + if (Shift == 1) { + // OF is the top two MSBs XOR'd together + SetRFLAG(_Xor(_Bfe(1, OpSize - 1, Src1), _Bfe(1, OpSize - 2, Src1))); + } + } +} + +} diff --git a/External/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp b/External/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp new file mode 100644 index 000000000..17e308d55 --- /dev/null +++ b/External/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp @@ -0,0 +1,2383 @@ +/* +$info$ +tags: frontend|x86-to-ir, opcodes|dispatcher-implementations +desc: Handles x86/64 Vector instructions to IR +$end_info$ +*/ + +#include "Interface/Core/OpcodeDispatcher.h" + +#include + +namespace FEXCore::IR { +#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op + +void OpDispatchBuilder::MOVVectorOp(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 1); + StoreResult(FPRClass, Op, Src, 1); +} + +void OpDispatchBuilder::MOVAPSOp(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + StoreResult(FPRClass, Op, Src, -1); +} + +void OpDispatchBuilder::MOVUPSOp(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 1); + StoreResult(FPRClass, Op, Src, 1); +} + +void OpDispatchBuilder::MOVLHPSOp(OpcodeArgs) { + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, 8); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8); + auto Result = _VInsElement(16, 8, 1, 0, Dest, Src); + StoreResult(FPRClass, Op, Result, 8); +} + +void OpDispatchBuilder::MOVHPDOp(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + // This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 128bit + if (Op->Dest.IsGPR()) { + // If the destination is a GPR then the source is memory + // xmm1[127:64] = src + OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1); + auto Result = _VInsElement(16, 8, 1, 0, Dest, Src); + StoreResult(FPRClass, Op, Result, -1); + } + else { + // In this case memory is the destination and the high bits of the XMM are source + // Mem64 = xmm1[127:64] + auto Result = _VExtractToGPR(16, 8, Src, 1); + StoreResult(GPRClass, Op, Result, -1); + } +} + +void OpDispatchBuilder::MOVLPOp(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8); + if (Op->Dest.IsGPR()) { + // xmm, xmm is movhlps special case + if (Op->Src[0].IsGPR()) { + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, 8, 16); + Src = _VExtractElement(16, 8, Src, 1); + auto Result = _VInsScalarElement(16, 8, 0, Dest, Src); + StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 16, 16); + } + else { + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, 8, 16); + auto Result = _VInsScalarElement(16, 8, 0, Dest, Src); + StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, 8, 16); + } + } + else { + StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 8, 8); + } +} + +void OpDispatchBuilder::MOVSHDUPOp(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8); + OrderedNode *Result = _VInsElement(16, 4, 3, 3, Src, Src); + Result = _VInsElement(16, 4, 2, 3, Result, Src); + Result = _VInsElement(16, 4, 1, 1, Result, Src); + Result = _VInsElement(16, 4, 0, 1, Result, Src); + StoreResult(FPRClass, Op, Result, -1); +} + +void OpDispatchBuilder::MOVSLDUPOp(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, 8); + OrderedNode *Result = _VInsElement(16, 4, 3, 2, Src, Src); + Result = _VInsElement(16, 4, 2, 2, Result, Src); + Result = _VInsElement(16, 4, 1, 0, Result, Src); + Result = _VInsElement(16, 4, 0, 0, Result, Src); + StoreResult(FPRClass, Op, Result, -1); +} + +void OpDispatchBuilder::MOVSSOp(OpcodeArgs) { + if (Op->Dest.IsGPR() && Op->Src[0].IsGPR()) { + // MOVSS xmm1, xmm2 + OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1); + OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1); + auto Result = _VInsScalarElement(16, 4, 0, Dest, Src); + StoreResult(FPRClass, Op, Result, -1); + } + else if (Op->Dest.IsGPR()) { + // MOVSS xmm1, mem32 + // xmm1[127:0] <- zext(mem32) + OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1); + StoreResult(FPRClass, Op, Src, -1); + } + else { + // MOVSS mem32, xmm1 + OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1); + StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 4, -1); + } +} + +void OpDispatchBuilder::MOVSDOp(OpcodeArgs) { + if (Op->Dest.IsGPR() && Op->Src[0].IsGPR()) { + // xmm1[63:0] <- xmm2[63:0] + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + auto Result = _VInsScalarElement(16, 8, 0, Dest, Src); + StoreResult(FPRClass, Op, Result, -1); + } + else if (Op->Dest.IsGPR()) { + // xmm1[127:0] <- zext(mem64) + OrderedNode *Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 8, Op->Flags, -1); + StoreResult(FPRClass, Op, Src, -1); + } + else { + // In this case memory is the destination and the low bits of the XMM are source + // Mem64 = xmm2[63:0] + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, 8, -1); + } +} + +template +void OpDispatchBuilder::PADDQOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + auto ALUOp = _VAdd(Size, ElementSize, Dest, Src); + StoreResult(FPRClass, Op, ALUOp, -1); +} + +template +void OpDispatchBuilder::PADDQOp<1>(OpcodeArgs); +template +void OpDispatchBuilder::PADDQOp<2>(OpcodeArgs); +template +void OpDispatchBuilder::PADDQOp<4>(OpcodeArgs); +template +void OpDispatchBuilder::PADDQOp<8>(OpcodeArgs); + +template +void OpDispatchBuilder::PSUBQOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + auto ALUOp = _VSub(Size, ElementSize, Dest, Src); + StoreResult(FPRClass, Op, ALUOp, -1); +} + +template +void OpDispatchBuilder::PSUBQOp<1>(OpcodeArgs); +template +void OpDispatchBuilder::PSUBQOp<2>(OpcodeArgs); +template +void OpDispatchBuilder::PSUBQOp<4>(OpcodeArgs); +template +void OpDispatchBuilder::PSUBQOp<8>(OpcodeArgs); + +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + auto ALUOp = _VAdd(Size, ElementSize, Dest, Src); + // Overwrite our IR's op type + ALUOp.first->Header.Op = IROp; + + StoreResult(FPRClass, Op, ALUOp, -1); +} + +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); + +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); + +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); + +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorALUOp(OpcodeArgs); + +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + // If OpSize == ElementSize then it only does the lower scalar op + auto ALUOp = _VAdd(ElementSize, ElementSize, Dest, Src); + // Overwrite our IR's op type + ALUOp.first->Header.Op = IROp; + + OrderedNode* Result = ALUOp; + + if (Size != ElementSize) { + // Insert the lower bits + Result = _VInsScalarElement(Size, ElementSize, 0, Dest, Result); + } + + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorScalarALUOp(OpcodeArgs); + +template +void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + if constexpr (Scalar) { + Size = ElementSize; + } + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + auto ALUOp = _VFSqrt(Size, ElementSize, Src); + // Overwrite our IR's op type + ALUOp.first->Header.Op = IROp; + + if constexpr (Scalar) { + // Insert the lower bits + auto Result = _VInsScalarElement(GetSrcSize(Op), ElementSize, 0, Dest, ALUOp); + StoreResult(FPRClass, Op, Result, -1); + } + else { + StoreResult(FPRClass, Op, ALUOp, -1); + } +} + +template +void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs); + +template +void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs); + +template +void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs); + +template +void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs); +template +void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs); + +void OpDispatchBuilder::MOVQOp(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + // This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 128bit + if (Op->Dest.IsGPR()) { + const auto gpr = Op->Dest.Data.GPR.GPR; + _StoreContext(FPRClass, 8, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][0]), Src); + auto Const = _Constant(0); + _StoreContext(GPRClass, 8, offsetof(FEXCore::Core::CPUState, xmm[gpr - FEXCore::X86State::REG_XMM_0][1]), Const); + } + else { + // This is simple, just store the result + StoreResult(FPRClass, Op, Src, -1); + } +} + +template +void OpDispatchBuilder::MOVMSKOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + uint8_t NumElements = Size / ElementSize; + + OrderedNode *CurrentVal = _Constant(0); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + for (unsigned i = 0; i < NumElements; ++i) { + // Extract the top bit of the element + OrderedNode *Tmp = _VExtractToGPR(16, ElementSize, Src, i); + Tmp = _Bfe(1, ElementSize * 8 - 1, Tmp); + + // Shift it to the correct location + Tmp = _Lshl(Tmp, _Constant(i)); + + // Or it with the current value + CurrentVal = _Or(CurrentVal, Tmp); + } + StoreResult(GPRClass, Op, CurrentVal, -1); +} + +template +void OpDispatchBuilder::MOVMSKOp<4>(OpcodeArgs); +template +void OpDispatchBuilder::MOVMSKOp<8>(OpcodeArgs); + +void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + //TODO: We could remove this VCastFromGOR + VInsGPR pair if we had a VDUPFromGPR instruction that maps directly to AArch64. + auto M = _Constant(0x80'40'20'10'08'04'02'01ULL); + OrderedNode *VMask = _VCastFromGPR(16, 8, M); + VMask = _VInsGPR(16, 8, VMask, M, 1); + + auto VCMP = _VCMPLTZ(Src, 16, 1); + auto VAnd = _VAnd(VCMP, VMask, 16, 1); + + auto VAdd1 = _VAddP(VAnd, VAnd, 16, 1); + auto VAdd2 = _VAddP(VAdd1, VAdd1, 8, 1); + auto VAdd3 = _VAddP(VAdd2, VAdd2, 8, 1); + + StoreResult(GPRClass, Op, _VExtractToGPR(16, 2, VAdd3, 0), -1); +} + +template +void OpDispatchBuilder::PUNPCKLOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + auto ALUOp = _VZip(Size, ElementSize, Dest, Src); + StoreResult(FPRClass, Op, ALUOp, -1); +} + +template +void OpDispatchBuilder::PUNPCKLOp<1>(OpcodeArgs); +template +void OpDispatchBuilder::PUNPCKLOp<2>(OpcodeArgs); +template +void OpDispatchBuilder::PUNPCKLOp<4>(OpcodeArgs); +template +void OpDispatchBuilder::PUNPCKLOp<8>(OpcodeArgs); + +template +void OpDispatchBuilder::PUNPCKHOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + auto ALUOp = _VZip2(Size, ElementSize, Dest, Src); + StoreResult(FPRClass, Op, ALUOp, -1); +} + +template +void OpDispatchBuilder::PUNPCKHOp<1>(OpcodeArgs); +template +void OpDispatchBuilder::PUNPCKHOp<2>(OpcodeArgs); +template +void OpDispatchBuilder::PUNPCKHOp<4>(OpcodeArgs); +template +void OpDispatchBuilder::PUNPCKHOp<8>(OpcodeArgs); + +void OpDispatchBuilder::PSHUFBOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + // PSHUFB doesn't 100% match VTBL behaviour + // VTBL will set the element zero if the index is greater than the number of elements + // In the array + // Bit 7 is the only bit that is supposed to set elements to zero with PSHUFB + // Mask the selection bits and top bit correctly + // Bits [6:4] is reserved for 128bit + // Bits [6:3] is reserved for 64bit + if (Size == 8) { + auto MaskVector = _VectorImm(0b1000'0111, Size, 1); + Src = _VAnd(Size, Size, Src, MaskVector); + } + else { + auto MaskVector = _VectorImm(0b1000'1111, Size, 1); + Src = _VAnd(Size, Size, Src, MaskVector); + } + auto Res = _VTBL1(Size, Dest, Src); + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::PSHUFDOp(OpcodeArgs) { + LOGMAN_THROW_A(ElementSize != 0, "What. No element size?"); + auto Size = GetSrcSize(Op); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + uint8_t Shuffle = Op->Src[1].Data.Literal.Value; + + uint8_t NumElements = Size / ElementSize; + + // 16bit is a bit special of a shuffle + // It only ever operates on half the register + // Then there is a high and low variant of the instruction to determine where the destination goes + // and where the source comes from + if constexpr (HalfSize) { + NumElements /= 2; + } + + uint8_t BaseElement = Low ? 0 : NumElements; + + auto Dest = Src; + for (uint8_t Element = 0; Element < NumElements; ++Element) { + Dest = _VInsElement(Size, ElementSize, BaseElement + Element, BaseElement + (Shuffle & 0b11), Dest, Src); + Shuffle >>= 2; + } + + StoreResult(FPRClass, Op, Dest, -1); +} + +template +void OpDispatchBuilder::PSHUFDOp<2, false, true>(OpcodeArgs); +template +void OpDispatchBuilder::PSHUFDOp<2, true, false>(OpcodeArgs); +template +void OpDispatchBuilder::PSHUFDOp<2, true, true>(OpcodeArgs); +template +void OpDispatchBuilder::PSHUFDOp<4, false, true>(OpcodeArgs); + +template +void OpDispatchBuilder::SHUFOp(OpcodeArgs) { + LOGMAN_THROW_A(ElementSize != 0, "What. No element size?"); + auto Size = GetSrcSize(Op); + OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + uint8_t Shuffle = Op->Src[1].Data.Literal.Value; + + uint8_t NumElements = Size / ElementSize; + + auto Dest = Src1; + std::array Srcs = { + }; + + for (int i = 0; i < (NumElements >> 1); ++i) { + Srcs[i] = Src1; + } + + for (int i = (NumElements >> 1); i < NumElements; ++i) { + Srcs[i] = Src2; + } + + // 32bit: + // [31:0] = Src1[Selection] + // [63:32] = Src1[Selection] + // [95:64] = Src2[Selection] + // [127:96] = Src2[Selection] + // 64bit: + // [63:0] = Src1[Selection] + // [127:64] = Src2[Selection] + uint8_t SelectionMask = NumElements - 1; + uint8_t ShiftAmount = std::popcount(SelectionMask); + for (uint8_t Element = 0; Element < NumElements; ++Element) { + Dest = _VInsElement(Size, ElementSize, Element, Shuffle & SelectionMask, Dest, Srcs[Element]); + Shuffle >>= ShiftAmount; + } + + StoreResult(FPRClass, Op, Dest, -1); +} + +template +void OpDispatchBuilder::SHUFOp<4>(OpcodeArgs); +template +void OpDispatchBuilder::SHUFOp<8>(OpcodeArgs); + +void OpDispatchBuilder::ANDNOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + // Dest = ~Src1 & Src2 + + Src1 = _VNot(Size, Size, Src1); + auto Dest = _VAnd(Size, Size, Src1, Src2); + + StoreResult(FPRClass, Op, Dest, -1); +} + +template +void OpDispatchBuilder::PINSROp(OpcodeArgs) { + auto Size = GetDstSize(Op); + + OrderedNode *Src{}; + if (Op->Src[0].IsGPR()) { + Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); + } + else { + // If loading from memory then we only load the element size + Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ElementSize, Op->Flags, -1); + } + OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetDstSize(Op), Op->Flags, -1); + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint64_t Index = Op->Src[1].Data.Literal.Value; + + uint8_t NumElements = Size / ElementSize; + Index &= NumElements - 1; + + // This maps 1:1 to an AArch64 NEON Op + auto ALUOp = _VInsGPR(Size, ElementSize, Dest, Src, Index); + StoreResult(FPRClass, Op, ALUOp, -1); +} + +template +void OpDispatchBuilder::PINSROp<1>(OpcodeArgs); +template +void OpDispatchBuilder::PINSROp<2>(OpcodeArgs); +template +void OpDispatchBuilder::PINSROp<4>(OpcodeArgs); +template +void OpDispatchBuilder::PINSROp<8>(OpcodeArgs); + +void OpDispatchBuilder::InsertPSOp(OpcodeArgs) { + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint8_t Imm = Op->Src[1].Data.Literal.Value; + uint8_t CountS = (Imm >> 6); + uint8_t CountD = (Imm >> 4) & 0b11; + uint8_t ZMask = Imm & 0xF; + + OrderedNode *Dest{}; + if (ZMask != 0xF) { + // Only need to load destination if it isn't a full zero + Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetDstSize(Op), Op->Flags, -1); + } + + if (!(ZMask & (1 << CountD))) { + // In the case that ZMask overwrites the destination element, then don't even insert + OrderedNode *Src{}; + if (Op->Src[0].IsGPR()) { + Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + } + else { + // If loading from memory then CountS is forced to zero + CountS = 0; + Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 4, Op->Flags, -1); + } + + Dest = _VInsElement(GetDstSize(Op), 4, CountD, CountS, Dest, Src); + } + + // ZMask happens after insert + if (ZMask == 0xF) { + Dest = _VectorImm(0, 16, 4); + } + else if (ZMask) { + auto Zero = _VectorImm(0, 16, 4); + for (size_t i = 0; i < 4; ++i) { + if (ZMask & (1 << i)) { + Dest = _VInsElement(GetDstSize(Op), 4, i, 0, Dest, Zero); + } + } + } + + StoreResult(FPRClass, Op, Dest, -1); +} + +template +void OpDispatchBuilder::PExtrOp(OpcodeArgs) { + const auto Size = GetSrcSize(Op); + + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint64_t Index = Op->Src[1].Data.Literal.Value; + + const uint8_t NumElements = Size / ElementSize; + Index &= NumElements - 1; + + OrderedNode *Result = _VExtractToGPR(16, ElementSize, Src, Index); + + if (Op->Dest.IsGPR()) { + const uint8_t GPRSize = CTX->GetGPRSize(); + + // If we are storing to a GPR then we zero extend it + if constexpr (ElementSize < 4) { + Result = _Bfe(GPRSize, ElementSize * 8, 0, Result); + } + StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, -1); + } + else { + // If we are storing to memory then we store the size of the element extracted + StoreResult(GPRClass, Op, Result, -1); + } +} + +template +void OpDispatchBuilder::PExtrOp<1>(OpcodeArgs); +template +void OpDispatchBuilder::PExtrOp<2>(OpcodeArgs); +template +void OpDispatchBuilder::PExtrOp<4>(OpcodeArgs); +template +void OpDispatchBuilder::PExtrOp<8>(OpcodeArgs); + +template +void OpDispatchBuilder::PSIGN(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + auto ZeroVec = _VectorZero(Size); + auto NegVec = _VNeg(Size, ElementSize, Dest); + + OrderedNode *CmpLT = _VCMPLTZ(Size, ElementSize, Src); + OrderedNode *CmpEQ = _VCMPEQZ(Size, ElementSize, Src); + OrderedNode *CmpGT = _VCMPGTZ(Size, ElementSize, Src); + + // Negative elements return -dest + CmpLT = _VAnd(Size, ElementSize, CmpLT, NegVec); + + // Zero elements return 0 + CmpEQ = _VAnd(Size, ElementSize, CmpEQ, ZeroVec); + + // Positive elements return dest + CmpGT = _VAnd(Size, ElementSize, CmpGT, Dest); + + // Or our results + OrderedNode *Res = _VOr(Size, ElementSize, CmpGT, _VOr(Size, ElementSize, CmpLT, CmpEQ)); + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::PSIGN<1>(OpcodeArgs); +template +void OpDispatchBuilder::PSIGN<2>(OpcodeArgs); +template +void OpDispatchBuilder::PSIGN<4>(OpcodeArgs); + +template +void OpDispatchBuilder::PSRLDOp(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + auto Size = GetSrcSize(Op); + + OrderedNode *Result{}; + + // Incoming element size for the shift source is always 8 + auto MaxShift = _VectorImm(ElementSize * 8, 8, 8); + Src = _VUMin(8, 8, MaxShift, Src); + Result = _VUShrS(Size, ElementSize, Dest, Src); + + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::PSRLDOp<2>(OpcodeArgs); +template +void OpDispatchBuilder::PSRLDOp<4>(OpcodeArgs); +template +void OpDispatchBuilder::PSRLDOp<8>(OpcodeArgs); + +template +void OpDispatchBuilder::PSRLI(OpcodeArgs) { + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint64_t ShiftConstant = Op->Src[1].Data.Literal.Value; + + auto Size = GetSrcSize(Op); + + auto Shift = _VUShrI(Size, ElementSize, Dest, ShiftConstant); + StoreResult(FPRClass, Op, Shift, -1); +} + +template +void OpDispatchBuilder::PSRLI<2>(OpcodeArgs); +template +void OpDispatchBuilder::PSRLI<4>(OpcodeArgs); +template +void OpDispatchBuilder::PSRLI<8>(OpcodeArgs); + +template +void OpDispatchBuilder::PSLLI(OpcodeArgs) { + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint64_t ShiftConstant = Op->Src[1].Data.Literal.Value; + + auto Size = GetSrcSize(Op); + + auto Shift = _VShlI(Size, ElementSize, Dest, ShiftConstant); + StoreResult(FPRClass, Op, Shift, -1); +} + +template +void OpDispatchBuilder::PSLLI<2>(OpcodeArgs); +template +void OpDispatchBuilder::PSLLI<4>(OpcodeArgs); +template +void OpDispatchBuilder::PSLLI<8>(OpcodeArgs); + +template +void OpDispatchBuilder::PSLL(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + auto Size = GetDstSize(Op); + + OrderedNode *Result{}; + + // Incoming element size for the shift source is always 8 + auto MaxShift = _VectorImm(ElementSize * 8, 8, 8); + Src = _VUMin(8, 8, MaxShift, Src); + Result = _VUShlS(Size, ElementSize, Dest, Src); + + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::PSLL<2>(OpcodeArgs); +template +void OpDispatchBuilder::PSLL<4>(OpcodeArgs); +template +void OpDispatchBuilder::PSLL<8>(OpcodeArgs); + +template +void OpDispatchBuilder::PSRAOp(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + auto Size = GetDstSize(Op); + + OrderedNode *Result{}; + + // Incoming element size for the shift source is always 8 + auto MaxShift = _VectorImm(ElementSize * 8, 8, 8); + Src = _VUMin(8, 8, MaxShift, Src); + Result = _VSShrS(Size, ElementSize, Dest, Src); + + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::PSRAOp<2>(OpcodeArgs); +template +void OpDispatchBuilder::PSRAOp<4>(OpcodeArgs); + +void OpDispatchBuilder::PSRLDQ(OpcodeArgs) { + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint64_t Shift = Op->Src[1].Data.Literal.Value; + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + auto Size = GetDstSize(Op); + + auto Result = _VSRI(Size, 16, Dest, Shift); + StoreResult(FPRClass, Op, Result, -1); +} + +void OpDispatchBuilder::PSLLDQ(OpcodeArgs) { + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint64_t Shift = Op->Src[1].Data.Literal.Value; + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + auto Size = GetDstSize(Op); + + auto Result = _VSLI(Size, 16, Dest, Shift); + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::PSRAIOp(OpcodeArgs) { + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint64_t Shift = Op->Src[1].Data.Literal.Value; + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + auto Size = GetDstSize(Op); + + auto Result = _VSShrI(Size, ElementSize, Dest, Shift); + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::PSRAIOp<2>(OpcodeArgs); +template +void OpDispatchBuilder::PSRAIOp<4>(OpcodeArgs); + +template +void OpDispatchBuilder::PAVGOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + + auto Result = _VURAvg(Size, ElementSize, Dest, Src); + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::PAVGOp<1>(OpcodeArgs); +template +void OpDispatchBuilder::PAVGOp<2>(OpcodeArgs); + +void OpDispatchBuilder::MOVDDUPOp(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Res = _SplatVector2(Src); + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::CVTGPR_To_FPR(OpcodeArgs) { + OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); + + size_t GPRSize = GetSrcSize(Op); + + Src = _Float_FromGPR_S(DstElementSize, GPRSize, Src); + + OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, 16, Op->Flags, -1); + + Src = _VInsScalarElement(16, DstElementSize, 0, Dest, Src); + + StoreResult(FPRClass, Op, Src, -1); +} + +template +void OpDispatchBuilder::CVTGPR_To_FPR<4>(OpcodeArgs); +template +void OpDispatchBuilder::CVTGPR_To_FPR<8>(OpcodeArgs); + +template +void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + // GPR size is determined by REX.W + // Source Element size is determined by instruction + size_t GPRSize = GetDstSize(Op); + + size_t ElementSize = SrcElementSize; + if constexpr (HostRoundingMode) { + Src = _Float_ToGPR_S(Src, ElementSize, GPRSize); + } + else { + Src = _Float_ToGPR_ZS(Src, ElementSize, GPRSize); + } + + StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Src, GPRSize, -1); +} + +template +void OpDispatchBuilder::CVTFPR_To_GPR<4, true>(OpcodeArgs); +template +void OpDispatchBuilder::CVTFPR_To_GPR<4, false>(OpcodeArgs); + +template +void OpDispatchBuilder::CVTFPR_To_GPR<8, true>(OpcodeArgs); +template +void OpDispatchBuilder::CVTFPR_To_GPR<8, false>(OpcodeArgs); + +template +void OpDispatchBuilder::Vector_CVT_Int_To_Float(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + size_t ElementSize = SrcElementSize; + size_t Size = GetDstSize(Op); + if constexpr (Widen) { + Src = _VSXTL(Src, Size, ElementSize); + ElementSize <<= 1; + } + + Src = _Vector_SToF(Src, Size, ElementSize); + + StoreResult(FPRClass, Op, Src, -1); +} + +template +void OpDispatchBuilder::Vector_CVT_Int_To_Float<4, true>(OpcodeArgs); +template +void OpDispatchBuilder::Vector_CVT_Int_To_Float<4, false>(OpcodeArgs); + +template +void OpDispatchBuilder::Vector_CVT_Float_To_Int(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + size_t ElementSize = SrcElementSize; + size_t Size = GetDstSize(Op); + + if constexpr (Narrow) { + Src = _Vector_FToF(Size, SrcElementSize >> 1, SrcElementSize, Src); + ElementSize >>= 1; + } + + if constexpr (HostRoundingMode) { + Src = _Vector_FToS(Src, Size, ElementSize); + } + else { + Src = _Vector_FToZS(Src, Size, ElementSize); + } + + StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, -1); +} + +template +void OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, false>(OpcodeArgs); +template +void OpDispatchBuilder::Vector_CVT_Float_To_Int<4, false, true>(OpcodeArgs); + +template +void OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, true>(OpcodeArgs); +template +void OpDispatchBuilder::Vector_CVT_Float_To_Int<8, true, false>(OpcodeArgs); + +template +void OpDispatchBuilder::Scalar_CVT_Float_To_Float(OpcodeArgs) { + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + Src = _Float_FToF(DstElementSize, SrcElementSize, Src); + Src = _VInsScalarElement(16, DstElementSize, 0, Dest, Src); + + StoreResult(FPRClass, Op, Src, -1); +} + +template +void OpDispatchBuilder::Scalar_CVT_Float_To_Float<4, 8>(OpcodeArgs); +template +void OpDispatchBuilder::Scalar_CVT_Float_To_Float<8, 4>(OpcodeArgs); + +template +void OpDispatchBuilder::Vector_CVT_Float_To_Float(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + size_t Size = GetDstSize(Op); + + if constexpr (DstElementSize > SrcElementSize) { + Src = _Vector_FToF(Size, SrcElementSize << 1, SrcElementSize, Src); + } + else { + Src = _Vector_FToF(Size, SrcElementSize >> 1, SrcElementSize, Src); + } + + StoreResult(FPRClass, Op, Src, -1); +} + +template +void OpDispatchBuilder::Vector_CVT_Float_To_Float<4, 8>(OpcodeArgs); +template +void OpDispatchBuilder::Vector_CVT_Float_To_Float<8, 4>(OpcodeArgs); + +template +void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + size_t ElementSize = SrcElementSize; + size_t DstSize = GetDstSize(Op); + if constexpr (Widen) { + Src = _VSXTL(Src, DstSize, ElementSize); + ElementSize <<= 1; + } + + // Always signed + Src = _Vector_SToF(Src, DstSize, ElementSize); + + OrderedNode *Dest{}; + if constexpr (Widen) { + Dest = Src; + } + else { + Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags, -1); + // Insert the lower bits + Dest = _VInsElement(GetDstSize(Op), 8, 0, 0, Dest, Src); + } + + StoreResult(FPRClass, Op, Dest, -1); +} + +template +void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, false>(OpcodeArgs); +template +void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float<4, true>(OpcodeArgs); + +template +void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + size_t ElementSize = SrcElementSize; + size_t Size = GetDstSize(Op); + + // Always narrows + Src = _Vector_FToF(Size, SrcElementSize >> 1, SrcElementSize, Src); + ElementSize >>= 1; + + if constexpr (HostRoundingMode) { + Src = _Vector_FToS(Src, Size, ElementSize); + } + else { + Src = _Vector_FToZS(Src, Size, ElementSize); + } + + StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, -1); +} + +template +void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, false>(OpcodeArgs); +template +void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<8, true>(OpcodeArgs); + +void OpDispatchBuilder::MASKMOVOp(OpcodeArgs) { + // Until we get correct PHI nodes this is required to be a loop unroll + const auto GPRSize = CTX->GetGPRSize(); + const auto Size = uint32_t{GetSrcSize(Op)} * 8; + + OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1); + + OrderedNode *MemDest = _LoadContext(GPRSize, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RDI]), GPRClass); + + const size_t NumElements = Size / 64; + for (size_t Element = 0; Element < NumElements; ++Element) { + // Extract the current element + auto SrcElement = _VExtractToGPR(GetSrcSize(Op), 8, Src, Element); + auto DestElement = _VExtractToGPR(GetSrcSize(Op), 8, Dest, Element); + + constexpr size_t NumSelectBits = 64 / 8; + for (size_t Select = 0; Select < NumSelectBits; ++Select) { + auto SelectMask = _Bfe(1, 8 * Select + 7, SrcElement); + auto CondJump = _CondJump(SelectMask, {COND_EQ}); + auto StoreBlock = CreateNewCodeBlockAfter(GetCurrentBlock()); + SetFalseJumpTarget(CondJump, StoreBlock); + SetCurrentCodeBlock(StoreBlock); + { + auto DestByte = _Bfe(8, 8 * Select, DestElement); + auto MemLocation = _Add(MemDest, _Constant(Element * 8 + Select)); + _StoreMemAutoTSO(GPRClass, 1, MemLocation, DestByte, 1); + } + auto Jump = _Jump(); + auto NextJumpTarget = CreateNewCodeBlockAfter(StoreBlock); + SetJumpTarget(Jump, NextJumpTarget); + SetTrueJumpTarget(CondJump, NextJumpTarget); + SetCurrentCodeBlock(NextJumpTarget); + } + } +} + +void OpDispatchBuilder::MOVBetweenGPR_FPR(OpcodeArgs) { + if (Op->Dest.IsGPR() && + Op->Dest.Data.GPR.GPR >= FEXCore::X86State::REG_XMM_0) { + OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); + // zext to 128bit + auto Converted = _VCastFromGPR(16, GetSrcSize(Op), Src); + StoreResult(FPRClass, Op, Op->Dest, Converted, -1); + } + else { + // Destination is GPR or mem + // Extract from XMM first + auto ElementSize = GetDstSize(Op); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0],Op->Flags, -1); + + Src = _VExtractToGPR(GetSrcSize(Op), ElementSize, Src, 0); + + StoreResult(GPRClass, Op, Op->Dest, Src, -1); + } +} + +template +void OpDispatchBuilder::VFCMPOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetDstSize(Op), Op->Flags, -1); + OrderedNode *Src2{}; + if constexpr (Scalar) { + Src2 = _VExtractElement(GetDstSize(Op), Size, Dest, 0); + } + else { + Src2 = Dest; + } + uint8_t CompType = Op->Src[1].Data.Literal.Value; + + OrderedNode *Result{}; + // This maps 1:1 to an AArch64 NEON Op + //auto ALUOp = _VCMPGT(Size, ElementSize, Dest, Src); + switch (CompType) { + case 0x00: case 0x08: case 0x10: case 0x18: // EQ + Result = _VFCMPEQ(Size, ElementSize, Src2, Src); + break; + case 0x01: case 0x09: case 0x11: case 0x19: // LT, GT(Swapped operand) + Result = _VFCMPLT(Size, ElementSize, Src2, Src); + break; + case 0x02: case 0x0A: case 0x12: case 0x1A: // LE, GE(Swapped operand) + Result = _VFCMPLE(Size, ElementSize, Src2, Src); + break; + case 0x03: case 0x0B: case 0x13: case 0x1B: // Unordered + Result = _VFCMPUNO(Size, ElementSize, Src2, Src); + break; + case 0x04: case 0x0C: case 0x14: case 0x1C: // NEQ + Result = _VFCMPNEQ(Size, ElementSize, Src2, Src); + break; + case 0x05: case 0x0D: case 0x15: case 0x1D: // NLT, NGT(Swapped operand) + Result = _VFCMPLT(Size, ElementSize, Src2, Src); + Result = _VNot(Size, ElementSize, Result); + break; + case 0x06: case 0x0E: case 0x16: case 0x1E: // NLE, NGE(Swapped operand) + Result = _VFCMPLE(Size, ElementSize, Src2, Src); + Result = _VNot(Size, ElementSize, Result); + break; + case 0x07: case 0x0F: case 0x17: case 0x1F: // Ordered + Result = _VFCMPORD(Size, ElementSize, Src2, Src); + break; + default: LOGMAN_MSG_A("Unknown Comparison type: %d", CompType); + } + + if constexpr (Scalar) { + // Insert the lower bits + Result = _VInsScalarElement(GetDstSize(Op), ElementSize, 0, Dest, Result); + } + + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::VFCMPOp<4, false>(OpcodeArgs); +template +void OpDispatchBuilder::VFCMPOp<4, true>(OpcodeArgs); +template +void OpDispatchBuilder::VFCMPOp<8, false>(OpcodeArgs); +template +void OpDispatchBuilder::VFCMPOp<8, true>(OpcodeArgs); + +void OpDispatchBuilder::FXSaveOp(OpcodeArgs) { + OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false); + Mem = AppendSegmentOffset(Mem, Op->Flags); + + // Saves 512bytes to the memory location provided + // Header changes depending on if REX.W is set or not + if (Op->Flags & X86Tables::DecodeFlags::FLAG_REX_WIDENING) { + // BYTE | 0 1 | 2 3 | 4 | 5 | 6 7 | 8 9 | a b | c d | e f | + // ------------------------------------------ + // 00 | FCW | FSW | FTW | | FOP | FIP | + // 16 | FDP | MXCSR | MXCSR_MASK| + } + else { + // BYTE | 0 1 | 2 3 | 4 | 5 | 6 7 | 8 9 | a b | c d | e f | + // ------------------------------------------ + // 00 | FCW | FSW | FTW | | FOP | FIP[31:0] | FCS | | + // 16 | FDP[31:0] | FDS | | MXCSR | MXCSR_MASK| + } + + { + auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass); + _StoreMem(GPRClass, 2, Mem, FCW, 2); + } + + { + // We must construct the FSW from our various bits + OrderedNode *MemLocation = _Add(Mem, _Constant(2)); + OrderedNode *FSW = _Constant(0); + auto Top = GetX87Top(); + FSW = _Or(FSW, _Lshl(Top, _Constant(11))); + + auto C0 = GetRFLAG(FEXCore::X86State::X87FLAG_C0_LOC); + auto C1 = GetRFLAG(FEXCore::X86State::X87FLAG_C1_LOC); + auto C2 = GetRFLAG(FEXCore::X86State::X87FLAG_C2_LOC); + auto C3 = GetRFLAG(FEXCore::X86State::X87FLAG_C3_LOC); + + FSW = _Or(FSW, _Lshl(C0, _Constant(8))); + FSW = _Or(FSW, _Lshl(C1, _Constant(9))); + FSW = _Or(FSW, _Lshl(C2, _Constant(10))); + FSW = _Or(FSW, _Lshl(C3, _Constant(14))); + _StoreMem(GPRClass, 2, MemLocation, FSW, 2); + } + + // BYTE | 0 1 | 2 3 | 4 | 5 | 6 7 | 8 9 | a b | c d | e f | + // ------------------------------------------ + // 32 | ST0/MM0 | + // 48 | ST1/MM1 | + // 64 | ST2/MM2 | + // 80 | ST3/MM3 | + // 96 | ST4/MM4 | + // 112 | ST5/MM5 | + // 128 | ST6/MM6 | + // 144 | ST7/MM7 | + // 160 | XMM0 + // 173 | XMM1 + // 192 | XMM2 + // 208 | XMM3 + // 224 | XMM4 + // 240 | XMM5 + // 256 | XMM6 + // 272 | XMM7 + // 288 | XMM8 + // 304 | XMM9 + // 320 | XMM10 + // 336 | XMM11 + // 352 | XMM12 + // 368 | XMM13 + // 384 | XMM14 + // 400 | XMM15 + // 416 | + // 432 | + // 448 | + // 464 | Available + // 480 | Available + // 496 | Available + // FCW: x87 FPU control word + // FSW: x87 FPU status word + // FTW: x87 FPU Tag word (Abridged) + // FOP: x87 FPU opcode. Lower 11 bits of the opcode + // FIP: x87 FPU instructyion pointer offset + // FCS: x87 FPU instruction pointer selector. If CPUID_0000_0007_0000_00000:EBX[bit 13] = 1 then this is deprecated and stores as 0 + // FDP: x87 FPU instruction operand (data) pointer offset + // FDS: x87 FPU instruction operand (data) pointer selector. Same deprecation as FCS + // MXCSR: If OSFXSR bit in CR4 is not set then this may not be saved + // MXCSR_MASK: Mask for writes to the MXCSR register + // If OSFXSR bit in CR4 is not set than FXSAVE /may/ not save the XMM registers + // This is implementation dependent + for (unsigned i = 0; i < 8; ++i) { + OrderedNode *MMReg = _LoadContext(16, offsetof(FEXCore::Core::CPUState, mm[i]), FPRClass); + OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32)); + + _StoreMem(FPRClass, 16, MemLocation, MMReg, 16); + } + for (unsigned i = 0; i < 16; ++i) { + OrderedNode *XMMReg = _LoadContext(16, offsetof(FEXCore::Core::CPUState, xmm[i]), FPRClass); + OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160)); + + _StoreMem(FPRClass, 16, MemLocation, XMMReg, 16); + } +} + +void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) { + OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false); + Mem = AppendSegmentOffset(Mem, Op->Flags); + + auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2); + _F80LoadFCW(NewFCW); + _StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW); + + { + OrderedNode *MemLocation = _Add(Mem, _Constant(2)); + auto NewFSW = _LoadMem(GPRClass, 2, MemLocation, 2); + + // Strip out the FSW information + auto Top = _Bfe(3, 11, NewFSW); + SetX87Top(Top); + + auto C0 = _Bfe(1, 8, NewFSW); + auto C1 = _Bfe(1, 9, NewFSW); + auto C2 = _Bfe(1, 10, NewFSW); + auto C3 = _Bfe(1, 14, NewFSW); + + SetRFLAG(C0); + SetRFLAG(C1); + SetRFLAG(C2); + SetRFLAG(C3); + } + + for (unsigned i = 0; i < 8; ++i) { + OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 32)); + auto MMReg = _LoadMem(FPRClass, 16, MemLocation, 16); + _StoreContext(FPRClass, 16, offsetof(FEXCore::Core::CPUState, mm[i]), MMReg); + } + for (unsigned i = 0; i < 16; ++i) { + OrderedNode *MemLocation = _Add(Mem, _Constant(i * 16 + 160)); + auto XMMReg = _LoadMem(FPRClass, 16, MemLocation, 16); + _StoreContext(FPRClass, 16, offsetof(FEXCore::Core::CPUState, xmm[i]), XMMReg); + } +} + +void OpDispatchBuilder::PAlignrOp(OpcodeArgs) { + OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + auto Size = GetDstSize(Op); + + uint8_t Index = Op->Src[1].Data.Literal.Value; + OrderedNode *Res{}; + if (Index >= (Size * 2)) { + // If the immediate is greater than both vectors combined then it zeroes the vector + Res = _VectorZero(Size); + } + else { + Res = _VExtr(Size, 1, Src1, Src2, Index); + } + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::UCOMISxOp(OpcodeArgs) { + OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + OrderedNode *Res = _FCmp(Src1, Src2, ElementSize, + (1 << FCMP_FLAG_EQ) | + (1 << FCMP_FLAG_LT) | + (1 << FCMP_FLAG_UNORDERED)); + + OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT); + OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ); + OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED); + + SetRFLAG(HostFlag_CF); + SetRFLAG(HostFlag_ZF); + SetRFLAG(HostFlag_Unordered); + + auto ZeroConst = _Constant(0); + SetRFLAG(ZeroConst); + SetRFLAG(ZeroConst); + SetRFLAG(ZeroConst); + + flagsOp = FLAGS_OP_FCMP; + flagsOpDest = Src1; + flagsOpSrc = Src2; + flagsOpSize = GetSrcSize(Op); +} + +template +void OpDispatchBuilder::UCOMISxOp<4>(OpcodeArgs); +template +void OpDispatchBuilder::UCOMISxOp<8>(OpcodeArgs); + +void OpDispatchBuilder::LDMXCSR(OpcodeArgs) { + OrderedNode *Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1); + // We only support the rounding mode being set + OrderedNode *RoundingMode = _Bfe(4, 3, 13, Dest); + _SetRoundingMode(RoundingMode); +} + +void OpDispatchBuilder::STMXCSR(OpcodeArgs) { + // Default MXCSR + OrderedNode *MXCSR = _Constant(32, 0x1F80); + OrderedNode *RoundingMode = _GetRoundingMode(); + MXCSR = _Bfi(4, 3, 13, MXCSR, RoundingMode); + + StoreResult(GPRClass, Op, MXCSR, -1); +} + +template +void OpDispatchBuilder::PACKUSOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *Res = _VSQXTUN(Size, ElementSize, Dest); + Res = _VSQXTUN2(Size, ElementSize, Res, Src); + + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::PACKUSOp<2>(OpcodeArgs); +template +void OpDispatchBuilder::PACKUSOp<4>(OpcodeArgs); + +template +void OpDispatchBuilder::PACKSSOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *Res = _VSQXTN(Size, ElementSize, Dest); + Res = _VSQXTN2(Size, ElementSize, Res, Src); + + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::PACKSSOp<2>(OpcodeArgs); +template +void OpDispatchBuilder::PACKSSOp<4>(OpcodeArgs); + +template +void OpDispatchBuilder::PMULLOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *Res{}; + + if (Size == 8) { + if constexpr (Signed) { + Res = _VSMull(16, ElementSize, Src1, Src2); + } + else { + Res = _VUMull(16, ElementSize, Src1, Src2); + } + } + else { + OrderedNode* Srcs1[2]{}; + OrderedNode* Srcs2[2]{}; + + Srcs1[0] = _VExtr(Size, ElementSize, Src1, Src1, 0); + Srcs1[1] = _VExtr(Size, ElementSize, Src1, Src1, 2); + + Srcs2[0] = _VExtr(Size, ElementSize, Src2, Src2, 0); + Srcs2[1] = _VExtr(Size, ElementSize, Src2, Src2, 2); + + Src1 = _VInsElement(Size, ElementSize, 1, 0, Srcs1[0], Srcs1[1]); + Src2 = _VInsElement(Size, ElementSize, 1, 0, Srcs2[0], Srcs2[1]); + + if constexpr (Signed) { + Res = _VSMull(Size, ElementSize, Src1, Src2); + } + else { + Res = _VUMull(Size, ElementSize, Src1, Src2); + } + } + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::PMULLOp<4, false>(OpcodeArgs); +template +void OpDispatchBuilder::PMULLOp<4, true>(OpcodeArgs); + +template +void OpDispatchBuilder::MOVQ2DQ(OpcodeArgs) { + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + // This instruction is a bit special in that if the source is MMX then it zexts to 128bit + if constexpr (ToXMM) { + Src = _VMov(Src, 16); + _StoreContext(FPRClass, 16, offsetof(FEXCore::Core::CPUState, xmm[Op->Dest.Data.GPR.GPR - FEXCore::X86State::REG_XMM_0][0]), Src); + } + else { + // This is simple, just store the result + StoreResult(FPRClass, Op, Src, -1); + } +} + +template +void OpDispatchBuilder::MOVQ2DQ(OpcodeArgs); +template +void OpDispatchBuilder::MOVQ2DQ(OpcodeArgs); + +template +void OpDispatchBuilder::PADDSOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *Res{}; + if constexpr (Signed) { + Res = _VSQAdd(Size, ElementSize, Dest, Src); + } + else { + Res = _VUQAdd(Size, ElementSize, Dest, Src); + } + + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::PADDSOp<1, false>(OpcodeArgs); +template +void OpDispatchBuilder::PADDSOp<1, true>(OpcodeArgs); +template +void OpDispatchBuilder::PADDSOp<2, false>(OpcodeArgs); +template +void OpDispatchBuilder::PADDSOp<2, true>(OpcodeArgs); + +template +void OpDispatchBuilder::PSUBSOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *Res{}; + if constexpr (Signed) { + Res = _VSQSub(Size, ElementSize, Dest, Src); + } + else { + Res = _VUQSub(Size, ElementSize, Dest, Src); + } + + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::PSUBSOp<1, false>(OpcodeArgs); +template +void OpDispatchBuilder::PSUBSOp<1, true>(OpcodeArgs); +template +void OpDispatchBuilder::PSUBSOp<2, false>(OpcodeArgs); +template +void OpDispatchBuilder::PSUBSOp<2, true>(OpcodeArgs); + +template +void OpDispatchBuilder::ADDSUBPOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *ResAdd{}; + OrderedNode *ResSub{}; + ResAdd = _VFAdd(Size, ElementSize, Dest, Src); + ResSub = _VFSub(Size, ElementSize, Dest, Src); + + // We now need to swizzle results + uint8_t NumElements = Size / ElementSize; + // Even elements are the sub result + // Odd elements are the add results + for (size_t i = 0; i < NumElements; i += 2) { + ResAdd = _VInsElement(Size, ElementSize, i, i, ResAdd, ResSub); + } + StoreResult(FPRClass, Op, ResAdd, -1); +} + +template +void OpDispatchBuilder::ADDSUBPOp<4>(OpcodeArgs); +template +void OpDispatchBuilder::ADDSUBPOp<8>(OpcodeArgs); + +void OpDispatchBuilder::PMADDWD(OpcodeArgs) { + // This is a pretty curious operation + // Does two MADD operations across 4 16bit signed integers and accumulates to 32bit integers in the destination + // + // x86 PMADDWD: xmm1, xmm2 + // xmm1[31:0] = (xmm1[15:0] * xmm2[15:0]) + (xmm1[31:16] * xmm2[31:16]) + // xmm1[63:32] = (xmm1[47:32] * xmm2[47:32]) + (xmm1[63:48] * xmm2[63:48]) + // etc.. for larger registers + + auto Size = GetSrcSize(Op); + + OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + if (Size == 8) { + Size <<= 1; + Src1 = _VBitcast(Size, 2, Src1); + Src2 = _VBitcast(Size, 2, Src2); + } + + auto Src1_L = _VSXTL(Size, 2, Src1); // [15:0 ], [31:16], [32:47 ], [63:48 ] + auto Src1_H = _VSXTL2(Size, 2, Src1); // [79:64], [95:80], [111:96], [127:112] + + auto Src2_L = _VSXTL(Size, 2, Src2); // [15:0 ], [31:16], [32:47 ], [63:48 ] + auto Src2_H = _VSXTL2(Size, 2, Src2); // [79:64], [95:80], [111:96], [127:112] + + auto Res_L = _VSMul(Size, 4, Src1_L, Src2_L); // [15:0 ], [31:16], [32:47 ], [63:48 ] : Original elements + auto Res_H = _VSMul(Size, 4, Src1_H, Src2_H); // [79:64], [95:80], [111:96], [127:112] : Original elements + + // [15:0 ] + [31:16], [32:47 ] + [63:48 ], [79:64] + [95:80], [111:96] + [127:112] + auto Res = _VAddP(Size, 4, Res_L, Res_H); + StoreResult(FPRClass, Op, Res, -1); +} + +void OpDispatchBuilder::PMADDUBSW(OpcodeArgs) { + // This is a pretty curious operation + // Does four MADD operations across 8 8bit signed and unsigned integers and accumulates to 16bit integers in the destination WITH saturation + // + // x86 PMADDUBSW: mm1, mm2 + // mm1[15:0] = SaturateSigned16(((s8)mm2[15:8] * (u8)mm1[15:8]) + ((s8)mm2[7:0] * (u8)mm1[7:0])) + // mm1[31:16] = SaturateSigned16(((s8)mm2[31:24] * (u8)mm1[31:24]) + ((s8)mm2[23:16] * (u8)mm1[23:16])) + // mm1[47:32] = SaturateSigned16(((s8)mm2[47:40] * (u8)mm1[47:40]) + ((s8)mm2[39:32] * (u8)mm1[39:32])) + // mm1[63:48] = SaturateSigned16(((s8)mm2[63:56] * (u8)mm1[63:56]) + ((s8)mm2[55:48] * (u8)mm1[55:48])) + // Extends to larger registers + auto Size = GetSrcSize(Op); + + OrderedNode *Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + if (Size == 8) { + // 64bit is more efficient + + // Src1 is unsigned + auto Src1_16b = _VUXTL(Size * 2, 1, Src1); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] + + // Src2 is signed + auto Src2_16b = _VSXTL(Size * 2, 1, Src2); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] + + auto ResMul_L = _VSMull(Size * 2, 2, Src1_16b, Src2_16b); + auto ResMul_H = _VSMull2(Size * 2, 2, Src1_16b, Src2_16b); + + // Now add pairwise across the vector + auto ResAdd = _VAddP(Size * 2, 4, ResMul_L, ResMul_H); + + // Add saturate back down to 16bit + OrderedNode *Res = _VSQXTN(Size * 2, 4, ResAdd); + StoreResult(FPRClass, Op, Res, -1); + } + else { + // Src1 is unsigned + auto Src1_16b_L = _VUXTL(Size, 1, Src1); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] + auto Src1_16b_H = _VUXTL2(Size, 1, Src1); // Offset to +64bits [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] + + // Src2 is signed + auto Src2_16b_L = _VSXTL(Size, 1, Src2); // [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] + auto Src2_16b_H = _VSXTL2(Size, 1, Src2); // Offset to +64bits [7:0 ], [15:8], [23:16], [31:24], [39:32], [47:40], [55:48], [63:56] + + auto ResMul_L = _VSMull(Size, 2, Src1_16b_L, Src2_16b_L); + auto ResMul_L_H = _VSMull2(Size, 2, Src1_16b_L, Src2_16b_L); + + auto ResMul_H = _VSMull(Size, 2, Src1_16b_H, Src2_16b_H); + auto ResMul_H_H = _VSMull2(Size, 2, Src1_16b_H, Src2_16b_H); + + // Now add pairwise across the vector + auto ResAdd_L = _VAddP(Size, 4, ResMul_L, ResMul_L_H); + auto ResAdd_H = _VAddP(Size, 4, ResMul_H, ResMul_H_H); + + // Add saturate back down to 16bit + OrderedNode *Res = _VSQXTN(Size, 4, ResAdd_L); + Res = _VSQXTN2(Size, 4, Res, ResAdd_H); + + StoreResult(FPRClass, Op, Res, -1); + } +} + +template +void OpDispatchBuilder::PMULHW(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *Res{}; + if (Size == 8) { + Dest = _VBitcast(Size * 2, 2, Dest); + Src = _VBitcast(Size * 2, 2, Src); + + // Implementation is more efficient for 8byte registers + if (Signed) + Res = _VSMull(Size * 2, 2, Dest, Src); + else + Res = _VUMull(Size * 2, 2, Dest, Src); + + Res = _VUShrNI(Size * 2, 4, Res, 16); + } + else { + // 128bit is less efficient + OrderedNode *ResultLow; + OrderedNode *ResultHigh; + if (Signed) { + ResultLow = _VSMull(Size, 2, Dest, Src); + ResultHigh = _VSMull2(Size, 2, Dest, Src); + } + else { + ResultLow = _VUMull(Size, 2, Dest, Src); + ResultHigh = _VUMull2(Size, 2, Dest, Src); + } + + // Combine the results + Res = _VUShrNI(Size, 4, ResultLow, 16); + Res = _VUShrNI2(Size, 4, Res, ResultHigh, 16); + } + + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::PMULHW(OpcodeArgs); +template +void OpDispatchBuilder::PMULHW(OpcodeArgs); + +void OpDispatchBuilder::PMULHRSW(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *Res{}; + if (Size == 8) { + // Implementation is more efficient for 8byte registers + Res = _VSMull(Size * 2, 2, Dest, Src); + Res = _VSShrI(Size * 2, 4, Res, 14); + auto OneVector = _VectorImm(1, Size * 2, 4); + Res = _VAdd(Size * 2, 4, Res, OneVector); + Res = _VUShrNI(Size * 2, 4, Res, 1); + } + else { + // 128bit is less efficient + OrderedNode *ResultLow; + OrderedNode *ResultHigh; + + ResultLow = _VSMull(Size, 2, Dest, Src); + ResultHigh = _VSMull2(Size, 2, Dest, Src); + + ResultLow = _VSShrI(Size, 4, ResultLow, 14); + ResultHigh = _VSShrI(Size, 4, ResultHigh, 14); + auto OneVector = _VectorImm(1, Size, 4); + + ResultLow = _VAdd(Size, 4, ResultLow, OneVector); + ResultHigh = _VAdd(Size, 4, ResultHigh, OneVector); + + // Combine the results + Res = _VUShrNI(Size, 4, ResultLow, 1); + Res = _VUShrNI2(Size, 4, Res, ResultHigh, 1); + } + + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::HADDP(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *Res = _VFAddP(Size, ElementSize, Dest, Src); + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::HADDP<4>(OpcodeArgs); +template +void OpDispatchBuilder::HADDP<8>(OpcodeArgs); + +template +void OpDispatchBuilder::HSUBP(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + // This is a bit complicated since AArch64 doesn't support a pairwise subtract + auto Dest_Neg = _VFNeg(Size, ElementSize, Dest); + auto Src_Neg = _VFNeg(Size, ElementSize, Src); + + // Now we need to swizzle the values + OrderedNode *Swizzle_Dest = Dest; + OrderedNode *Swizzle_Src = Src; + + if constexpr (ElementSize == 4) { + Swizzle_Dest = _VInsElement(Size, ElementSize, 1, 1, Swizzle_Dest, Dest_Neg); + Swizzle_Dest = _VInsElement(Size, ElementSize, 3, 3, Swizzle_Dest, Dest_Neg); + + Swizzle_Src = _VInsElement(Size, ElementSize, 1, 1, Swizzle_Src, Src_Neg); + Swizzle_Src = _VInsElement(Size, ElementSize, 3, 3, Swizzle_Src, Src_Neg); + } + else { + Swizzle_Dest = _VInsElement(Size, ElementSize, 1, 1, Swizzle_Dest, Dest_Neg); + Swizzle_Src = _VInsElement(Size, ElementSize, 1, 1, Swizzle_Src, Src_Neg); + } + + OrderedNode *Res = _VFAddP(Size, ElementSize, Swizzle_Dest, Swizzle_Src); + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::HSUBP<4>(OpcodeArgs); +template +void OpDispatchBuilder::HSUBP<8>(OpcodeArgs); + +template +void OpDispatchBuilder::PHADD(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *Res = _VAddP(Size, ElementSize, Dest, Src); + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::PHADD<2>(OpcodeArgs); +template +void OpDispatchBuilder::PHADD<4>(OpcodeArgs); + +template +void OpDispatchBuilder::PHSUB(OpcodeArgs) { + auto Size = GetSrcSize(Op); + uint8_t NumElements = Size / ElementSize; + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + // This is a bit complicated since AArch64 doesn't support a pairwise subtract + auto Dest_Neg = _VNeg(Size, ElementSize, Dest); + auto Src_Neg = _VNeg(Size, ElementSize, Src); + + // Now we need to swizzle the values + OrderedNode *Swizzle_Dest = Dest; + OrderedNode *Swizzle_Src = Src; + + // Odd elements turn in to negated elements + for (size_t i = 1; i < NumElements; i += 2) { + Swizzle_Dest = _VInsElement(Size, ElementSize, i, i, Swizzle_Dest, Dest_Neg); + Swizzle_Src = _VInsElement(Size, ElementSize, i, i, Swizzle_Src, Src_Neg); + } + + OrderedNode *Res = _VAddP(Size, ElementSize, Swizzle_Dest, Swizzle_Src); + StoreResult(FPRClass, Op, Res, -1); +} + +template +void OpDispatchBuilder::PHSUB<2>(OpcodeArgs); +template +void OpDispatchBuilder::PHSUB<4>(OpcodeArgs); + +void OpDispatchBuilder::PHADDS(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + if (Size == 8) { + // Implementation is more efficient for 8byte registers + auto Dest_Larger = _VSXTL(Size * 2, 2, Dest); + auto Src_Larger = _VSXTL(Size * 2, 2, Src); + + OrderedNode *AddRes = _VAddP(Size * 2, 4, Dest_Larger, Src_Larger); + + // Saturate back down to the result + OrderedNode *Res = _VSQXTN(Size * 2, 4, AddRes); + StoreResult(FPRClass, Op, Res, -1); + } + else { + auto Dest_Larger = _VSXTL(Size, 2, Dest); + auto Dest_Larger_H = _VSXTL2(Size, 2, Dest); + + auto Src_Larger = _VSXTL(Size, 2, Src); + auto Src_Larger_H = _VSXTL2(Size, 2, Src); + + OrderedNode *AddRes_L = _VAddP(Size, 4, Dest_Larger, Dest_Larger_H); + OrderedNode *AddRes_H = _VAddP(Size, 4, Src_Larger, Src_Larger_H); + + // Saturate back down to the result + OrderedNode *Res = _VSQXTN(Size, 4, AddRes_L); + Res = _VSQXTN2(Size, 4, Res, AddRes_H); + + StoreResult(FPRClass, Op, Res, -1); + } +} + +void OpDispatchBuilder::PHSUBS(OpcodeArgs) { + auto Size = GetSrcSize(Op); + uint8_t ElementSize = 2; + uint8_t NumElements = Size / ElementSize; + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + // This is a bit complicated since AArch64 doesn't support a pairwise subtract + auto Dest_Neg = _VNeg(Size, ElementSize, Dest); + auto Src_Neg = _VNeg(Size, ElementSize, Src); + + // Now we need to swizzle the values + OrderedNode *Swizzle_Dest = Dest; + OrderedNode *Swizzle_Src = Src; + + // Odd elements turn in to negated elements + for (size_t i = 1; i < NumElements; i += 2) { + Swizzle_Dest = _VInsElement(Size, ElementSize, i, i, Swizzle_Dest, Dest_Neg); + Swizzle_Src = _VInsElement(Size, ElementSize, i, i, Swizzle_Src, Src_Neg); + } + + Dest = Swizzle_Dest; + Src = Swizzle_Src; + + if (Size == 8) { + // Implementation is more efficient for 8byte registers + auto Dest_Larger = _VSXTL(Size * 2, 2, Dest); + auto Src_Larger = _VSXTL(Size * 2, 2, Src); + + OrderedNode *AddRes = _VAddP(Size * 2, 4, Dest_Larger, Src_Larger); + + // Saturate back down to the result + OrderedNode *Res = _VSQXTN(Size * 2, 4, AddRes); + StoreResult(FPRClass, Op, Res, -1); + } + else { + auto Dest_Larger = _VSXTL(Size, 2, Dest); + auto Dest_Larger_H = _VSXTL2(Size, 2, Dest); + + auto Src_Larger = _VSXTL(Size, 2, Src); + auto Src_Larger_H = _VSXTL2(Size, 2, Src); + + OrderedNode *AddRes_L = _VAddP(Size, 4, Dest_Larger, Dest_Larger_H); + OrderedNode *AddRes_H = _VAddP(Size, 4, Src_Larger, Src_Larger_H); + + // Saturate back down to the result + OrderedNode *Res = _VSQXTN(Size, 4, AddRes_L); + Res = _VSQXTN2(Size, 4, Res, AddRes_H); + + StoreResult(FPRClass, Op, Res, -1); + } +} + +void OpDispatchBuilder::PSADBW(OpcodeArgs) { + // The documentation is actually incorrect in how this instruction operates + // It strongly implies that the `abs(dest[i] - src[i])` operates in 8bit space + // but it actually operates in more than 8bit space + // This can be seen with `abs(0 - 0xFF)` returning a different result depending + // on bit length + auto Size = GetSrcSize(Op); + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *Result{}; + + if (Size == 8) { + Dest = _VUXTL(Size*2, 1, Dest); + Src = _VUXTL(Size*2, 1, Src); + + OrderedNode *SubResult = _VSub(Size*2, 2, Dest, Src); + OrderedNode *AbsResult = _VAbs(Size*2, 2, SubResult); + + // Now vector-wide add the results for each + Result = _VAddV(Size * 2, 2, AbsResult); + } + else { + OrderedNode *Dest_Low = _VUXTL(Size, 1, Dest); + OrderedNode *Dest_High = _VUXTL2(Size, 1, Dest); + + OrderedNode *Src_Low = _VUXTL(Size, 1, Src); + OrderedNode *Src_High = _VUXTL2(Size, 1, Src); + + OrderedNode *SubResult_Low = _VSub(Size, 2, Dest_Low, Src_Low); + OrderedNode *SubResult_High = _VSub(Size, 2, Dest_High, Src_High); + + OrderedNode *AbsResult_Low = _VAbs(Size, 2, SubResult_Low); + OrderedNode *AbsResult_High = _VAbs(Size, 2, SubResult_High); + + // Now vector pairwise add all four of these + OrderedNode * Result_Low = _VAddV(Size, 2, AbsResult_Low); + OrderedNode * Result_High = _VAddV(Size, 2, AbsResult_High); + + Result = _VInsElement(Size, 8, 1, 0, Result_Low, Result_High); + } + + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::ExtendVectorElements(OpcodeArgs) { + auto Size = GetDstSize(Op); + + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *Result {Src}; + + for (size_t CurrentElementSize = ElementSize; + CurrentElementSize != DstElementSize; + CurrentElementSize <<= 1) { + if constexpr (Signed) { + Result = _VSXTL(Result, Size, CurrentElementSize); + } + else { + Result = _VUXTL(Result, Size, CurrentElementSize); + } + } + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::ExtendVectorElements<1, 2, false>(OpcodeArgs); +template +void OpDispatchBuilder::ExtendVectorElements<1, 4, false>(OpcodeArgs); +template +void OpDispatchBuilder::ExtendVectorElements<1, 8, false>(OpcodeArgs); +template +void OpDispatchBuilder::ExtendVectorElements<2, 4, false>(OpcodeArgs); +template +void OpDispatchBuilder::ExtendVectorElements<2, 8, false>(OpcodeArgs); +template +void OpDispatchBuilder::ExtendVectorElements<4, 8, false>(OpcodeArgs); + +template +void OpDispatchBuilder::ExtendVectorElements<1, 2, true>(OpcodeArgs); +template +void OpDispatchBuilder::ExtendVectorElements<1, 4, true>(OpcodeArgs); +template +void OpDispatchBuilder::ExtendVectorElements<1, 8, true>(OpcodeArgs); +template +void OpDispatchBuilder::ExtendVectorElements<2, 4, true>(OpcodeArgs); +template +void OpDispatchBuilder::ExtendVectorElements<2, 8, true>(OpcodeArgs); +template +void OpDispatchBuilder::ExtendVectorElements<4, 8, true>(OpcodeArgs); + +template +void OpDispatchBuilder::VectorRound(OpcodeArgs) { + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint64_t Mode = Op->Src[1].Data.Literal.Value; + uint64_t RoundControlSource = (Mode >> 2) & 1; + uint64_t RoundControl = Mode & 0b11; + + auto Size = GetSrcSize(Op); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + if (RoundControlSource) { + RoundControl = 0; // MXCSR + } + + std::array SourceModes = { + FEXCore::IR::Round_Nearest, + FEXCore::IR::Round_Negative_Infinity, + FEXCore::IR::Round_Positive_Infinity, + FEXCore::IR::Round_Towards_Zero, + FEXCore::IR::Round_Host, + }; + + Src = _Vector_FToI(Src, SourceModes[(RoundControlSource << 2) | RoundControl], Size, ElementSize); + + if constexpr (Scalar) { + // Insert the lower bits + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + auto Result = _VInsScalarElement(GetDstSize(Op), ElementSize, 0, Dest, Src); + StoreResult(FPRClass, Op, Result, -1); + } + else { + StoreResult(FPRClass, Op, Src, -1); + } +} + +template +void OpDispatchBuilder::VectorRound<4, false>(OpcodeArgs); +template +void OpDispatchBuilder::VectorRound<8, false>(OpcodeArgs); + +template +void OpDispatchBuilder::VectorRound<4, true>(OpcodeArgs); +template +void OpDispatchBuilder::VectorRound<8, true>(OpcodeArgs); + +template +void OpDispatchBuilder::VectorBlend(OpcodeArgs) { + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint8_t Select = Op->Src[1].Data.Literal.Value; + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + for (size_t i = 0; i < (16 / ElementSize); ++i) { + if (Select & (1 << i)) { + // This could be optimized if it becomes costly + Dest = _VInsElement(16, ElementSize, i, i, Dest, Src); + } + } + StoreResult(FPRClass, Op, Dest, -1); +} + +template +void OpDispatchBuilder::VectorBlend<2>(OpcodeArgs); +template +void OpDispatchBuilder::VectorBlend<4>(OpcodeArgs); +template +void OpDispatchBuilder::VectorBlend<8>(OpcodeArgs); + +template +void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + // The mask is hardcoded to be xmm0 in this instruction + OrderedNode *Mask = _LoadContext(16, offsetof(FEXCore::Core::CPUState, xmm[0]), FPRClass); + // Each element is selected by the high bit of that element size + // Dest[ElementIdx] = Xmm0[ElementIndex][HighBit] ? Src : Dest; + // + // To emulate this on AArch64 + // Arithmetic shift right by the element size, then use BSL to select the registers + Mask = _VSShrI(Size, ElementSize, Mask, (ElementSize * 8) - 1); + auto Result = _VBSL(Mask, Src, Dest); + + StoreResult(FPRClass, Op, Result, -1); +} +template +void OpDispatchBuilder::VectorVariableBlend<1>(OpcodeArgs); +template +void OpDispatchBuilder::VectorVariableBlend<4>(OpcodeArgs); +template +void OpDispatchBuilder::VectorVariableBlend<8>(OpcodeArgs); + +void OpDispatchBuilder::PTestOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + OrderedNode *Test1 = _VAnd(Dest, Src, Size, 1); + OrderedNode *Test2 = _VBic(Src, Dest, Size, 1); + + Test1 = _VPopcount(Size, 1, Test1); + Test2 = _VPopcount(Size, 1, Test2); + + // Element size doesn't matter here + // x86-64 doesn't support a horizontal byte add though + Test1 = _VAddV(Size, 2, Test1); + Test2 = _VAddV(Size, 2, Test2); + + Test1 = _VExtractToGPR(16, 2, Test1, 0); + Test2 = _VExtractToGPR(16, 2, Test2, 0); + + auto ZeroConst = _Constant(0); + auto OneConst = _Constant(1); + + Test1 = _Select(FEXCore::IR::COND_EQ, + Test1, ZeroConst, OneConst, ZeroConst); + + Test2 = _Select(FEXCore::IR::COND_EQ, + Test2, ZeroConst, OneConst, ZeroConst); + + SetRFLAG(Test1); + SetRFLAG(Test2); +} + +void OpDispatchBuilder::PHMINPOSUWOp(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + auto Min = _VUMinV(Size, 2, Src); + + std::array Indexes { + _Constant(0), + _Constant(1), + _Constant(2), + _Constant(3), + _Constant(4), + _Constant(5), + _Constant(6), + _Constant(7), + }; + + auto Pos = Indexes[7]; + auto MinGPR = _VExtractToGPR(16, 2, Min, 0); + + // Calculate position + // This doesn't match with ARM behaviour at all + // Instruction returns the minimum matching index + for (size_t i = 8; i > 0; --i) { + auto Element = _VExtractToGPR(16, 2, Src, i - 1); + Pos = _Select(FEXCore::IR::COND_EQ, + Element, MinGPR, Indexes[i - 1], Pos); + } + + // Insert the minimum in to bits [15:0] + OrderedNode *Result = _VMov(Min, 2); + + // Insert position in to bits [18:16] + Result = _VInsGPR(16, 2, Result, Pos, 1); + + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::DPPOp(OpcodeArgs) { + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint8_t Mask = Op->Src[1].Data.Literal.Value; + uint8_t SrcMask = Mask >> 4; + uint8_t DstMask = Mask & 0xF; + + OrderedNode *ZeroVec = _VectorZero(16); + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + // First step is to do an FMUL + OrderedNode *Temp = _VFMul(16, ElementSize, Dest, Src); + + // Now we zero out elements based on src mask + for (size_t i = 0; i < (16 / ElementSize); ++i) { + if ((SrcMask & (1 << i)) == 0) { + Temp = _VInsElement(16, ElementSize, i, 0, Temp, ZeroVec); + } + } + + // Now we need to do a horizontal add of the elements + // We only have pairwise float add so this needs to be done in steps + Temp = _VFAddP(16, ElementSize, Temp, ZeroVec); + + if constexpr (ElementSize == 4) { + // For 32-bit float we need one more step to add all four results together + Temp = _VFAddP(16, ElementSize, Temp, ZeroVec); + } + + // Now using the destination mask we choose where the result ends up + // It can duplicate and zero results + auto Result = ZeroVec; + + for (size_t i = 0; i < (16 / ElementSize); ++i) { + if (DstMask & (1 << i)) { + Result = _VInsElement(16, ElementSize, i, 0, Result, Temp); + } + } + + StoreResult(FPRClass, Op, Result, -1); +} + +template +void OpDispatchBuilder::DPPOp<4>(OpcodeArgs); +template +void OpDispatchBuilder::DPPOp<8>(OpcodeArgs); + +void OpDispatchBuilder::MPSADBWOp(OpcodeArgs) { + LOGMAN_THROW_A(Op->Src[1].IsLiteral(), "Src1 needs to be literal here"); + uint8_t Select = Op->Src[1].Data.Literal.Value; + + // Src1 needs to be in byte offset + uint8_t Select_Dest = ((Select & 0b100) >> 2) * 32 / 8; + uint8_t Select_Src2 = Select & 0b11; + + OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); + OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + + // Src2 will grab a 32bit element and duplicate it across the 128bits + OrderedNode *DupSrc = _VDupElement(16, 4, Src, Select_Src2); + + // Src1/Dest needs a bunch of magic + + // Shift right by selected bytes + // This will give us Dest[15:0], and Dest[79:64] + OrderedNode *Dest1 = _VExtr(16, 1, Dest, Dest, Select_Dest + 0); + // This will give us Dest[31:16], and Dest[95:80] + OrderedNode *Dest2 = _VExtr(16, 1, Dest, Dest, Select_Dest + 1); + // This will give us Dest[47:32], and Dest[111:96] + OrderedNode *Dest3 = _VExtr(16, 1, Dest, Dest, Select_Dest + 2); + // This will give us Dest[63:48], and Dest[127:112] + OrderedNode *Dest4 = _VExtr(16, 1, Dest, Dest, Select_Dest + 3); + + // For each shifted section, we now have two 32-bit values per vector that can be used + // Dest1.S[0] and Dest1.S[1] = Bytes - 0,1,2,3:4,5,6,7 + // Dest2.S[0] and Dest2.S[1] = Bytes - 1,2,3,4:5,6,7,8 + // Dest3.S[0] and Dest3.S[1] = Bytes - 2,3,4,5:6,7,8,9 + // Dest4.S[0] and Dest4.S[1] = Bytes - 3,4,5,6:7,8,9,10 + Dest1 = _VUABDL(16, 1, Dest1, DupSrc); + Dest2 = _VUABDL(16, 1, Dest2, DupSrc); + Dest3 = _VUABDL(16, 1, Dest3, DupSrc); + Dest4 = _VUABDL(16, 1, Dest4, DupSrc); + + // Dest[1,2,3,4] Now contains the data prior to combining + // Temp[0,1,2,3] for each step + + // Each destination now has 16bit x 8 elements in it that were the absolute difference for each byte + // Needs each to be 16bit to store the next step + // Next stage is to sum pairwise + // Dest1: + // ADDP Dest2, Dest1: TmpCombine1 + // ADDP Dest4, Dest3: TmpCombine2 + // TmpCombine1.8H[0] = Dest1.8H[0] + Dest1.8H[1]; + // TmpCombine1.8H[1] = Dest1.8H[2] + Dest1.8H[3]; + // TmpCombine1.8H[2] = Dest1.8H[4] + Dest1.8H[5]; + // TmpCombine1.8H[3] = Dest1.8H[6] + Dest1.8H[7]; + // TmpCombine1.8H[4] = Dest2.8H[0] + Dest2.8H[1]; + // TmpCombine1.8H[5] = Dest2.8H[2] + Dest2.8H[3]; + // TmpCombine1.8H[6] = Dest2.8H[4] + Dest2.8H[5]; + // TmpCombine1.8H[7] = Dest2.8H[6] + Dest2.8H[7]; + // + // ADDP TmpCombine2, TmpCombine1: FinalCombine + // FinalCombine.8H[0] = TmpCombine1.8H[0] + TmpCombine1.8H[1] + // FinalCombine.8H[1] = TmpCombine1.8H[2] + TmpCombine1.8H[3] + // FinalCombine.8H[2] = TmpCombine1.8H[4] + TmpCombine1.8H[5] + // FinalCombine.8H[3] = TmpCombine1.8H[6] + TmpCombine1.8H[7] + // FinalCombine.8H[4] = TmpCombine2.8H[0] + TmpCombine2.8H[1] + // FinalCombine.8H[5] = TmpCombine2.8H[2] + TmpCombine2.8H[3] + // FinalCombine.8H[6] = TmpCombine2.8H[4] + TmpCombine2.8H[5] + // FinalCombine.8H[7] = TmpCombine2.8H[6] + TmpCombine2.8H[7] + + auto TmpCombine1 = _VAddP(16, 2, Dest1, Dest2); + auto TmpCombine2 = _VAddP(16, 2, Dest3, Dest4); + + auto FinalCombine = _VAddP(16, 2, TmpCombine1, TmpCombine2); + + // This now contains our results but they are in the wrong order. + // We need to swizzle the results in to the correct ordering + // Result.8H[0] = FinalCombine.8H[0] + // Result.8H[1] = FinalCombine.8H[2] + // Result.8H[2] = FinalCombine.8H[4] + // Result.8H[3] = FinalCombine.8H[6] + // Result.8H[4] = FinalCombine.8H[1] + // Result.8H[5] = FinalCombine.8H[3] + // Result.8H[6] = FinalCombine.8H[5] + // Result.8H[7] = FinalCombine.8H[7] + + auto Even = _VUnZip(16, 2, FinalCombine, FinalCombine); + auto Odd = _VUnZip2(16, 2, FinalCombine, FinalCombine); + auto Result = _VInsElement(16, 8, 1, 0, Even, Odd); + + StoreResult(FPRClass, Op, Result, -1); +} + +} diff --git a/External/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp b/External/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp new file mode 100644 index 000000000..7e3051d39 --- /dev/null +++ b/External/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp @@ -0,0 +1,1276 @@ +/* +$info$ +tags: frontend|x86-to-ir, opcodes|dispatcher-implementations +desc: Handles x86/64 x87 to IR +$end_info$ +*/ + +#include "Interface/Core/OpcodeDispatcher.h" + +#include + +namespace FEXCore::IR { +#define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op + +OrderedNode *OpDispatchBuilder::GetX87Top() { + // Yes, we are storing 3 bits in a single flag register. + // Deal with it + return _LoadContext(1, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC, GPRClass); +} + +void OpDispatchBuilder::SetX87Top(OrderedNode *Value) { + _StoreContext(GPRClass, 1, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC, Value); +} + +template +void OpDispatchBuilder::FLD(OpcodeArgs) { + // Update TOP + auto orig_top = GetX87Top(); + auto mask = _Constant(7); + + size_t read_width = (width == 80) ? 16 : width / 8; + + OrderedNode *data{}; + + if (!Op->Src[0].IsNone()) { + // Read from memory + data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], read_width, Op->Flags, -1); + } + else { + // Implicit arg + auto offset = _Constant(Op->OP & 7); + data = _And(_Add(orig_top, offset), mask); + data = _LoadContextIndexed(data, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + } + OrderedNode *converted = data; + + // Convert to 80bit float + if constexpr (width == 32 || width == 64) { + converted = _F80CVTTo(data, width / 8); + } + + auto top = _And(_Sub(orig_top, _Constant(1)), mask); + SetX87Top(top); + // Write to ST[TOP] + _StoreContextIndexed(converted, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + //_StoreContext(converted, 16, offsetof(FEXCore::Core::CPUState, mm[7][0])); +} + +template +void OpDispatchBuilder::FLD<32>(OpcodeArgs); +template +void OpDispatchBuilder::FLD<64>(OpcodeArgs); +template +void OpDispatchBuilder::FLD<80>(OpcodeArgs); + +void OpDispatchBuilder::FBLD(OpcodeArgs) { + // Update TOP + auto orig_top = GetX87Top(); + auto mask = _Constant(7); + auto top = _And(_Sub(orig_top, _Constant(1)), mask); + SetX87Top(top); + + // Read from memory + OrderedNode *data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], 16, Op->Flags, -1); + OrderedNode *converted = _F80BCDLoad(data); + _StoreContextIndexed(converted, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +void OpDispatchBuilder::FBSTP(OpcodeArgs) { + auto orig_top = GetX87Top(); + auto data = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + OrderedNode *converted = _F80BCDStore(data); + + StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, 10, 1); + + auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7)); + SetX87Top(top); +} + +template +void OpDispatchBuilder::FLD_Const(OpcodeArgs) { + // Update TOP + auto orig_top = GetX87Top(); + auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7)); + SetX87Top(top); + + auto low = _Constant(Lower); + auto high = _Constant(Upper); + OrderedNode *data = _VCastFromGPR(16, 8, low); + data = _VInsGPR(16, 8, data, high, 1); + // Write to ST[TOP] + _StoreContextIndexed(data, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +template +void OpDispatchBuilder::FLD_Const<0x8000'0000'0000'0000, 0b0'011'1111'1111'1111>(OpcodeArgs); // 1.0 +template +void OpDispatchBuilder::FLD_Const<0xD49A'784B'CD1B'8AFE, 0x4000>(OpcodeArgs); // log2l(10) +template +void OpDispatchBuilder::FLD_Const<0xB8AA'3B29'5C17'F0BC, 0x3FFF>(OpcodeArgs); // log2l(e) +template +void OpDispatchBuilder::FLD_Const<0xC90F'DAA2'2168'C235, 0x4000>(OpcodeArgs); // pi +template +void OpDispatchBuilder::FLD_Const<0x9A20'9A84'FBCF'F799, 0x3FFD>(OpcodeArgs); // log10l(2) +template +void OpDispatchBuilder::FLD_Const<0xB172'17F7'D1CF'79AC, 0x3FFE>(OpcodeArgs); // log(2) +template +void OpDispatchBuilder::FLD_Const<0, 0>(OpcodeArgs); // 0.0 + +void OpDispatchBuilder::FILD(OpcodeArgs) { + // Update TOP + auto orig_top = GetX87Top(); + auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7)); + SetX87Top(top); + + size_t read_width = GetSrcSize(Op); + + // Read from memory + auto data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], read_width, Op->Flags, -1); + + auto zero = _Constant(0); + + // Sign extend to 64bits + if (read_width != 8) + data = _Sext(read_width * 8, data); + + // Extract sign and make interger absolute + auto sign = _Select(COND_SLT, data, zero, _Constant(0x8000), zero); + auto absolute = _Select(COND_SLT, data, zero, _Sub(zero, data), data); + + // left justify the absolute interger + auto shift = _Sub(_Constant(63), _FindMSB(absolute)); + auto shifted = _Lshl(absolute, shift); + + auto adjusted_exponent = _Sub(_Constant(0x3fff + 63), shift); + auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent); + auto upper = _Or(sign, zeroed_exponent); + + + OrderedNode *converted = _VCastFromGPR(16, 8, shifted); + converted = _VInsElement(16, 8, 1, 0, converted, _VCastFromGPR(16, 8, upper)); + + // Write to ST[TOP] + _StoreContextIndexed(converted, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +template +void OpDispatchBuilder::FST(OpcodeArgs) { + auto orig_top = GetX87Top(); + auto data = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + if constexpr (width == 80) { + StoreResult_WithOpSize(FPRClass, Op, Op->Dest, data, 10, 1); + } + else if constexpr (width == 32 || width == 64) { + auto result = _F80CVT(data, width / 8); + StoreResult_WithOpSize(FPRClass, Op, Op->Dest, result, width / 8, 1); + } + + if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { + auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7)); + SetX87Top(top); + } +} + +template +void OpDispatchBuilder::FST<32>(OpcodeArgs); +template +void OpDispatchBuilder::FST<64>(OpcodeArgs); +template +void OpDispatchBuilder::FST<80>(OpcodeArgs); + +template +void OpDispatchBuilder::FIST(OpcodeArgs) { + auto Size = GetSrcSize(Op); + + auto orig_top = GetX87Top(); + OrderedNode *data = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + data = _F80CVTInt(data, Truncate, Size); + + StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, 1); + + if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { + auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7)); + SetX87Top(top); + } +} + +template +void OpDispatchBuilder::FIST(OpcodeArgs); +template +void OpDispatchBuilder::FIST(OpcodeArgs); + +template +void OpDispatchBuilder::FADD(OpcodeArgs) { + auto top = GetX87Top(); + OrderedNode *StackLocation = top; + + OrderedNode *arg{}; + OrderedNode *b{}; + + auto mask = _Constant(7); + + if (!Op->Src[0].IsNone()) { + // Memory arg + if constexpr (width == 16 || width == 32 || width == 64) { + if constexpr (Integer) { + arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); + b = _F80CVTToInt(arg, width / 8); + } + else { + arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + b = _F80CVTTo(arg, width / 8); + } + } + } else { + // Implicit arg + auto offset = _Constant(Op->OP & 7); + arg = _And(_Add(top, offset), mask); + if constexpr (ResInST0 == OpResult::RES_STI) { + StackLocation = arg; + } + b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + } + + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + auto result = _F80Add(a, b); + + if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { + top = _And(_Add(top, _Constant(1)), mask); + SetX87Top(top); + } + + // Write to ST[TOP] + _StoreContextIndexed(result, StackLocation, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +template +void OpDispatchBuilder::FADD<32, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FADD<64, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FADD<80, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FADD<80, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs); + +template +void OpDispatchBuilder::FADD<16, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FADD<32, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); + +template +void OpDispatchBuilder::FMUL(OpcodeArgs) { + auto top = GetX87Top(); + OrderedNode *StackLocation = top; + OrderedNode *arg{}; + OrderedNode *b{}; + + auto mask = _Constant(7); + + if (!Op->Src[0].IsNone()) { + // Memory arg + + if constexpr (width == 16 || width == 32 || width == 64) { + if constexpr (Integer) { + arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); + b = _F80CVTToInt(arg, width / 8); + } + else { + arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + b = _F80CVTTo(arg, width / 8); + } + } + } else { + // Implicit arg + auto offset = _Constant(Op->OP & 7); + arg = _And(_Add(top, offset), mask); + if constexpr (ResInST0 == OpResult::RES_STI) { + StackLocation = arg; + } + + b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + } + + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + auto result = _F80Mul(a, b); + + if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { + top = _And(_Add(top, _Constant(1)), mask); + SetX87Top(top); + } + + // Write to ST[TOP] + _StoreContextIndexed(result, StackLocation, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +template +void OpDispatchBuilder::FMUL<32, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FMUL<64, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FMUL<80, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FMUL<80, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs); + +template +void OpDispatchBuilder::FMUL<16, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FMUL<32, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); + +template +void OpDispatchBuilder::FDIV(OpcodeArgs) { + auto top = GetX87Top(); + OrderedNode *StackLocation = top; + OrderedNode *arg{}; + OrderedNode *b{}; + + auto mask = _Constant(7); + + if (!Op->Src[0].IsNone()) { + // Memory arg + + if constexpr (width == 16 || width == 32 || width == 64) { + if constexpr (Integer) { + arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); + b = _F80CVTToInt(arg, width / 8); + } + else { + arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + b = _F80CVTTo(arg, width / 8); + } + } + } else { + // Implicit arg + auto offset = _Constant(Op->OP & 7); + arg = _And(_Add(top, offset), mask); + if constexpr (ResInST0 == OpResult::RES_STI) { + StackLocation = arg; + } + + b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + } + + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + OrderedNode *result{}; + if constexpr (reverse) { + result = _F80Div(b, a); + } + else { + result = _F80Div(a, b); + } + + if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { + top = _And(_Add(top, _Constant(1)), mask); + SetX87Top(top); + } + + // Write to ST[TOP] + _StoreContextIndexed(result, StackLocation, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +template +void OpDispatchBuilder::FDIV<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FDIV<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); + +template +void OpDispatchBuilder::FDIV<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FDIV<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); + +template +void OpDispatchBuilder::FDIV<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FDIV<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); + +template +void OpDispatchBuilder::FDIV<80, false, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs); +template +void OpDispatchBuilder::FDIV<80, false, true, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs); + +template +void OpDispatchBuilder::FDIV<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FDIV<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); + +template +void OpDispatchBuilder::FDIV<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FDIV<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); + +template +void OpDispatchBuilder::FSUB(OpcodeArgs) { + auto top = GetX87Top(); + OrderedNode *StackLocation = top; + OrderedNode *arg{}; + OrderedNode *b{}; + + auto mask = _Constant(7); + + if (!Op->Src[0].IsNone()) { + // Memory arg + + if constexpr (width == 16 || width == 32 || width == 64) { + if constexpr (Integer) { + arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); + b = _F80CVTToInt(arg, width / 8); + } + else { + arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + b = _F80CVTTo(arg, width / 8); + } + } + } else { + // Implicit arg + auto offset = _Constant(Op->OP & 7); + arg = _And(_Add(top, offset), mask); + if constexpr (ResInST0 == OpResult::RES_STI) { + StackLocation = arg; + } + b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + } + + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + OrderedNode *result{}; + if constexpr (reverse) { + result = _F80Sub(b, a); + } + else { + result = _F80Sub(a, b); + } + + if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { + top = _And(_Add(top, _Constant(1)), mask); + SetX87Top(top); + } + + // Write to ST[TOP] + _StoreContextIndexed(result, StackLocation, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +template +void OpDispatchBuilder::FSUB<32, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FSUB<32, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); + +template +void OpDispatchBuilder::FSUB<64, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FSUB<64, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); + +template +void OpDispatchBuilder::FSUB<80, false, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FSUB<80, false, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); + +template +void OpDispatchBuilder::FSUB<80, false, false, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs); +template +void OpDispatchBuilder::FSUB<80, false, true, OpDispatchBuilder::OpResult::RES_STI>(OpcodeArgs); + +template +void OpDispatchBuilder::FSUB<16, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FSUB<16, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); + +template +void OpDispatchBuilder::FSUB<32, true, false, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); +template +void OpDispatchBuilder::FSUB<32, true, true, OpDispatchBuilder::OpResult::RES_ST0>(OpcodeArgs); + +void OpDispatchBuilder::FCHS(OpcodeArgs) { + auto top = GetX87Top(); + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + auto low = _Constant(0); + auto high = _Constant(0b1'000'0000'0000'0000); + OrderedNode *data = _VCastFromGPR(16, 8, low); + data = _VInsGPR(16, 8, data, high, 1); + + auto result = _VXor(a, data, 16, 1); + + // Write to ST[TOP] + _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +void OpDispatchBuilder::FABS(OpcodeArgs) { + auto top = GetX87Top(); + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + auto low = _Constant(~0ULL); + auto high = _Constant(0b0'111'1111'1111'1111); + OrderedNode *data = _VCastFromGPR(16, 8, low); + data = _VInsGPR(16, 8, data, high, 1); + + auto result = _VAnd(a, data, 16, 1); + + // Write to ST[TOP] + _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +void OpDispatchBuilder::FTST(OpcodeArgs) { + auto top = GetX87Top(); + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + auto low = _Constant(0); + OrderedNode *data = _VCastFromGPR(16, 8, low); + + OrderedNode *Res = _F80Cmp(a, data, + (1 << FCMP_FLAG_EQ) | + (1 << FCMP_FLAG_LT) | + (1 << FCMP_FLAG_UNORDERED)); + + OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT); + OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ); + OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED); + HostFlag_CF = _Or(HostFlag_CF, HostFlag_Unordered); + HostFlag_ZF = _Or(HostFlag_ZF, HostFlag_Unordered); + + SetRFLAG(HostFlag_CF); + SetRFLAG(_Constant(0)); + SetRFLAG(HostFlag_Unordered); + SetRFLAG(HostFlag_ZF); +} + +void OpDispatchBuilder::FRNDINT(OpcodeArgs) { + auto top = GetX87Top(); + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + auto result = _F80Round(a); + + // Write to ST[TOP] + _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +void OpDispatchBuilder::FXTRACT(OpcodeArgs) { + auto orig_top = GetX87Top(); + auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7)); + SetX87Top(top); + + auto a = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + auto exp = _F80XTRACT_EXP(a); + auto sig = _F80XTRACT_SIG(a); + + // Write to ST[TOP] + _StoreContextIndexed(exp, orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + _StoreContextIndexed(sig, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +void OpDispatchBuilder::FNINIT(OpcodeArgs) { + // Init FCW to 0x037 + auto NewFCW = _Constant(16, 0x037); + _F80LoadFCW(NewFCW); + _StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW); + + // Init FSW to 0 + SetX87Top(_Constant(0)); + + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + + // XXX: Add FTW support +} + +template +void OpDispatchBuilder::FCOMI(OpcodeArgs) { + auto top = GetX87Top(); + auto mask = _Constant(7); + + OrderedNode *arg{}; + OrderedNode *b{}; + + if (!Op->Src[0].IsNone()) { + // Memory arg + if constexpr (width == 16 || width == 32 || width == 64) { + if constexpr (Integer) { + arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); + b = _F80CVTToInt(arg, width / 8); + } + else { + arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1); + b = _F80CVTTo(arg, width / 8); + } + } + } else { + // Implicit arg + auto offset = _Constant(Op->OP & 7); + arg = _And(_Add(top, offset), mask); + b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + } + + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + OrderedNode *Res = _F80Cmp(a, b, + (1 << FCMP_FLAG_EQ) | + (1 << FCMP_FLAG_LT) | + (1 << FCMP_FLAG_UNORDERED)); + + OrderedNode *HostFlag_CF = _GetHostFlag(Res, FCMP_FLAG_LT); + OrderedNode *HostFlag_ZF = _GetHostFlag(Res, FCMP_FLAG_EQ); + OrderedNode *HostFlag_Unordered = _GetHostFlag(Res, FCMP_FLAG_UNORDERED); + HostFlag_CF = _Or(HostFlag_CF, HostFlag_Unordered); + HostFlag_ZF = _Or(HostFlag_ZF, HostFlag_Unordered); + + if constexpr (whichflags == FCOMIFlags::FLAGS_X87) { + SetRFLAG(HostFlag_CF); + SetRFLAG(_Constant(0)); + SetRFLAG(HostFlag_Unordered); + SetRFLAG(HostFlag_ZF); + } + else { + SetRFLAG(HostFlag_CF); + SetRFLAG(HostFlag_ZF); + SetRFLAG(HostFlag_Unordered); + } + + + if constexpr (poptwice) { + top = _And(_Add(top, _Constant(2)), mask); + SetX87Top(top); + } + else if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { + top = _And(_Add(top, _Constant(1)), mask); + SetX87Top(top); + } +} + +template +void OpDispatchBuilder::FCOMI<32, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs); + +template +void OpDispatchBuilder::FCOMI<64, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs); + +template +void OpDispatchBuilder::FCOMI<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs); +template +void OpDispatchBuilder::FCOMI<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_RFLAGS, false>(OpcodeArgs); +template +void OpDispatchBuilder::FCOMI<80, false, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, true>(OpcodeArgs); + +template +void OpDispatchBuilder::FCOMI<16, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs); + +template +void OpDispatchBuilder::FCOMI<32, true, OpDispatchBuilder::FCOMIFlags::FLAGS_X87, false>(OpcodeArgs); + + +void OpDispatchBuilder::FXCH(OpcodeArgs) { + auto top = GetX87Top(); + OrderedNode* arg; + + auto mask = _Constant(7); + + // Implicit arg + auto offset = _Constant(Op->OP & 7); + arg = _And(_Add(top, offset), mask); + + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + auto b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + // Write to ST[TOP] + _StoreContextIndexed(b, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + _StoreContextIndexed(a, arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +void OpDispatchBuilder::FST(OpcodeArgs) { + auto top = GetX87Top(); + OrderedNode* arg; + + auto mask = _Constant(7); + + // Implicit arg + auto offset = _Constant(Op->OP & 7); + arg = _And(_Add(top, offset), mask); + + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + // Write to ST[TOP] + _StoreContextIndexed(a, arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { + top = _And(_Add(top, _Constant(1)), _Constant(7)); + SetX87Top(top); + } +} + +template +void OpDispatchBuilder::X87UnaryOp(OpcodeArgs) { + auto top = GetX87Top(); + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + auto result = _F80Round(a); + // Overwrite the op + result.first->Header.Op = IROp; + + // Write to ST[TOP] + _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +template +void OpDispatchBuilder::X87UnaryOp(OpcodeArgs); +template +void OpDispatchBuilder::X87UnaryOp(OpcodeArgs); +template +void OpDispatchBuilder::X87UnaryOp(OpcodeArgs); +template +void OpDispatchBuilder::X87UnaryOp(OpcodeArgs); + +template +void OpDispatchBuilder::X87BinaryOp(OpcodeArgs) { + auto top = GetX87Top(); + + auto mask = _Constant(7); + OrderedNode *st1 = _And(_Add(top, _Constant(1)), mask); + + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + st1 = _LoadContextIndexed(st1, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + auto result = _F80Add(a, st1); + // Overwrite the op + result.first->Header.Op = IROp; + + if constexpr (IROp == IR::OP_F80FPREM) { + //TODO: Set C0 to Q2, C3 to Q1, C1 to Q0 + SetRFLAG(_Constant(0)); + } + + // Write to ST[TOP] + _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +template +void OpDispatchBuilder::X87BinaryOp(OpcodeArgs); +template +void OpDispatchBuilder::X87BinaryOp(OpcodeArgs); +template +void OpDispatchBuilder::X87BinaryOp(OpcodeArgs); + +template +void OpDispatchBuilder::X87ModifySTP(OpcodeArgs) { + auto orig_top = GetX87Top(); + if (Inc) { + auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7)); + SetX87Top(top); + } + else { + auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7)); + SetX87Top(top); + } +} + +template +void OpDispatchBuilder::X87ModifySTP(OpcodeArgs); +template +void OpDispatchBuilder::X87ModifySTP(OpcodeArgs); + +void OpDispatchBuilder::X87SinCos(OpcodeArgs) { + auto orig_top = GetX87Top(); + auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7)); + SetX87Top(top); + + auto a = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + auto sin = _F80SIN(a); + auto cos = _F80COS(a); + + // Write to ST[TOP] + _StoreContextIndexed(sin, orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + _StoreContextIndexed(cos, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +void OpDispatchBuilder::X87FYL2X(OpcodeArgs) { + bool Plus1 = Op->OP == 0x01F9; // FYL2XP + + auto orig_top = GetX87Top(); + auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7)); + SetX87Top(top); + + OrderedNode *st0 = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + OrderedNode *st1 = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + if (Plus1) { + auto low = _Constant(0x8000'0000'0000'0000); + auto high = _Constant(0b0'011'1111'1111'1111); + OrderedNode *data = _VCastFromGPR(16, 8, low); + data = _VInsGPR(16, 8, data, high, 1); + st0 = _F80Add(st0, data); + } + + auto result = _F80FYL2X(st0, st1); + + // Write to ST[TOP] + _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +void OpDispatchBuilder::X87TAN(OpcodeArgs) { + auto orig_top = GetX87Top(); + auto top = _And(_Sub(orig_top, _Constant(1)), _Constant(7)); + SetX87Top(top); + + auto a = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + auto result = _F80TAN(a); + + auto low = _Constant(0x8000'0000'0000'0000); + auto high = _Constant(0b0'011'1111'1111'1111); + OrderedNode *data = _VCastFromGPR(16, 8, low); + data = _VInsGPR(16, 8, data, high, 1); + + // Write to ST[TOP] + _StoreContextIndexed(result, orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + _StoreContextIndexed(data, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +void OpDispatchBuilder::X87ATAN(OpcodeArgs) { + auto orig_top = GetX87Top(); + auto top = _And(_Add(orig_top, _Constant(1)), _Constant(7)); + SetX87Top(top); + + auto a = _LoadContextIndexed(orig_top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + OrderedNode *st1 = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + auto result = _F80ATAN(st1, a); + + // Write to ST[TOP] + _StoreContextIndexed(result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +void OpDispatchBuilder::X87LDENV(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false); + Mem = AppendSegmentOffset(Mem, Op->Flags); + + auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2); + _F80LoadFCW(NewFCW); + _StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW); + + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 1)); + auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size); + + // Strip out the FSW information + auto Top = _Bfe(3, 11, NewFSW); + SetX87Top(Top); + + auto C0 = _Bfe(1, 8, NewFSW); + auto C1 = _Bfe(1, 9, NewFSW); + auto C2 = _Bfe(1, 10, NewFSW); + auto C3 = _Bfe(1, 14, NewFSW); + + SetRFLAG(C0); + SetRFLAG(C1); + SetRFLAG(C2); + SetRFLAG(C3); +} + +void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) { + // 14 bytes for 16bit + // 2 Bytes : FCW + // 2 Bytes : FSW + // 2 bytes : FTW + // 2 bytes : Instruction offset + // 2 bytes : Instruction CS selector + // 2 bytes : Data offset + // 2 bytes : Data selector + + // 28 bytes for 32bit + // 4 bytes : FCW + // 4 bytes : FSW + // 4 bytes : FTW + // 4 bytes : Instruction pointer + // 2 bytes : instruction pointer selector + // 2 bytes : Opcode + // 4 bytes : data pointer offset + // 4 bytes : data pointer selector + + auto Size = GetDstSize(Op); + OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false); + Mem = AppendSegmentOffset(Mem, Op->Flags); + + { + auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass); + _StoreMem(GPRClass, Size, Mem, FCW, Size); + } + + { + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 1)); + // We must construct the FSW from our various bits + OrderedNode *FSW = _Constant(0); + auto Top = GetX87Top(); + FSW = _Or(FSW, _Lshl(Top, _Constant(11))); + + auto C0 = GetRFLAG(FEXCore::X86State::X87FLAG_C0_LOC); + auto C1 = GetRFLAG(FEXCore::X86State::X87FLAG_C1_LOC); + auto C2 = GetRFLAG(FEXCore::X86State::X87FLAG_C2_LOC); + auto C3 = GetRFLAG(FEXCore::X86State::X87FLAG_C3_LOC); + + FSW = _Or(FSW, _Lshl(C0, _Constant(8))); + FSW = _Or(FSW, _Lshl(C1, _Constant(9))); + FSW = _Or(FSW, _Lshl(C2, _Constant(10))); + FSW = _Or(FSW, _Lshl(C3, _Constant(14))); + _StoreMem(GPRClass, Size, MemLocation, FSW, Size); + } + + auto ZeroConst = _Constant(0); + + { + // FTW + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 2)); + _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); + } + + { + // Instruction Offset + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 3)); + _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); + } + + { + // Instruction CS selector (+ Opcode) + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 4)); + _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); + } + + { + // Data pointer offset + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 5)); + _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); + } + + { + // Data pointer selector + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 6)); + _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); + } +} + +void OpDispatchBuilder::X87FLDCW(OpcodeArgs) { + OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); + _F80LoadFCW(NewFCW); + _StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW); +} + +void OpDispatchBuilder::X87FSTCW(OpcodeArgs) { + auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass); + + StoreResult(GPRClass, Op, FCW, -1); +} + +void OpDispatchBuilder::X87LDSW(OpcodeArgs) { + OrderedNode *NewFSW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1); + // Strip out the FSW information + auto Top = _Bfe(3, 11, NewFSW); + SetX87Top(Top); + + auto C0 = _Bfe(1, 8, NewFSW); + auto C1 = _Bfe(1, 9, NewFSW); + auto C2 = _Bfe(1, 10, NewFSW); + auto C3 = _Bfe(1, 14, NewFSW); + + SetRFLAG(C0); + SetRFLAG(C1); + SetRFLAG(C2); + SetRFLAG(C3); +} + +void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) { + // We must construct the FSW from our various bits + OrderedNode *FSW = _Constant(0); + auto Top = GetX87Top(); + FSW = _Or(FSW, _Lshl(Top, _Constant(11))); + + auto C0 = GetRFLAG(FEXCore::X86State::X87FLAG_C0_LOC); + auto C1 = GetRFLAG(FEXCore::X86State::X87FLAG_C1_LOC); + auto C2 = GetRFLAG(FEXCore::X86State::X87FLAG_C2_LOC); + auto C3 = GetRFLAG(FEXCore::X86State::X87FLAG_C3_LOC); + + FSW = _Or(FSW, _Lshl(C0, _Constant(8))); + FSW = _Or(FSW, _Lshl(C1, _Constant(9))); + FSW = _Or(FSW, _Lshl(C2, _Constant(10))); + FSW = _Or(FSW, _Lshl(C3, _Constant(14))); + + StoreResult(GPRClass, Op, FSW, -1); +} + +void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) { + // 14 bytes for 16bit + // 2 Bytes : FCW + // 2 Bytes : FSW + // 2 bytes : FTW + // 2 bytes : Instruction offset + // 2 bytes : Instruction CS selector + // 2 bytes : Data offset + // 2 bytes : Data selector + + // 28 bytes for 32bit + // 4 bytes : FCW + // 4 bytes : FSW + // 4 bytes : FTW + // 4 bytes : Instruction pointer + // 2 bytes : instruction pointer selector + // 2 bytes : Opcode + // 4 bytes : data pointer offset + // 4 bytes : data pointer selector + + auto Size = GetDstSize(Op); + OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, -1, false); + Mem = AppendSegmentOffset(Mem, Op->Flags); + + OrderedNode *Top = GetX87Top(); + { + auto FCW = _LoadContext(2, offsetof(FEXCore::Core::CPUState, FCW), GPRClass); + _StoreMem(GPRClass, Size, Mem, FCW, Size); + } + + { + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 1)); + // We must construct the FSW from our various bits + OrderedNode *FSW = _Constant(0); + FSW = _Or(FSW, _Lshl(Top, _Constant(11))); + + auto C0 = GetRFLAG(FEXCore::X86State::X87FLAG_C0_LOC); + auto C1 = GetRFLAG(FEXCore::X86State::X87FLAG_C1_LOC); + auto C2 = GetRFLAG(FEXCore::X86State::X87FLAG_C2_LOC); + auto C3 = GetRFLAG(FEXCore::X86State::X87FLAG_C3_LOC); + + FSW = _Or(FSW, _Lshl(C0, _Constant(8))); + FSW = _Or(FSW, _Lshl(C1, _Constant(9))); + FSW = _Or(FSW, _Lshl(C2, _Constant(10))); + FSW = _Or(FSW, _Lshl(C3, _Constant(14))); + _StoreMem(GPRClass, Size, MemLocation, FSW, Size); + } + + auto ZeroConst = _Constant(0); + + { + // FTW + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 2)); + _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); + } + + { + // Instruction Offset + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 3)); + _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); + } + + { + // Instruction CS selector (+ Opcode) + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 4)); + _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); + } + + { + // Data pointer offset + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 5)); + _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); + } + + { + // Data pointer selector + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 6)); + _StoreMem(GPRClass, Size, MemLocation, ZeroConst, Size); + } + + OrderedNode *ST0Location = _Add(Mem, _Constant(Size * 7)); + + auto OneConst = _Constant(1); + auto SevenConst = _Constant(7); + auto TenConst = _Constant(10); + for (int i = 0; i < 7; ++i) { + auto data = _LoadContextIndexed(Top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + _StoreMem(FPRClass, 16, ST0Location, data, 1); + ST0Location = _Add(ST0Location, TenConst); + Top = _And(_Add(Top, OneConst), SevenConst); + } + + // The final st(7) needs a bit of special handling here + auto data = _LoadContextIndexed(Top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + // ST7 broken in to two parts + // Lower 64bits [63:0] + // upper 16 bits [79:64] + _StoreMem(FPRClass, 8, ST0Location, data, 1); + ST0Location = _Add(ST0Location, _Constant(8)); + auto topBytes = _VExtractElement(16, 2, data, 4); + _StoreMem(FPRClass, 2, ST0Location, topBytes, 1); + + // reset to default + FNINIT(Op); +} + +void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) { + auto Size = GetSrcSize(Op); + OrderedNode *Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1, false); + Mem = AppendSegmentOffset(Mem, Op->Flags); + + auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2); + _F80LoadFCW(NewFCW); + _StoreContext(GPRClass, 2, offsetof(FEXCore::Core::CPUState, FCW), NewFCW); + + OrderedNode *MemLocation = _Add(Mem, _Constant(Size * 1)); + auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size); + + // Strip out the FSW information + OrderedNode *Top = _Bfe(3, 11, NewFSW); + SetX87Top(Top); + + auto C0 = _Bfe(1, 8, NewFSW); + auto C1 = _Bfe(1, 9, NewFSW); + auto C2 = _Bfe(1, 10, NewFSW); + auto C3 = _Bfe(1, 14, NewFSW); + + SetRFLAG(C0); + SetRFLAG(C1); + SetRFLAG(C2); + SetRFLAG(C3); + + OrderedNode *ST0Location = _Add(Mem, _Constant(Size * 7)); + + auto OneConst = _Constant(1); + auto SevenConst = _Constant(7); + auto TenConst = _Constant(10); + + auto low = _Constant(~0ULL); + auto high = _Constant(0xFFFF); + OrderedNode *Mask = _VCastFromGPR(16, 8, low); + Mask = _VInsGPR(16, 8, Mask, high, 1); + + for (int i = 0; i < 7; ++i) { + OrderedNode *Reg = _LoadMem(FPRClass, 16, ST0Location, 1); + // Mask off the top bits + Reg = _VAnd(16, 16, Reg, Mask); + + _StoreContextIndexed(Reg, Top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + + ST0Location = _Add(ST0Location, TenConst); + Top = _And(_Add(Top, OneConst), SevenConst); + } + + // The final st(7) needs a bit of special handling here + // ST7 broken in to two parts + // Lower 64bits [63:0] + // upper 16 bits [79:64] + + OrderedNode *Reg = _LoadMem(FPRClass, 8, ST0Location, 1); + ST0Location = _Add(ST0Location, _Constant(8)); + OrderedNode *RegHigh = _LoadMem(FPRClass, 2, ST0Location, 1); + Reg = _VInsElement(16, 2, 4, 0, Reg, RegHigh); + _StoreContextIndexed(Reg, Top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +void OpDispatchBuilder::X87FXAM(OpcodeArgs) { + auto top = GetX87Top(); + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + OrderedNode *Result = _VExtractToGPR(16, 8, a, 1); + + // Extract the sign bit + Result = _Lshr(Result, _Constant(15)); + SetRFLAG(Result); + + // Claim this is a normal number + // We don't support anything else + auto ZeroConst = _Constant(0); + auto OneConst = _Constant(1); + SetRFLAG(ZeroConst); + SetRFLAG(OneConst); + SetRFLAG(ZeroConst); +} + +void OpDispatchBuilder::X87FCMOV(OpcodeArgs) { + enum CompareType { + COMPARE_ZERO, + COMPARE_NOTZERO, + }; + uint32_t FLAGMask{}; + CompareType Type = COMPARE_ZERO; + OrderedNode *SrcCond; + + auto ZeroConst = _Constant(0); + auto OneConst = _Constant(1); + + uint16_t Opcode = Op->OP & 0b1111'1111'1000; + switch (Opcode) { + case 0x3'C0: + FLAGMask = 1 << FEXCore::X86State::RFLAG_CF_LOC; + Type = COMPARE_ZERO; + break; + case 0x2'C0: + FLAGMask = 1 << FEXCore::X86State::RFLAG_CF_LOC; + Type = COMPARE_NOTZERO; + break; + case 0x2'C8: + FLAGMask = 1 << FEXCore::X86State::RFLAG_ZF_LOC; + Type = COMPARE_NOTZERO; + break; + case 0x3'C8: + FLAGMask = 1 << FEXCore::X86State::RFLAG_ZF_LOC; + Type = COMPARE_ZERO; + break; + case 0x2'D0: + FLAGMask = (1 << FEXCore::X86State::RFLAG_ZF_LOC) | (1 << FEXCore::X86State::RFLAG_CF_LOC); + Type = COMPARE_NOTZERO; + break; + case 0x3'D0: + FLAGMask = (1 << FEXCore::X86State::RFLAG_ZF_LOC) | (1 << FEXCore::X86State::RFLAG_CF_LOC); + Type = COMPARE_ZERO; + break; + case 0x2'D8: + FLAGMask = 1 << FEXCore::X86State::RFLAG_PF_LOC; + Type = COMPARE_NOTZERO; + break; + case 0x3'D8: + FLAGMask = 1 << FEXCore::X86State::RFLAG_PF_LOC; + Type = COMPARE_ZERO; + break; + default: + LOGMAN_MSG_A("Unhandled FCMOV op: 0x%x", Opcode); + break; + } + + auto MaskConst = _Constant(FLAGMask); + + auto RFLAG = GetPackedRFLAG(false); + + auto AndOp = _And(RFLAG, MaskConst); + switch (Type) { + case COMPARE_ZERO: { + SrcCond = _Select(FEXCore::IR::COND_EQ, + AndOp, ZeroConst, OneConst, ZeroConst); + break; + } + case COMPARE_NOTZERO: { + SrcCond = _Select(FEXCore::IR::COND_EQ, + AndOp, ZeroConst, ZeroConst, OneConst); + break; + } + } + + SrcCond = _Sbfe(1, 0, SrcCond); + + OrderedNode *VecCond = _VCastFromGPR(16, 8, SrcCond); + VecCond = _VInsGPR(16, 8, VecCond, SrcCond, 1); + + auto top = GetX87Top(); + OrderedNode* arg; + + auto mask = _Constant(7); + + // Implicit arg + auto offset = _Constant(Op->OP & 7); + arg = _And(_Add(top, offset), mask); + + auto a = _LoadContextIndexed(top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + auto b = _LoadContextIndexed(arg, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); + auto Result = _VBSL(VecCond, b, a); + + // Write to ST[TOP] + _StoreContextIndexed(Result, top, 16, offsetof(FEXCore::Core::CPUState, mm[0][0]), 16, FPRClass); +} + +} From c2a97d4fc1c3c9485a67184afbc20b49262256c8 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Thu, 8 Jul 2021 10:33:36 -0700 Subject: [PATCH 2/2] Minor header shuffling to improve compilation time Shaves a few seconds off compile time --- External/FEXCore/Source/Interface/Context/Context.cpp | 1 + .../Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp | 2 ++ .../FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.cpp | 2 ++ .../Source/Interface/Core/Interpreter/InterpreterClass.h | 1 - External/FEXCore/Source/Interface/Core/JIT/Arm64/BranchOps.cpp | 2 ++ External/FEXCore/Source/Interface/Core/JIT/Arm64/JIT.cpp | 1 + External/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h | 1 - .../FEXCore/Source/Interface/Core/JIT/x86_64/BranchOps.cpp | 2 ++ External/FEXCore/Source/Interface/Core/JIT/x86_64/JIT.cpp | 1 + External/FEXCore/Source/Interface/Core/JIT/x86_64/JITClass.h | 1 - .../FEXCore/Source/Interface/Core/X86Tables/H0F38Tables.cpp | 2 ++ External/FEXCore/Source/Interface/Core/X86Tables/X86Tables.h | 3 +++ External/FEXCore/include/FEXCore/Debug/X86Tables.h | 3 --- Source/Tests/UnitTestGenerator.cpp | 2 +- 14 files changed, 17 insertions(+), 7 deletions(-) diff --git a/External/FEXCore/Source/Interface/Context/Context.cpp b/External/FEXCore/Source/Interface/Context/Context.cpp index 119c867bb..f0115a741 100644 --- a/External/FEXCore/Source/Interface/Context/Context.cpp +++ b/External/FEXCore/Source/Interface/Context/Context.cpp @@ -2,6 +2,7 @@ #include "Interface/Context/Context.h" #include "Interface/Core/Core.h" #include "Interface/Core/OpcodeDispatcher.h" +#include "Interface/Core/X86Tables/X86Tables.h" #include #include diff --git a/External/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp b/External/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp index 3eb37ba8c..64f856530 100644 --- a/External/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp +++ b/External/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp @@ -1,3 +1,5 @@ +#include "Interface/Core/LookupCache.h" + #include "Interface/Core/ArchHelpers/MContext.h" #include "Interface/Core/Dispatcher/Arm64Dispatcher.h" diff --git a/External/FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.cpp b/External/FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.cpp index 4a9f9623b..770c1726d 100644 --- a/External/FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.cpp +++ b/External/FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.cpp @@ -1,3 +1,5 @@ +#include "Interface/Core/LookupCache.h" + #include "Interface/Core/Dispatcher/X86Dispatcher.h" #include "Interface/Core/Interpreter/InterpreterClass.h" diff --git a/External/FEXCore/Source/Interface/Core/Interpreter/InterpreterClass.h b/External/FEXCore/Source/Interface/Core/Interpreter/InterpreterClass.h index 4ef0ae53e..c0796bc03 100644 --- a/External/FEXCore/Source/Interface/Core/Interpreter/InterpreterClass.h +++ b/External/FEXCore/Source/Interface/Core/Interpreter/InterpreterClass.h @@ -1,6 +1,5 @@ #pragma once -#include "Interface/Core/LookupCache.h" #include "Interface/Core/InternalThreadState.h" #include "Interface/Core/Dispatcher/Dispatcher.h" diff --git a/External/FEXCore/Source/Interface/Core/JIT/Arm64/BranchOps.cpp b/External/FEXCore/Source/Interface/Core/JIT/Arm64/BranchOps.cpp index a66043643..9a4893ccb 100644 --- a/External/FEXCore/Source/Interface/Core/JIT/Arm64/BranchOps.cpp +++ b/External/FEXCore/Source/Interface/Core/JIT/Arm64/BranchOps.cpp @@ -4,6 +4,8 @@ tags: backend|arm64 $end_info$ */ +#include "Interface/Core/LookupCache.h" + #include "Interface/Core/JIT/Arm64/JITClass.h" #include "Interface/Core/InternalThreadState.h" diff --git a/External/FEXCore/Source/Interface/Core/JIT/Arm64/JIT.cpp b/External/FEXCore/Source/Interface/Core/JIT/Arm64/JIT.cpp index 74195cdd6..c62be8e66 100644 --- a/External/FEXCore/Source/Interface/Core/JIT/Arm64/JIT.cpp +++ b/External/FEXCore/Source/Interface/Core/JIT/Arm64/JIT.cpp @@ -11,6 +11,7 @@ $end_info$ */ #include "Interface/Context/Context.h" +#include "Interface/Core/LookupCache.h" #include "Interface/Core/ArchHelpers/Arm64.h" #include "Interface/Core/ArchHelpers/MContext.h" diff --git a/External/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h b/External/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h index 002a7bc23..f6dee7866 100644 --- a/External/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h +++ b/External/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h @@ -6,7 +6,6 @@ $end_info$ #pragma once -#include "Interface/Core/LookupCache.h" #include "Interface/Core/ArchHelpers/Arm64Emitter.h" #include "Interface/Core/Dispatcher/Dispatcher.h" diff --git a/External/FEXCore/Source/Interface/Core/JIT/x86_64/BranchOps.cpp b/External/FEXCore/Source/Interface/Core/JIT/x86_64/BranchOps.cpp index 30f033608..d27712e65 100644 --- a/External/FEXCore/Source/Interface/Core/JIT/x86_64/BranchOps.cpp +++ b/External/FEXCore/Source/Interface/Core/JIT/x86_64/BranchOps.cpp @@ -4,6 +4,8 @@ tags: backend|x86-64 $end_info$ */ +#include "Interface/Core/LookupCache.h" + #include "Interface/Core/JIT/x86_64/JITClass.h" #include "Interface/IR/Passes/RegisterAllocationPass.h" diff --git a/External/FEXCore/Source/Interface/Core/JIT/x86_64/JIT.cpp b/External/FEXCore/Source/Interface/Core/JIT/x86_64/JIT.cpp index db5c257e0..9d6af5d11 100644 --- a/External/FEXCore/Source/Interface/Core/JIT/x86_64/JIT.cpp +++ b/External/FEXCore/Source/Interface/Core/JIT/x86_64/JIT.cpp @@ -6,6 +6,7 @@ $end_info$ */ #include "Interface/Context/Context.h" +#include "Interface/Core/LookupCache.h" #include "Interface/Core/Dispatcher/X86Dispatcher.h" #include "Interface/Core/JIT/x86_64/JITClass.h" diff --git a/External/FEXCore/Source/Interface/Core/JIT/x86_64/JITClass.h b/External/FEXCore/Source/Interface/Core/JIT/x86_64/JITClass.h index 318fc07d4..151a83531 100644 --- a/External/FEXCore/Source/Interface/Core/JIT/x86_64/JITClass.h +++ b/External/FEXCore/Source/Interface/Core/JIT/x86_64/JITClass.h @@ -6,7 +6,6 @@ $end_info$ #pragma once -#include "Interface/Core/LookupCache.h" #include "Interface/Core/BlockSamplingData.h" #include "Interface/Core/Dispatcher/Dispatcher.h" diff --git a/External/FEXCore/Source/Interface/Core/X86Tables/H0F38Tables.cpp b/External/FEXCore/Source/Interface/Core/X86Tables/H0F38Tables.cpp index 8a304bf75..6719d82be 100644 --- a/External/FEXCore/Source/Interface/Core/X86Tables/H0F38Tables.cpp +++ b/External/FEXCore/Source/Interface/Core/X86Tables/H0F38Tables.cpp @@ -4,6 +4,8 @@ tags: frontend|x86-tables $end_info$ */ +#include + #include "Interface/Core/X86Tables/X86Tables.h" namespace FEXCore::X86Tables { diff --git a/External/FEXCore/Source/Interface/Core/X86Tables/X86Tables.h b/External/FEXCore/Source/Interface/Core/X86Tables/X86Tables.h index 527dd8a59..d5265f727 100644 --- a/External/FEXCore/Source/Interface/Core/X86Tables/X86Tables.h +++ b/External/FEXCore/Source/Interface/Core/X86Tables/X86Tables.h @@ -6,6 +6,7 @@ $end_info$ #pragma once #include +#include #include @@ -113,5 +114,7 @@ static inline void GenerateX87Table(X86InstInfo *FinalTable, U16U8InfoStruct con } }; +void InitializeInfoTables(Context::OperatingMode Mode); + } diff --git a/External/FEXCore/include/FEXCore/Debug/X86Tables.h b/External/FEXCore/include/FEXCore/Debug/X86Tables.h index 2606e2018..8d6e1d2a8 100644 --- a/External/FEXCore/include/FEXCore/Debug/X86Tables.h +++ b/External/FEXCore/include/FEXCore/Debug/X86Tables.h @@ -1,6 +1,5 @@ #pragma once -#include #include #include @@ -500,6 +499,4 @@ extern FEX_DEFAULT_VISIBILITY X86InstInfo XOPTableGroupOps[MAX_XOP_GROUP_TABLE_S // EVEX extern FEX_DEFAULT_VISIBILITY X86InstInfo EVEXTableOps[MAX_EVEX_TABLE_SIZE]; - -FEX_DEFAULT_VISIBILITY void InitializeInfoTables(Context::OperatingMode Mode); } diff --git a/Source/Tests/UnitTestGenerator.cpp b/Source/Tests/UnitTestGenerator.cpp index 0c6cb4951..024695dd6 100644 --- a/Source/Tests/UnitTestGenerator.cpp +++ b/Source/Tests/UnitTestGenerator.cpp @@ -2407,7 +2407,7 @@ int main(int argc, char **argv, char **const envp) { LOGMAN_THROW_A(!Args.empty(), "Not enough arguments"); - FEXCore::X86Tables::InitializeInfoTables(FEXCore::Context::MODE_64BIT); + FEXCore::Context::InitializeStaticTables(FEXCore::Context::MODE_64BIT); Code.reserve(4096*128); Filepath = Args[0];