// SPDX-License-Identifier: MIT #pragma once #include "Interface/Core/Frontend.h" #include "Interface/Core/X86Tables/X86Tables.h" #include "Interface/Context/Context.h" #include "Interface/IR/IREmitter.h" #include #include #include #include #include #include #include #include #include #include #include #include #include namespace FEXCore::IR { class Pass; class PassManager; enum class MemoryAccessType { // Choose TSO or Non-TSO depending on access type DEFAULT, // TSO access behaviour TSO, // Non-TSO access behaviour NONTSO, // Non-temporal streaming STREAM, }; enum class BTAction { BTNone, BTClear, BTSet, BTComplement, }; struct LoadSourceOptions { // Alignment of the load in bytes. -1 signifies unaligned int8_t Align = -1; // Whether or not to load the data if a memory access occurs. // If set to false, then the address that would have been loaded from // will be returned instead. // // Note: If returning the address, make sure to apply the segment offset // after with AppendSegmentOffset(). // bool LoadData = true; // Use to force a load even if the underlying type isn't loadable. bool ForceLoad = false; // Specifies the access type of the load. MemoryAccessType AccessType = MemoryAccessType::DEFAULT; // Whether or not a zero extend should clear the upper bits // in the register (e.g. an 8-bit load would clear the upper 24 bits // or 56 bits depending on the operating mode). // If true, no zero-extension occurs. bool AllowUpperGarbage = false; }; struct AddressMode { Ref Base {nullptr}; Ref Index {nullptr}; MemOffsetType IndexType = MEM_OFFSET_SXTX; uint8_t IndexScale = 1; int64_t Offset = 0; // Size in bytes for the address calculation. 8 for an arm64 hardware mode. uint8_t AddrSize; bool NonTSO; }; class OpDispatchBuilder final : public IREmitter { friend class FEXCore::IR::Pass; friend class FEXCore::IR::PassManager; public: enum class FlagsGenerationType : uint8_t { TYPE_NONE, TYPE_SUB, TYPE_MUL, TYPE_UMUL, TYPE_LOGICAL, TYPE_LSHLI, TYPE_LSHRI, TYPE_LSHRDI, TYPE_ASHRI, TYPE_BEXTR, TYPE_BLSI, TYPE_BLSMSK, TYPE_BLSR, TYPE_POPCOUNT, TYPE_BZHI, TYPE_ZCNT, TYPE_RDRAND, }; Ref GetNewJumpBlock(uint64_t RIP) { auto it = JumpTargets.find(RIP); LOGMAN_THROW_A_FMT(it != JumpTargets.end(), "Couldn't find block generated for 0x{:x}", RIP); return it->second.BlockEntry; } void SetNewBlockIfChanged(uint64_t RIP) { auto it = JumpTargets.find(RIP); if (it == JumpTargets.end()) { return; } it->second.HaveEmitted = true; if (CurrentCodeBlock->Wrapped(DualListData.ListBegin()).ID() == it->second.BlockEntry->Wrapped(DualListData.ListBegin()).ID()) { return; } // We have hit a RIP that is a jump target // Thus we need to end up in a new block SetCurrentCodeBlock(it->second.BlockEntry); } void StartNewBlock() { // If we loaded flags but didn't change them, invalidate the cached copy and move on. // Changes get stored out by CalculateDeferredFlags. CachedNZCV = nullptr; PossiblySetNZCVBits = ~0U; FlushRegisterCache(); // New block needs to reset segment telemetry. SegmentsNeedReadCheck = ~0U; // Need to clear any named constants that were cached. ClearCachedNamedConstants(); } IRPair Jump() { FlushRegisterCache(); return _Jump(); } IRPair Jump(Ref _TargetBlock) { FlushRegisterCache(); return _Jump(_TargetBlock); } IRPair CondJump(Ref _Cmp1, Ref _Cmp2, Ref _TrueBlock, Ref _FalseBlock, CondClassType _Cond = {COND_NEQ}, uint8_t _CompareSize = 0) { FlushRegisterCache(); return _CondJump(_Cmp1, _Cmp2, _TrueBlock, _FalseBlock, _Cond, _CompareSize); } IRPair CondJump(Ref ssa0, CondClassType cond = {COND_NEQ}) { FlushRegisterCache(); return _CondJump(ssa0, cond); } IRPair CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClassType cond = {COND_NEQ}) { FlushRegisterCache(); return _CondJump(ssa0, ssa1, ssa2, cond); } IRPair CondJumpNZCV(CondClassType Cond) { FlushRegisterCache(); // The jump will ignore the sources, so it doesn't matter what we put here. // Put an inline constant so RA+codegen will ignore altogether. auto Placeholder = _InlineConstant(0); return _CondJump(Placeholder, Placeholder, InvalidNode, InvalidNode, Cond, 0, true); } IRPair ExitFunction(Ref NewRIP) { FlushRegisterCache(); return _ExitFunction(NewRIP); } IRPair Break(BreakDefinition Reason) { FlushRegisterCache(); return _Break(Reason); } IRPair Thunk(Ref ArgPtr, SHA256Sum ThunkNameHash) { FlushRegisterCache(); return _Thunk(ArgPtr, ThunkNameHash); } bool FinishOp(uint64_t NextRIP, bool LastOp) { // If we are switching to a new block and this current block has yet to set a RIP // Then we need to insert an unconditional jump from the current block to the one we are going to // This happens most frequently when an instruction jumps backwards to another location // eg: // // nop dword [rax], eax // .label: // rdi, 0x8 // cmp qword [rdi-8], 0 // jne .label if (LastOp && !BlockSetRIP) { auto it = JumpTargets.find(NextRIP); if (it == JumpTargets.end()) { const uint8_t GPRSize = CTX->GetGPRSize(); // If we don't have a jump target to a new block then we have to leave // Set the RIP to the next instruction and leave auto RelocatedNextRIP = _EntrypointOffset(IR::SizeToOpSize(GPRSize), NextRIP - Entry); ExitFunction(RelocatedNextRIP); } else if (it != JumpTargets.end()) { Jump(it->second.BlockEntry); return true; } } if (LastOp) { LOGMAN_THROW_A_FMT(IsDeferredFlagsStored(), "FinishOp: Deferred flags weren't generated at end of block"); } BlockSetRIP = false; return false; } static bool CanHaveSideEffects(const FEXCore::X86Tables::X86InstInfo* TableInfo, FEXCore::X86Tables::DecodedOp Op) { if (TableInfo && TableInfo->Flags & X86Tables::InstFlags::FLAGS_DEBUG_MEM_ACCESS) { // If it is marked as having memory access then always say it has a side-effect. // Not always true but better to be safe. return true; } auto CanHaveSideEffects = false; auto HasPotentialMemoryAccess = [](const X86Tables::DecodedOperand& Operand) -> bool { if (Operand.IsNone()) { return false; } // This isn't guaranteed that all of these types will access memory, but be safe. return Operand.IsGPRDirect() || Operand.IsGPRIndirect() || Operand.IsRIPRelative() || Operand.IsSIB(); }; CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Dest); CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[0]); CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[1]); CanHaveSideEffects |= HasPotentialMemoryAccess(Op->Src[2]); return CanHaveSideEffects; } template void ForeachDirection(F&& Routine) { // Otherwise, prepare to branch. auto Zero = _Constant(0); // If the shift is zero, do not touch the flags. auto ForwardBlock = CreateNewCodeBlockAfter(GetCurrentBlock()); auto BackwardBlock = CreateNewCodeBlockAfter(ForwardBlock); auto ExitBlock = CreateNewCodeBlockAfter(BackwardBlock); auto DF = GetRFLAG(X86State::RFLAG_DF_RAW_LOC); CondJump(DF, Zero, ForwardBlock, BackwardBlock, {COND_EQ}); for (auto D = 0; D < 2; ++D) { SetCurrentCodeBlock(D ? BackwardBlock : ForwardBlock); StartNewBlock(); { Routine(D ? -1 : 1); Jump(ExitBlock); } } SetCurrentCodeBlock(ExitBlock); StartNewBlock(); } OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx); OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator& Allocator); void ResetWorkingList(); void ResetDecodeFailure() { NeedsBlockEnd = DecodeFailure = false; } bool HadDecodeFailure() const { return DecodeFailure; } bool NeedsBlockEnder() const { return NeedsBlockEnd; } void ResetHandledLock() { HandledLock = false; } bool HasHandledLock() const { return HandledLock; } void SetDumpIR(bool DumpIR) { ShouldDump = DumpIR; } bool ShouldDumpIR() const { return ShouldDump; } void BeginFunction(uint64_t RIP, const fextl::vector* Blocks, uint32_t NumInstructions); void Finalize(); // Dispatch builder functions #define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op void UnhandledOp(OpcodeArgs); template void MOVGPROp(OpcodeArgs); void MOVGPRNTOp(OpcodeArgs); void MOVVectorAlignedOp(OpcodeArgs); void MOVVectorUnalignedOp(OpcodeArgs); void MOVVectorNTOp(OpcodeArgs); template void ALUOp(OpcodeArgs); void INTOp(OpcodeArgs); template void SyscallOp(OpcodeArgs); void ThunkOp(OpcodeArgs); void LEAOp(OpcodeArgs); void NOPOp(OpcodeArgs); void RETOp(OpcodeArgs); void IRETOp(OpcodeArgs); void CallbackReturnOp(OpcodeArgs); void SecondaryALUOp(OpcodeArgs); template void ADCOp(OpcodeArgs); template void SBBOp(OpcodeArgs); void SALCOp(OpcodeArgs); void PUSHOp(OpcodeArgs); void PUSHREGOp(OpcodeArgs); void PUSHAOp(OpcodeArgs); template void PUSHSegmentOp(OpcodeArgs); void POPOp(OpcodeArgs); void POPAOp(OpcodeArgs); template void POPSegmentOp(OpcodeArgs); void LEAVEOp(OpcodeArgs); void CALLOp(OpcodeArgs); void CALLAbsoluteOp(OpcodeArgs); void CondJUMPOp(OpcodeArgs); void CondJUMPRCXOp(OpcodeArgs); void LoopOp(OpcodeArgs); void JUMPOp(OpcodeArgs); void JUMPAbsoluteOp(OpcodeArgs); template void TESTOp(OpcodeArgs); void MOVSXDOp(OpcodeArgs); void MOVSXOp(OpcodeArgs); void MOVZXOp(OpcodeArgs); template void CMPOp(OpcodeArgs); void SETccOp(OpcodeArgs); void CQOOp(OpcodeArgs); void CDQOp(OpcodeArgs); void XCHGOp(OpcodeArgs); void SAHFOp(OpcodeArgs); void LAHFOp(OpcodeArgs); template void MOVSegOp(OpcodeArgs); void FLAGControlOp(OpcodeArgs); void MOVOffsetOp(OpcodeArgs); void CMOVOp(OpcodeArgs); void CPUIDOp(OpcodeArgs); void XGetBVOp(OpcodeArgs); uint32_t LoadConstantShift(X86Tables::DecodedOp Op, bool Is1Bit); void SHLOp(OpcodeArgs); template void SHLImmediateOp(OpcodeArgs); void SHROp(OpcodeArgs); template void SHRImmediateOp(OpcodeArgs); void SHLDOp(OpcodeArgs); void SHLDImmediateOp(OpcodeArgs); void SHRDOp(OpcodeArgs); void SHRDImmediateOp(OpcodeArgs); template void ASHROp(OpcodeArgs); template void RotateOp(OpcodeArgs); void RCROp1Bit(OpcodeArgs); void RCROp8x1Bit(OpcodeArgs); void RCROp(OpcodeArgs); void RCRSmallerOp(OpcodeArgs); void RCLOp1Bit(OpcodeArgs); void RCLOp(OpcodeArgs); void RCLSmallerOp(OpcodeArgs); template void BTOp(OpcodeArgs); void IMUL1SrcOp(OpcodeArgs); void IMUL2SrcOp(OpcodeArgs); void IMULOp(OpcodeArgs); void STOSOp(OpcodeArgs); void MOVSOp(OpcodeArgs); void CMPSOp(OpcodeArgs); void LODSOp(OpcodeArgs); void SCASOp(OpcodeArgs); void BSWAPOp(OpcodeArgs); void PUSHFOp(OpcodeArgs); void POPFOp(OpcodeArgs); struct CycleCounterPair { Ref CounterLow; Ref CounterHigh; }; CycleCounterPair CycleCounter(); void RDTSCOp(OpcodeArgs); void INCOp(OpcodeArgs); void DECOp(OpcodeArgs); void NEGOp(OpcodeArgs); void DIVOp(OpcodeArgs); void IDIVOp(OpcodeArgs); void BSFOp(OpcodeArgs); void BSROp(OpcodeArgs); void CMPXCHGOp(OpcodeArgs); void CMPXCHGPairOp(OpcodeArgs); void MULOp(OpcodeArgs); void NOTOp(OpcodeArgs); void XADDOp(OpcodeArgs); void PopcountOp(OpcodeArgs); void DAAOp(OpcodeArgs); void DASOp(OpcodeArgs); void AAAOp(OpcodeArgs); void AASOp(OpcodeArgs); void AAMOp(OpcodeArgs); void AADOp(OpcodeArgs); void XLATOp(OpcodeArgs); template void RDRANDOp(OpcodeArgs); enum class Segment { FS, GS, }; template void ReadSegmentReg(OpcodeArgs); template void WriteSegmentReg(OpcodeArgs); void EnterOp(OpcodeArgs); void SGDTOp(OpcodeArgs); void SMSWOp(OpcodeArgs); // SSE void MOVLPOp(OpcodeArgs); void MOVHPDOp(OpcodeArgs); void MOVSDOp(OpcodeArgs); void MOVSSOp(OpcodeArgs); template void VectorALUOp(OpcodeArgs); template void VectorALUROp(OpcodeArgs); template void VectorUnaryOp(OpcodeArgs); template void VectorUnaryDuplicateOp(OpcodeArgs); void MOVQOp(OpcodeArgs); void MOVQMMXOp(OpcodeArgs); template void MOVMSKOp(OpcodeArgs); void MOVMSKOpOne(OpcodeArgs); template void PUNPCKLOp(OpcodeArgs); template void PUNPCKHOp(OpcodeArgs); void PSHUFBOp(OpcodeArgs); template void PSHUFWOp(OpcodeArgs); void PSHUFW8ByteOp(OpcodeArgs); void PSHUFDOp(OpcodeArgs); template void PSRLDOp(OpcodeArgs); template void PSRLI(OpcodeArgs); template void PSLLI(OpcodeArgs); template void PSLL(OpcodeArgs); template void PSRAOp(OpcodeArgs); void PSRLDQ(OpcodeArgs); void PSLLDQ(OpcodeArgs); template void PSRAIOp(OpcodeArgs); void MOVDDUPOp(OpcodeArgs); template void CVTGPR_To_FPR(OpcodeArgs); template void CVTFPR_To_GPR(OpcodeArgs); template void Vector_CVT_Int_To_Float(OpcodeArgs); template void Scalar_CVT_Float_To_Float(OpcodeArgs); template void Vector_CVT_Float_To_Float(OpcodeArgs); template void Vector_CVT_Float_To_Int(OpcodeArgs); void MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs); template void XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs); void MASKMOVOp(OpcodeArgs); void MOVBetweenGPR_FPR(OpcodeArgs); void TZCNT(OpcodeArgs); void LZCNT(OpcodeArgs); template void VFCMPOp(OpcodeArgs); template void SHUFOp(OpcodeArgs); template void PINSROp(OpcodeArgs); void InsertPSOp(OpcodeArgs); template void PExtrOp(OpcodeArgs); template void PSIGN(OpcodeArgs); template void VPSIGN(OpcodeArgs); // BMI1 Ops void ANDNBMIOp(OpcodeArgs); void BEXTRBMIOp(OpcodeArgs); void BLSIBMIOp(OpcodeArgs); void BLSMSKBMIOp(OpcodeArgs); void BLSRBMIOp(OpcodeArgs); // BMI2 Ops void BMI2Shift(OpcodeArgs); void BZHI(OpcodeArgs); void MULX(OpcodeArgs); void PDEP(OpcodeArgs); void PEXT(OpcodeArgs); void RORX(OpcodeArgs); // ADX Ops void ADXOp(OpcodeArgs); // AVX Ops template void AVXVectorALUOp(OpcodeArgs); template void AVXVectorUnaryOp(OpcodeArgs); template void AVXVectorRound(OpcodeArgs); template void AVXScalar_CVT_Float_To_Float(OpcodeArgs); template void AVXVector_CVT_Float_To_Int(OpcodeArgs); template void AVXVector_CVT_Int_To_Float(OpcodeArgs); template void VectorScalarInsertALUOp(OpcodeArgs); template void AVXVectorScalarInsertALUOp(OpcodeArgs); template void VectorScalarUnaryInsertALUOp(OpcodeArgs); template void AVXVectorScalarUnaryInsertALUOp(OpcodeArgs); template void AVXVector_CVT_Float_To_Float(OpcodeArgs); void InsertMMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs); template void InsertCVTGPR_To_FPR(OpcodeArgs); template void AVXInsertCVTGPR_To_FPR(OpcodeArgs); template void InsertScalar_CVT_Float_To_Float(OpcodeArgs); template void AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs); template void InsertScalarRound(OpcodeArgs); template void AVXInsertScalarRound(OpcodeArgs); template void InsertScalarFCMPOp(OpcodeArgs); template void AVXInsertScalarFCMPOp(OpcodeArgs); template void AVXCVTGPR_To_FPR(OpcodeArgs); template void AVXVFCMPOp(OpcodeArgs); template void VADDSUBPOp(OpcodeArgs); void VAESDecOp(OpcodeArgs); void VAESDecLastOp(OpcodeArgs); void VAESEncOp(OpcodeArgs); void VAESEncLastOp(OpcodeArgs); void VANDNOp(OpcodeArgs); void VBLENDPDOp(OpcodeArgs); void VPBLENDDOp(OpcodeArgs); void VPBLENDWOp(OpcodeArgs); template void VBROADCASTOp(OpcodeArgs); template void VDPPOp(OpcodeArgs); void VEXTRACT128Op(OpcodeArgs); template void VHADDPOp(OpcodeArgs); template void VHSUBPOp(OpcodeArgs); void VINSERTOp(OpcodeArgs); void VINSERTPSOp(OpcodeArgs); template void VMASKMOVOp(OpcodeArgs); void VMOVHPOp(OpcodeArgs); void VMOVLPOp(OpcodeArgs); void VMOVDDUPOp(OpcodeArgs); void VMOVSHDUPOp(OpcodeArgs); void VMOVSLDUPOp(OpcodeArgs); void VMOVSDOp(OpcodeArgs); void VMOVSSOp(OpcodeArgs); void VMOVAPS_VMOVAPDOp(OpcodeArgs); void VMOVUPS_VMOVUPDOp(OpcodeArgs); void VMPSADBWOp(OpcodeArgs); template void VPACKSSOp(OpcodeArgs); template void VPACKUSOp(OpcodeArgs); void VPALIGNROp(OpcodeArgs); void VPCMPESTRIOp(OpcodeArgs); void VPCMPESTRMOp(OpcodeArgs); void VPCMPISTRIOp(OpcodeArgs); void VPCMPISTRMOp(OpcodeArgs); void VPERM2Op(OpcodeArgs); void VPERMDOp(OpcodeArgs); void VPERMQOp(OpcodeArgs); template void VPERMILImmOp(OpcodeArgs); template void VPERMILRegOp(OpcodeArgs); void VPHADDSWOp(OpcodeArgs); template void VPHSUBOp(OpcodeArgs); void VPHSUBSWOp(OpcodeArgs); void VPINSRBOp(OpcodeArgs); void VPINSRDQOp(OpcodeArgs); void VPINSRWOp(OpcodeArgs); void VPMADDUBSWOp(OpcodeArgs); void VPMADDWDOp(OpcodeArgs); template void VPMASKMOVOp(OpcodeArgs); void VPMULHRSWOp(OpcodeArgs); template void VPMULHWOp(OpcodeArgs); template void VPMULLOp(OpcodeArgs); void VPSADBWOp(OpcodeArgs); void VPSHUFBOp(OpcodeArgs); template void VPSHUFWOp(OpcodeArgs); template void VPSLLOp(OpcodeArgs); void VPSLLDQOp(OpcodeArgs); template void VPSLLIOp(OpcodeArgs); void VPSLLVOp(OpcodeArgs); template void VPSRAOp(OpcodeArgs); template void VPSRAIOp(OpcodeArgs); void VPSRAVDOp(OpcodeArgs); void VPSRLVOp(OpcodeArgs); template void VPSRLDOp(OpcodeArgs); void VPSRLDQOp(OpcodeArgs); template void VPUNPCKHOp(OpcodeArgs); template void VPUNPCKLOp(OpcodeArgs); template void VPSRLIOp(OpcodeArgs); template void VSHUFOp(OpcodeArgs); template void VTESTPOp(OpcodeArgs); void VZEROOp(OpcodeArgs); // X87 Ops Ref ReconstructFSW(); // Returns new x87 stack top from FSW. Ref ReconstructX87StateFromFSW(Ref FSW); template void FLD(OpcodeArgs); template void FLD_Const(OpcodeArgs); void FBLD(OpcodeArgs); void FBSTP(OpcodeArgs); void FILD(OpcodeArgs); template void FST(OpcodeArgs); void FST(OpcodeArgs); template void FIST(OpcodeArgs); enum class OpResult { RES_ST0, RES_STI, }; template void FADD(OpcodeArgs); template void FMUL(OpcodeArgs); template void FDIV(OpcodeArgs); template void FSUB(OpcodeArgs); void FCHS(OpcodeArgs); void FABS(OpcodeArgs); void FTST(OpcodeArgs); void FRNDINT(OpcodeArgs); void FXTRACT(OpcodeArgs); void FNINIT(OpcodeArgs); template void X87UnaryOp(OpcodeArgs); template void X87BinaryOp(OpcodeArgs); template void X87ModifySTP(OpcodeArgs); void X87SinCos(OpcodeArgs); void X87FYL2X(OpcodeArgs); void X87TAN(OpcodeArgs); void X87ATAN(OpcodeArgs); void X87LDENV(OpcodeArgs); void X87FLDCW(OpcodeArgs); void X87FNSTENV(OpcodeArgs); void X87FSTCW(OpcodeArgs); void X87LDSW(OpcodeArgs); void X87FNSTSW(OpcodeArgs); void X87FNSAVE(OpcodeArgs); void X87FRSTOR(OpcodeArgs); void X87FXAM(OpcodeArgs); void X87FCMOV(OpcodeArgs); void X87EMMS(OpcodeArgs); void X87FFREE(OpcodeArgs); void FXCH(OpcodeArgs); enum class FCOMIFlags { FLAGS_X87, FLAGS_RFLAGS, }; template void FCOMI(OpcodeArgs); // F64 X87 Ops template void FLDF64(OpcodeArgs); template void FLDF64_Const(OpcodeArgs); void FBLDF64(OpcodeArgs); void FBSTPF64(OpcodeArgs); void FILDF64(OpcodeArgs); template void FSTF64(OpcodeArgs); void FSTF64(OpcodeArgs); template void FISTF64(OpcodeArgs); template void FADDF64(OpcodeArgs); template void FMULF64(OpcodeArgs); template void FDIVF64(OpcodeArgs); template void FSUBF64(OpcodeArgs); void FCHSF64(OpcodeArgs); void FABSF64(OpcodeArgs); void FTSTF64(OpcodeArgs); void FRNDINTF64(OpcodeArgs); void FXTRACTF64(OpcodeArgs); void FNINITF64(OpcodeArgs); void FSQRTF64(OpcodeArgs); template void X87UnaryOpF64(OpcodeArgs); template void X87BinaryOpF64(OpcodeArgs); void X87SinCosF64(OpcodeArgs); void X87FLDCWF64(OpcodeArgs); void X87FYL2XF64(OpcodeArgs); void X87TANF64(OpcodeArgs); void X87ATANF64(OpcodeArgs); void X87FNSAVEF64(OpcodeArgs); void X87FRSTORF64(OpcodeArgs); void X87FXAMF64(OpcodeArgs); void X87LDENVF64(OpcodeArgs); template void FCOMIF64(OpcodeArgs); void FXSaveOp(OpcodeArgs); void FXRStoreOp(OpcodeArgs); Ref XSaveBase(X86Tables::DecodedOp Op); void XSaveOp(OpcodeArgs); void PAlignrOp(OpcodeArgs); template void UCOMISxOp(OpcodeArgs); void LDMXCSR(OpcodeArgs); void STMXCSR(OpcodeArgs); template void PACKUSOp(OpcodeArgs); template void PACKSSOp(OpcodeArgs); template void PMULLOp(OpcodeArgs); template void MOVQ2DQ(OpcodeArgs); template void ADDSUBPOp(OpcodeArgs); void PFNACCOp(OpcodeArgs); void PFPNACCOp(OpcodeArgs); void PSWAPDOp(OpcodeArgs); template void VPFCMPOp(OpcodeArgs); void PI2FWOp(OpcodeArgs); void PF2IWOp(OpcodeArgs); void PMULHRWOp(OpcodeArgs); void PMADDWD(OpcodeArgs); void PMADDUBSW(OpcodeArgs); template void PMULHW(OpcodeArgs); void PMULHRSW(OpcodeArgs); void MOVBEOp(OpcodeArgs); template void HSUBP(OpcodeArgs); template void PHSUB(OpcodeArgs); void PHADDS(OpcodeArgs); void PHSUBS(OpcodeArgs); void CLWB(OpcodeArgs); void CLFLUSHOPT(OpcodeArgs); void LoadFenceOrXRSTOR(OpcodeArgs); void MemFenceOrXSAVEOPT(OpcodeArgs); void StoreFenceOrCLFlush(OpcodeArgs); void CLZeroOp(OpcodeArgs); void RDTSCPOp(OpcodeArgs); void RDPIDOp(OpcodeArgs); template void Prefetch(OpcodeArgs); void PSADBW(OpcodeArgs); Ref BitwiseAtLeastTwo(Ref A, Ref B, Ref C); void SHA1NEXTEOp(OpcodeArgs); void SHA1MSG1Op(OpcodeArgs); void SHA1MSG2Op(OpcodeArgs); void SHA1RNDS4Op(OpcodeArgs); void SHA256MSG1Op(OpcodeArgs); void SHA256MSG2Op(OpcodeArgs); void SHA256RNDS2Op(OpcodeArgs); void AESImcOp(OpcodeArgs); void AESEncOp(OpcodeArgs); void AESEncLastOp(OpcodeArgs); void AESDecOp(OpcodeArgs); void AESDecLastOp(OpcodeArgs); void AESKeyGenAssist(OpcodeArgs); template void ExtendVectorElements(OpcodeArgs); template void VectorRound(OpcodeArgs); template void VectorBlend(OpcodeArgs); template void VectorVariableBlend(OpcodeArgs); void PTestOp(OpcodeArgs); void PHMINPOSUWOp(OpcodeArgs); template void DPPOp(OpcodeArgs); void MPSADBWOp(OpcodeArgs); void PCLMULQDQOp(OpcodeArgs); void VPCLMULQDQOp(OpcodeArgs); void CRC32(OpcodeArgs); void BreakOp(OpcodeArgs, FEXCore::IR::BreakDefinition BreakDefinition); void UnimplementedOp(OpcodeArgs); void PermissionRestrictedOp(OpcodeArgs); // AVX 128-bit operations Ref AVX128_LoadXMMRegister(uint32_t XMM, bool High); void AVX128_StoreXMMRegister(uint32_t XMM, const Ref Src, bool High); struct RefPair { Ref Low, High; }; RefPair AVX128_LoadSource_WithOpSize(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags, bool NeedsHigh, MemoryAccessType AccessType = MemoryAccessType::DEFAULT); void AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::DecodedOp Op, const FEXCore::X86Tables::DecodedOperand& Operand, const RefPair Src, MemoryAccessType AccessType = MemoryAccessType::DEFAULT); void InstallAVX128Handlers(); void AVX128_VMOVScalarImpl(OpcodeArgs, size_t ElementSize); void AVX128_VectorALUImpl(OpcodeArgs, IROps IROp, size_t ElementSize); void AVX128_VectorUnaryImpl(OpcodeArgs, IROps IROp, size_t ElementSize); void AVX128_VectorUnaryImpl(OpcodeArgs, size_t SrcSize, size_t ElementSize, std::function Helper); void AVX128_VectorBinaryImpl(OpcodeArgs, size_t SrcSize, size_t ElementSize, std::function Helper); void AVX128_VMOVAPS(OpcodeArgs); void AVX128_VMOVSD(OpcodeArgs); void AVX128_VMOVSS(OpcodeArgs); template void AVX128_VectorALU(OpcodeArgs); template void AVX128_VectorUnary(OpcodeArgs); void AVX128_VZERO(OpcodeArgs); void AVX128_MOVVectorNT(OpcodeArgs); void AVX128_MOVQ(OpcodeArgs); void AVX128_VMOVLP(OpcodeArgs); void AVX128_VMOVHP(OpcodeArgs); void AVX128_VMOVDDUP(OpcodeArgs); void AVX128_VMOVSLDUP(OpcodeArgs); void AVX128_VMOVSHDUP(OpcodeArgs); template void AVX128_VBROADCAST(OpcodeArgs); // End of AVX 128-bit implementation void InvalidOp(OpcodeArgs); void SetPackedRFLAG(bool Lower8, Ref Src); Ref GetPackedRFLAG(uint32_t FlagsMask = ~0U); void SetMultiblock(bool _Multiblock) { Multiblock = _Multiblock; } static inline constexpr unsigned IndexNZCV(unsigned BitOffset) { switch (BitOffset) { case FEXCore::X86State::RFLAG_OF_RAW_LOC: return 28; case FEXCore::X86State::RFLAG_CF_RAW_LOC: return 29; case FEXCore::X86State::RFLAG_ZF_RAW_LOC: return 30; case FEXCore::X86State::RFLAG_SF_RAW_LOC: return 31; default: FEX_UNREACHABLE; } } protected: void SaveNZCV(IROps Op = OP_DUMMY) override { /* Some opcodes are conservatively marked as clobbering flags, but in fact * do not clobber flags in certain conditions. Check for that here as an * optimization. */ switch (Op) { case OP_VFMINSCALARINSERT: case OP_VFMAXSCALARINSERT: /* On AFP platforms, becomes fmin/fmax and preserves NZCV. Otherwise * becomes fcmp and clobbers. */ if (CTX->HostFeatures.SupportsAFP) { return; } break; default: break; } // Invariant: When executing instructions that clobber NZCV, the flags must // be resident in a GPR, which is equivalent to CachedNZCV != nullptr. Get // the NZCV which fills the cache if necessary. if (CachedNZCV == nullptr) { GetNZCV(); } // Assume we'll need a reload. NZCVDirty = true; } private: struct JumpTargetInfo { Ref BlockEntry; bool HaveEmitted; }; FEXCore::Context::ContextImpl* CTX {}; constexpr static unsigned FullNZCVMask = (1U << FEXCore::X86State::RFLAG_CF_RAW_LOC) | (1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC) | (1U << FEXCore::X86State::RFLAG_SF_RAW_LOC) | (1U << FEXCore::X86State::RFLAG_OF_RAW_LOC); static bool ContainsNZCV(unsigned BitMask) { return (BitMask & FullNZCVMask) != 0; } static bool IsNZCV(unsigned BitOffset) { return BitOffset < 32 && ContainsNZCV(1U << BitOffset); } Ref CachedNZCV {}; bool NZCVDirty {}; uint32_t PossiblySetNZCVBits {}; fextl::map JumpTargets; bool HandledLock {false}; bool DecodeFailure {false}; bool NeedsBlockEnd {false}; // Used during new op bringup bool ShouldDump {false}; void ALUOpImpl(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx); // Opcode helpers for generalizing behavior across VEX and non-VEX variants. Ref ADDSUBPOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2); void AVXVectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize); void AVXVectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize); template void AVXVectorVariableBlend(OpcodeArgs); void AVXVariableShiftImpl(OpcodeArgs, IROps IROp); Ref AESKeyGenAssistImpl(OpcodeArgs); Ref CVTGPR_To_FPRImpl(OpcodeArgs, size_t DstElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op); Ref DPPOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2, const X86Tables::DecodedOperand& Imm, size_t ElementSize); Ref VDPPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2, const X86Tables::DecodedOperand& Imm); Ref ExtendVectorElementsImpl(OpcodeArgs, size_t ElementSize, size_t DstElementSize, bool Signed); Ref HSUBPOpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op); Ref InsertPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2, const X86Tables::DecodedOperand& Imm); Ref MPSADBWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op, const X86Tables::DecodedOperand& ImmOp); Ref PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2, const X86Tables::DecodedOperand& Imm, bool IsAVX); void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask); Ref PHADDSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2); Ref PHMINPOSUWOpImpl(OpcodeArgs); Ref PHSUBOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2, size_t ElementSize); Ref PHSUBSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op); Ref PINSROpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op, const X86Tables::DecodedOperand& Imm); Ref PMADDWDOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2); Ref PMADDUBSWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op); Ref PMULHRSWOpImpl(OpcodeArgs, Ref Src1, Ref Src2); Ref PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2); Ref PMULLOpImpl(OpcodeArgs, size_t ElementSize, bool Signed, Ref Src1, Ref Src2); Ref PSADBWOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op); Ref PSHUFBOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2); Ref PSIGNImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2); Ref PSLLIImpl(OpcodeArgs, size_t ElementSize, Ref Src, uint64_t Shift); Ref PSLLImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref ShiftVec); Ref PSRAOpImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref ShiftVec); Ref PSRLDOpImpl(OpcodeArgs, size_t ElementSize, Ref Src, Ref ShiftVec); Ref SHUFOpImpl(OpcodeArgs, size_t ElementSize, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2, const X86Tables::DecodedOperand& Imm); void VMASKMOVOpImpl(OpcodeArgs, size_t ElementSize, size_t DataSize, bool IsStore, const X86Tables::DecodedOperand& MaskOp, const X86Tables::DecodedOperand& DataOp); void MOVScalarOpImpl(OpcodeArgs, size_t ElementSize); void VMOVScalarOpImpl(OpcodeArgs, size_t ElementSize); Ref VFCMPOpImpl(OpcodeArgs, size_t ElementSize, Ref Src1, Ref Src2, uint8_t CompType); void VTESTOpImpl(OpcodeArgs, size_t ElementSize); void VectorALUOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize); void VectorALUROpImpl(OpcodeArgs, IROps IROp, size_t ElementSize); void VectorUnaryOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize); void VectorUnaryDuplicateOpImpl(OpcodeArgs, IROps IROp, size_t ElementSize); // x86 ALU scalar operations operate in three different ways // - AVX512: Writemask shenanigans that we don't care about. // - AVX/VEX: Two source // - Example 32bit VADDSS Dest, Src1, Src2 // - Dest[31:0] = Src1[31:0] + Src2[31:0] // - Dest[127:32] = Src1[127:32] // - SSE: Scalar operation inserts in to the low bits, upper bits completely unaffected. // - Example 32bit ADDSS Dest, Src // - Dest[31:0] = Dest[31:0] + Src[31:0] // - Dest[{256,128}:32] = (Unmodified) Ref VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IROp, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op, bool ZeroUpperBits); Ref VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IROps IROp, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op, bool ZeroUpperBits); Ref InsertCVTGPR_To_FPRImpl(OpcodeArgs, size_t DstSize, size_t DstElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op, bool ZeroUpperBits); Ref InsertScalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstSize, size_t DstElementSize, size_t SrcElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op, bool ZeroUpperBits); Ref InsertScalarRoundImpl(OpcodeArgs, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op, uint64_t Mode, bool ZeroUpperBits); Ref InsertScalarFCMPOpImpl(OpcodeArgs, size_t DstSize, size_t ElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op, uint8_t CompType, bool ZeroUpperBits); Ref VectorRoundImpl(OpcodeArgs, size_t ElementSize, Ref Src, uint64_t Mode); Ref Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op); void Vector_CVT_Float_To_FloatImpl(OpcodeArgs, size_t DstElementSize, size_t SrcElementSize, bool IsAVX); Ref Vector_CVT_Float_To_IntImpl(OpcodeArgs, size_t SrcElementSize, bool Narrow, bool HostRoundingMode); Ref Vector_CVT_Int_To_FloatImpl(OpcodeArgs, size_t SrcElementSize, bool Widen); void XSaveOpImpl(OpcodeArgs); void SaveX87State(OpcodeArgs, Ref MemBase); void SaveSSEState(Ref MemBase); void SaveMXCSRState(Ref MemBase); void SaveAVXState(Ref MemBase); void XRstorOpImpl(OpcodeArgs); void RestoreX87State(Ref MemBase); void RestoreSSEState(Ref MemBase); void RestoreMXCSRState(Ref MXCSR); void RestoreAVXState(Ref MemBase); void DefaultX87State(OpcodeArgs); void DefaultSSEState(); void DefaultAVXState(); Ref GetMXCSR(); #undef OpcodeArgs Ref AppendSegmentOffset(Ref Value, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false); Ref GetSegment(uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false); void UpdatePrefixFromSegment(Ref Segment, uint32_t SegmentReg); Ref LoadGPRRegister(uint32_t GPR, int8_t Size = -1, uint8_t Offset = 0, bool AllowUpperGarbage = false); Ref LoadXMMRegister(uint32_t XMM); void StoreGPRRegister(uint32_t GPR, const Ref Src, int8_t Size = -1, uint8_t Offset = 0); void StoreXMMRegister(uint32_t XMM, const Ref Src); Ref GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0); AddressMode AddSegmentToAddress(AddressMode A, uint32_t Flags); Ref LoadEffectiveAddress(AddressMode A, bool AllowUpperGarbage = false); AddressMode SelectAddressMode(AddressMode A, bool AtomicTSO, bool Vector, unsigned AccessSize); bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) { // Literals are immediates as sources but memory addresses as destinations. return !(Load && Operand.IsLiteral()) && !Operand.IsGPR(); } bool IsNonTSOReg(MemoryAccessType Access, uint8_t Reg) { return Access == MemoryAccessType::DEFAULT && Reg == X86State::REG_RSP; } AddressMode DecodeAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, MemoryAccessType AccessType, bool IsLoad); Ref LoadSource(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags, const LoadSourceOptions& Options = {}); Ref LoadSource_WithOpSize(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint8_t OpSize, uint32_t Flags, const LoadSourceOptions& Options = {}); void StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const FEXCore::X86Tables::DecodedOperand& Operand, const Ref Src, uint8_t OpSize, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT); void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const FEXCore::X86Tables::DecodedOperand& Operand, const Ref Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT); void StoreResult(FEXCore::IR::RegisterClassType Class, FEXCore::X86Tables::DecodedOp Op, const Ref Src, int8_t Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT); // In several instances, it's desirable to get a base address with the segment offset // applied to it. This pulls all the common-case appending into a single set of functions. [[nodiscard]] Ref MakeSegmentAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint8_t OpSize) { Ref Mem = LoadSource_WithOpSize(GPRClass, Op, Operand, OpSize, Op->Flags, {.LoadData = false}); return AppendSegmentOffset(Mem, Op->Flags); } [[nodiscard]] Ref MakeSegmentAddress(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand) { return MakeSegmentAddress(Op, Operand, GetSrcSize(Op)); } [[nodiscard]] Ref MakeSegmentAddress(X86State::X86Reg Reg, uint32_t Flags, uint32_t DefaultPrefix = 0, bool Override = false) { Ref Address = LoadGPRRegister(Reg); return AppendSegmentOffset(Address, Flags, DefaultPrefix, Override); } constexpr OpSize GetGuestVectorLength() const { return CTX->HostFeatures.SupportsSVE256 ? OpSize::i256Bit : OpSize::i128Bit; } [[nodiscard]] static uint32_t GPROffset(X86State::X86Reg reg) { LOGMAN_THROW_AA_FMT(reg <= X86State::X86Reg::REG_R15, "Invalid reg used"); return static_cast(offsetof(Core::CPUState, gregs[static_cast(reg)])); } [[nodiscard]] static uint32_t MMBaseOffset() { return static_cast(offsetof(Core::CPUState, mm[0][0])); } [[nodiscard]] uint8_t GetDstSize(X86Tables::DecodedOp Op) const; [[nodiscard]] uint8_t GetSrcSize(X86Tables::DecodedOp Op) const; [[nodiscard]] uint32_t GetDstBitSize(X86Tables::DecodedOp Op) const; [[nodiscard]] uint32_t GetSrcBitSize(X86Tables::DecodedOp Op) const; [[nodiscard]] IR::OpSize OpSizeFromDst(X86Tables::DecodedOp Op) const { return IR::SizeToOpSize(GetDstSize(Op)); } [[nodiscard]] IR::OpSize OpSizeFromSrc(X86Tables::DecodedOp Op) const { return IR::SizeToOpSize(GetSrcSize(Op)); } static inline constexpr unsigned NZCVIndexMask(unsigned BitMask) { unsigned NZCVMask {}; if (BitMask & (1U << FEXCore::X86State::RFLAG_OF_RAW_LOC)) { NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_OF_RAW_LOC); } if (BitMask & (1U << FEXCore::X86State::RFLAG_CF_RAW_LOC)) { NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC); } if (BitMask & (1U << FEXCore::X86State::RFLAG_ZF_RAW_LOC)) { NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_ZF_RAW_LOC); } if (BitMask & (1U << FEXCore::X86State::RFLAG_SF_RAW_LOC)) { NZCVMask |= 1U << IndexNZCV(FEXCore::X86State::RFLAG_SF_RAW_LOC); } return NZCVMask; } // Set flag tracking to prepare for an operation that directly writes NZCV. If // some bits are known to be zeroed, the PossiblySetNZCVBits mask can be // passed. Otherwise, it defaults to assuming all bits may be set after // (this is conservative). void HandleNZCVWrite(uint32_t _PossiblySetNZCVBits = ~0) { InvalidateDeferredFlags(); CachedNZCV = nullptr; PossiblySetNZCVBits = _PossiblySetNZCVBits; NZCVDirty = false; } // Set flag tracking to prepare for a read-modify-write operation on NZCV. void HandleNZCV_RMW(uint32_t _PossiblySetNZCVBits = ~0) { CalculateDeferredFlags(); if (NZCVDirty && CachedNZCV) { _StoreNZCV(CachedNZCV); } HandleNZCVWrite(_PossiblySetNZCVBits); } // Special case of the above where we are known to zero C/V void HandleNZ00Write() { HandleNZCVWrite((1u << 31) | (1u << 30)); } Ref GetNZCV() { if (!CachedNZCV) { CachedNZCV = _LoadNZCV(); } return CachedNZCV; } void SetNZCV(Ref Value) { CachedNZCV = Value; NZCVDirty = true; } void ZeroNZCV() { CachedNZCV = _Constant(0); PossiblySetNZCVBits = 0; NZCVDirty = true; } void SetNZ_ZeroCV(unsigned SrcSize, Ref Res) { HandleNZ00Write(); _TestNZ(IR::SizeToOpSize(SrcSize), Res, Res); } void InsertNZCV(unsigned BitOffset, Ref Value, signed FlagOffset, bool MustMask) { signed Bit = IndexNZCV(BitOffset); // If NZCV is not dirty, we always want to use rmif, it's 1 instruction to // implement this. But if NZCV is dirty, it might still be cheaper to copy // the GPR flags to NZCV and rmif. This is a heuristic for cases where we // expect that 2 instruction sequence to be a win (versus something like // bfe+mov+bfi+mov which can happen with our RA..). It's not totally // conservative but it's pretty good in practice. bool PreferRmif = !NZCVDirty || FlagOffset || MustMask || (PossiblySetNZCVBits & (1u << Bit)); if (CTX->HostFeatures.SupportsFlagM && PreferRmif) { // Update NZCV if (NZCVDirty && CachedNZCV) { _StoreNZCV(CachedNZCV); } CachedNZCV = nullptr; NZCVDirty = false; // Insert as NZCV. signed RmifBit = Bit - 28; _RmifNZCV(Value, (64 + FlagOffset - RmifBit) % 64, 1u << RmifBit); CachedNZCV = nullptr; } else { // Insert as GPR if (FlagOffset || MustMask) { Value = _Bfe(OpSize::i64Bit, 1, FlagOffset, Value); } if (PossiblySetNZCVBits == 0) { SetNZCV(_Lshl(OpSize::i64Bit, Value, _Constant(Bit))); } else if ((PossiblySetNZCVBits & (1u << Bit)) == 0) { SetNZCV(_Orlshl(OpSize::i32Bit, GetNZCV(), Value, Bit)); } else { SetNZCV(_Bfi(OpSize::i32Bit, 1, Bit, GetNZCV(), Value)); } } PossiblySetNZCVBits |= (1u << Bit); } void CarryInvert() { unsigned Bit = IndexNZCV(FEXCore::X86State::RFLAG_CF_RAW_LOC); if (CTX->HostFeatures.SupportsFlagM && !NZCVDirty) { // Invert as NZCV. _CarryInvert(); CachedNZCV = nullptr; } else { // Invert as a GPR SetNZCV(_Xor(OpSize::i32Bit, GetNZCV(), _Constant(1u << Bit))); } PossiblySetNZCVBits |= 1u << Bit; } template void SetRFLAG(Ref Value, unsigned ValueOffset = 0, bool MustMask = false) { SetRFLAG(Value, BitOffset, ValueOffset, MustMask); } void SetRFLAG(Ref Value, unsigned BitOffset, unsigned ValueOffset = 0, bool MustMask = false) { if (IsNZCV(BitOffset)) { InsertNZCV(BitOffset, Value, ValueOffset, MustMask); } else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) { _StoreRegister(Value, Core::CPUState::PF_AS_GREG, GPRClass, CTX->GetGPRSize()); } else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) { _StoreRegister(Value, Core::CPUState::AF_AS_GREG, GPRClass, CTX->GetGPRSize()); } else { if (ValueOffset || MustMask) { Value = _Bfe(OpSize::i32Bit, 1, ValueOffset, Value); } // For DF, we need to transform 0/1 into 1/-1 if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) { Value = _SubShift(OpSize::i64Bit, _Constant(1), Value, ShiftType::LSL, 1); } _StoreFlag(Value, BitOffset); } } void SetAF(unsigned Constant) { // AF is stored in bit 4 of the AF flag byte, with garbage in the other // bits. This allows us to defer the extract in the usual case. When it is // read, bit 4 is extracted. In order to write a constant value of AF, that // means we need to left-shift here to compensate. SetRFLAG(_Constant(Constant << 4)); } void ZeroPF_AF(); void InvalidateAF() { _InvalidateFlags((1u << X86State::RFLAG_AF_RAW_LOC)); } void InvalidatePF_AF() { _InvalidateFlags((1u << X86State::RFLAG_PF_RAW_LOC) | (1u << X86State::RFLAG_AF_RAW_LOC)); } CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) { switch (BitOffset) { case FEXCore::X86State::RFLAG_SF_RAW_LOC: return Invert ? CondClassType {COND_PL} : CondClassType {COND_MI}; case FEXCore::X86State::RFLAG_ZF_RAW_LOC: return Invert ? CondClassType {COND_NEQ} : CondClassType {COND_EQ}; case FEXCore::X86State::RFLAG_CF_RAW_LOC: return Invert ? CondClassType {COND_ULT} : CondClassType {COND_UGE}; case FEXCore::X86State::RFLAG_OF_RAW_LOC: return Invert ? CondClassType {COND_FNU} : CondClassType {COND_FU}; default: FEX_UNREACHABLE; } } void FlushRegisterCache() { CalculateDeferredFlags(); } Ref GetRFLAG(unsigned BitOffset, bool Invert = false) { if (IsNZCV(BitOffset)) { if (!(PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset)))) { return _Constant(Invert ? 1 : 0); } else if (NZCVDirty) { auto Value = _Bfe(OpSize::i32Bit, 1, IndexNZCV(BitOffset), GetNZCV()); if (Invert) { return _Xor(OpSize::i32Bit, Value, _Constant(1)); } else { return Value; } } else { return _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(BitOffset, Invert), _Constant(1), _Constant(0)); } } else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) { return _LoadRegister(Core::CPUState::PF_AS_GREG, GPRClass, CTX->GetGPRSize()); } else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) { return _LoadRegister(Core::CPUState::AF_AS_GREG, GPRClass, CTX->GetGPRSize()); } else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) { // Recover the sign bit, it is the logical DF value return _Lshr(OpSize::i64Bit, _LoadDF(), _Constant(63)); } else { return _LoadFlag(BitOffset); } } // Returns (DF ? -Size : Size) Ref LoadDir(const unsigned Size) { auto Dir = _LoadDF(); auto Shift = FEXCore::ilog2(Size); if (Shift) { return _Lshl(IR::SizeToOpSize(CTX->GetGPRSize()), Dir, _Constant(Shift)); } else { return Dir; } } // Returns DF ? (X - Size) : (X + Size) Ref OffsetByDir(Ref X, const unsigned Size) { auto Shift = FEXCore::ilog2(Size); return _AddShift(OpSize::i64Bit, X, _LoadDF(), ShiftType::LSL, Shift); } // Set SSE comparison flags based on the result set by Arm FCMP. This converts // NZCV from the Arm representation to an eXternal representation that's // totally not a euphemism for x86 or anything, nuh-uh. void ConvertNZCVToSSE() { if (CTX->HostFeatures.SupportsFlagM2) { LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp"); // We need to set PF according to the unordered flag. We'd rather do this // after axflag, since some impls fuse fcmp+axflag, so we want to do this // after. We can recover "unordered" after axflag as (Z && !C), but // there's no condition code for this so it would take 2 instructions // instead of one, which seems worse than doing 1 op before and breaking // the fusion. // // We set PF to unordered (V), but our PF representation is inverted so we // actually set to !V. This is one instruction with the VC cond code. Ref PFInvert = _NZCVSelect(OpSize::i32Bit, CondClassType {COND_FNU}, _Constant(1), _Constant(0)); SetRFLAG(PFInvert); // For the rest, this one weird a64 instruction maps exactly to what x86 // needs. What a coincidence! _AXFlag(); PossiblySetNZCVBits = ~0; // It does assume we invert CF internally, which is still TODO for us. For // now, add a cfinv to deal. Hopefully we delete this later. CarryInvert(); } else { Ref Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC); Ref C_inv = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true); Ref V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC); // We want to zero SF/OF, and then set CF/ZF. Zeroing up front lets us do // this all with shifted-or's on non-flagm platforms. ZeroNZCV(); SetRFLAG(_Or(OpSize::i32Bit, C_inv, V)); SetRFLAG(_Or(OpSize::i32Bit, Z, V)); // Note that we store PF inverted. SetRFLAG(_Xor(OpSize::i32Bit, V, _Constant(1))); } } // Set x87 comparison flags based on the result set by Arm FCMP. Clobbers // NZCV on flagm2 platforms. void ConvertNZCVToX87() { Ref V = GetRFLAG(FEXCore::X86State::RFLAG_OF_RAW_LOC); if (CTX->HostFeatures.SupportsFlagM2) { LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp"); // Convert to x86 flags, saves us from or'ing after. _AXFlag(); PossiblySetNZCVBits = ~0; // Copy the values. CF is inverted from the axflag result, ZF is as-is. SetRFLAG(GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC, true)); SetRFLAG(GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC)); } else { Ref Z = GetRFLAG(FEXCore::X86State::RFLAG_ZF_RAW_LOC); Ref N = GetRFLAG(FEXCore::X86State::RFLAG_SF_RAW_LOC); SetRFLAG(_Or(OpSize::i32Bit, N, V)); SetRFLAG(_Or(OpSize::i32Bit, Z, V)); } SetRFLAG(_Constant(0)); SetRFLAG(V); } // Helper to store a variable shift and calculate its flags for a variable // shift, with correct PF handling. void HandleShift(X86Tables::DecodedOp Op, Ref Result, Ref Dest, ShiftType Shift, Ref Src) { auto OldPF = GetRFLAG(X86State::RFLAG_PF_RAW_LOC); HandleNZCV_RMW(); CalculatePF(_ShiftFlags(OpSizeFromSrc(Op), Result, Dest, Shift, Src, OldPF)); StoreResult(GPRClass, Op, Result, -1); } std::pair ExtractPair(OpSize Size, Ref Pair) { // Extract high first. This is a hack to improve coalescing. Ref Hi = _ExtractElementPair(Size, Pair, 1); Ref Lo = _ExtractElementPair(Size, Pair, 0); return std::make_pair(Lo, Hi); } // Helper to derive Dest by a given builder-using Expression with the opcode // replaced with NewOp. Useful for generic building code. Not safe in general. // but does the right handling of ImplicitFlagClobber at least and must be // used instead of raw Op mutation. #define DeriveOp(Dest, NewOp, Expr) \ if (ImplicitFlagClobber(NewOp)) SaveNZCV(NewOp); \ auto Dest = (Expr); \ Dest.first->Header.Op = (NewOp) // Named constant cache for the current block. // Different arrays for sizes 1,2,4,8,16,32. Ref CachedNamedVectorConstants[FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_MAX][6] {}; struct IndexNamedVectorMapKey { uint32_t Index {}; FEXCore::IR::IndexNamedVectorConstant NamedIndexedConstant; uint8_t log2_size_in_bytes {}; uint16_t _pad {}; bool operator==(const IndexNamedVectorMapKey&) const = default; }; struct IndexNamedVectorMapKeyHasher { std::size_t operator()(const IndexNamedVectorMapKey& k) const noexcept { return XXH3_64bits(&k, sizeof(k)); } }; fextl::unordered_map CachedIndexedNamedVectorConstants; // Load and cache a named vector constant. Ref LoadAndCacheNamedVectorConstant(uint8_t Size, FEXCore::IR::NamedVectorConstant NamedConstant) { auto log2_size_bytes = FEXCore::ilog2(Size); if (CachedNamedVectorConstants[NamedConstant][log2_size_bytes]) { return CachedNamedVectorConstants[NamedConstant][log2_size_bytes]; } auto Constant = _LoadNamedVectorConstant(Size, NamedConstant); CachedNamedVectorConstants[NamedConstant][log2_size_bytes] = Constant; return Constant; } Ref LoadAndCacheIndexedNamedVectorConstant(uint8_t Size, FEXCore::IR::IndexNamedVectorConstant NamedIndexedConstant, uint32_t Index) { IndexNamedVectorMapKey Key { .Index = Index, .NamedIndexedConstant = NamedIndexedConstant, .log2_size_in_bytes = FEXCore::ilog2(Size), }; auto it = CachedIndexedNamedVectorConstants.find(Key); if (it != CachedIndexedNamedVectorConstants.end()) { return it->second; } auto Constant = _LoadNamedVectorIndexedConstant(Size, NamedIndexedConstant, Index); CachedIndexedNamedVectorConstants.insert_or_assign(Key, Constant); return Constant; } Ref LoadUncachedZeroVector(uint8_t Size) { return _LoadNamedVectorConstant(Size, IR::NamedVectorConstant::NAMED_VECTOR_ZERO); } Ref LoadZeroVector(uint8_t Size) { return LoadAndCacheNamedVectorConstant(Size, IR::NamedVectorConstant::NAMED_VECTOR_ZERO); } // Reset the named vector constants cache array. // These are only cached per block. void ClearCachedNamedConstants() { memset(CachedNamedVectorConstants, 0, sizeof(CachedNamedVectorConstants)); CachedIndexedNamedVectorConstants.clear(); } std::pair DecodeNZCVCondition(uint8_t OP) const; Ref SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue); Ref SelectCC(uint8_t OP, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue); /** * @name Deferred RFLAG calculation and generation. * * Only handles the six flags that ALU ops typically generate. * Specifically: CF, PF, AF, ZF, SF, OF * These six flags are heavily generated through basic ALU ops and balloon the IR if not early eliminated. * This tracking structure only tracks single blocks and requires RFLAGS calculation at block-ending ops. * Some flags generating ALU ops only touch part of the registers, In these cases it will do calculation up front. * This means we still need our IR passes to eliminate all redundant flags accesses but this light OpcodeDispatcher optimization * doesn't take it to that level. * @{ */ // Deferred flag generation tracking structure. // This structure is used to track RFlags from ALU ops for invalidation. // // Future ideas: Use an invalidation mask to do partial generation of flags. // Particularly for the instructions that don't do the full set of flags calculations. // These instructions currently calculate the deferred RFLAGS immediately then overwrite rflags state. // RCLSE IR pass will catch and remove redundant rflags stores like this currently. struct DeferredFlagData { // What type of flags to generate FlagsGenerationType Type {FlagsGenerationType::TYPE_NONE}; // Source size of the op uint8_t SrcSize; // Every flag generation type has a result Ref Res {}; union { // UMUL, BEXTR, BLSI, POPCOUNT, ZCNT, RDRAND struct { } NoSource; // MUL, BLSR, BLSMSKB, BZHI struct { Ref Src1; } OneSource; // Logical struct { Ref Src1; Ref Src2; } TwoSource; // LSHLI, LSHRI, ASHRI struct { Ref Src1; uint64_t Imm; } OneSrcImmediate; // ADD, SUB struct { Ref Src1; Ref Src2; bool UpdateCF; } TwoSrcImmediate; } Sources {}; }; DeferredFlagData CurrentDeferredFlags {}; /** * @brief Takes the current deferred flag state and stores the result in to RFLAGS. * * Once executed there will no longer be any deferred flag state and RFLAGS will have the correct flags in it. * Necessary to do when leaving a IR block, or if an instruction is doing a partial overwrite of the flags. */ void CalculateDeferredFlags(uint32_t FlagsToCalculateMask = ~0U); /** * @brief Invalidates the current deferred flags structure. * * If the emulated instruction is going to overwrite all of the flags but isn't tracked using the deferred flag system * then use this function to stop tracking the current active deferred flags. */ void InvalidateDeferredFlags() { CurrentDeferredFlags.Type = FlagsGenerationType::TYPE_NONE; // No NZCV bits will be set, they are all invalid. PossiblySetNZCVBits = 0; } /** * @brief Checks if there is any deferred flag state active. * * @return True if RFLAGs contains the flags. False if deferred flags is tracking the data. */ bool IsDeferredFlagsStored() const { return CurrentDeferredFlags.Type == FlagsGenerationType::TYPE_NONE; } template void Calculate_ShiftVariable(Ref Shift, F&& Calculate) { // RCR can call this with constants, so handle that without branching. uint64_t Const; if (IsValueConstant(WrapNode(Shift), &Const)) { if (Const) { Calculate(); } return; } // Otherwise, prepare to branch. uint32_t OldSetNZCVBits = PossiblySetNZCVBits; auto Zero = _Constant(0); // If the shift is zero, do not touch the flags. auto SetBlock = CreateNewCodeBlockAfter(GetCurrentBlock()); auto EndBlock = CreateNewCodeBlockAfter(SetBlock); CondJump(Shift, Zero, EndBlock, SetBlock, {COND_EQ}); SetCurrentCodeBlock(SetBlock); StartNewBlock(); { Calculate(); Jump(EndBlock); } SetCurrentCodeBlock(EndBlock); StartNewBlock(); PossiblySetNZCVBits |= OldSetNZCVBits; } /** * @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs. * @{ */ Ref LoadPFRaw(bool Invert); Ref LoadAF(); void FixupAF(); void SetAFAndFixup(Ref AF); Ref CalculateAFForDecimal(Ref A); void CalculatePF(Ref Res); void CalculateAF(Ref Src1, Ref Src2); void CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2, bool Sub); Ref CalculateFlags_ADC(uint8_t SrcSize, Ref Src1, Ref Src2); Ref CalculateFlags_SBB(uint8_t SrcSize, Ref Src1, Ref Src2); Ref CalculateFlags_SUB(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF = true); Ref CalculateFlags_ADD(uint8_t SrcSize, Ref Src1, Ref Src2, bool UpdateCF = true); void CalculateFlags_MUL(uint8_t SrcSize, Ref Res, Ref High); void CalculateFlags_UMUL(Ref High); void CalculateFlags_Logical(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2); void CalculateFlags_ShiftLeft(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2); void CalculateFlags_ShiftLeftImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift); void CalculateFlags_ShiftRight(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2); void CalculateFlags_ShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift); void CalculateFlags_ShiftRightDoubleImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift); void CalculateFlags_ShiftRightImmediateCommon(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift); void CalculateFlags_SignShiftRightImmediate(uint8_t SrcSize, Ref Res, Ref Src1, uint64_t Shift); void CalculateFlags_BEXTR(Ref Src); void CalculateFlags_BLSI(uint8_t SrcSize, Ref Src); void CalculateFlags_BLSMSK(uint8_t SrcSize, Ref Res, Ref Src); void CalculateFlags_BLSR(uint8_t SrcSize, Ref Res, Ref Src); void CalculateFlags_POPCOUNT(Ref Src); void CalculateFlags_BZHI(uint8_t SrcSize, Ref Result, Ref Src); void CalculateFlags_ZCNT(uint8_t SrcSize, Ref Result); void CalculateFlags_RDRAND(Ref Src); /** @} */ /** * @name These functions generated deferred RFLAGs tracking. * * Depending on the operation it may force a RFLAGs calculation before storing the new deferred state. * @{ */ void GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, Ref Src1, Ref Src2, bool UpdateCF = true) { if (!UpdateCF) { // If we aren't updating CF then we need to calculate flags. Invalidation mask would make this not required. CalculateDeferredFlags(); } CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_SUB, .SrcSize = GetSrcSize(Op), .Sources = { .TwoSrcImmediate = { .Src1 = Src1, .Src2 = Src2, .UpdateCF = UpdateCF, }, }, }; } void GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref High) { CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_MUL, .SrcSize = GetSrcSize(Op), .Res = Res, .Sources = { .OneSource = { .Src1 = High, }, }, }; } void GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, Ref High) { CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_UMUL, .SrcSize = GetSrcSize(Op), .Res = High, }; } void GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src1, Ref Src2) { CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_LOGICAL, .SrcSize = GetSrcSize(Op), .Res = Res, .Sources = { .TwoSource = { .Src1 = Src1, .Src2 = Src2, }, }, }; } void GenerateFlags_ShiftLeftImmediate(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src1, uint64_t Shift) { // No flags changed if shift is zero. if (Shift == 0) { return; } CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_LSHLI, .SrcSize = GetSrcSize(Op), .Res = Res, .Sources = { .OneSrcImmediate = { .Src1 = Src1, .Imm = Shift, }, }, }; } void GenerateFlags_SignShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src1, uint64_t Shift) { // No flags changed if shift is zero. if (Shift == 0) { return; } CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_ASHRI, .SrcSize = GetSrcSize(Op), .Res = Res, .Sources = { .OneSrcImmediate = { .Src1 = Src1, .Imm = Shift, }, }, }; } void GenerateFlags_ShiftRightImmediate(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src1, uint64_t Shift) { // No flags changed if shift is zero. if (Shift == 0) { return; } CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_LSHRI, .SrcSize = GetSrcSize(Op), .Res = Res, .Sources = { .OneSrcImmediate = { .Src1 = Src1, .Imm = Shift, }, }, }; } void GenerateFlags_ShiftRightDoubleImmediate(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src1, uint64_t Shift) { // No flags changed if shift is zero. if (Shift == 0) { return; } CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_LSHRDI, .SrcSize = GetSrcSize(Op), .Res = Res, .Sources = { .OneSrcImmediate = { .Src1 = Src1, .Imm = Shift, }, }, }; } void GenerateFlags_BEXTR(FEXCore::X86Tables::DecodedOp Op, Ref Src) { CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_BEXTR, .SrcSize = GetSrcSize(Op), .Res = Src, }; } void GenerateFlags_BLSI(FEXCore::X86Tables::DecodedOp Op, Ref Src) { CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_BLSI, .SrcSize = GetSrcSize(Op), .Res = Src, }; } void GenerateFlags_BLSMSK(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src) { CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_BLSMSK, .SrcSize = GetSrcSize(Op), .Res = Res, .Sources = { .OneSource = { .Src1 = Src, }, }, }; } void GenerateFlags_BLSR(FEXCore::X86Tables::DecodedOp Op, Ref Res, Ref Src) { CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_BLSR, .SrcSize = GetSrcSize(Op), .Res = Res, .Sources = { .OneSource = { .Src1 = Src, }, }, }; } void GenerateFlags_POPCOUNT(FEXCore::X86Tables::DecodedOp Op, Ref Src) { CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_POPCOUNT, .SrcSize = GetSrcSize(Op), .Res = Src, }; } void GenerateFlags_BZHI(FEXCore::X86Tables::DecodedOp Op, Ref Result, Ref Src) { CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_BZHI, .SrcSize = GetSrcSize(Op), .Res = Result, .Sources = { .OneSource = { .Src1 = Src, }, }, }; } void GenerateFlags_ZCNT(FEXCore::X86Tables::DecodedOp Op, Ref Src) { CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_ZCNT, .SrcSize = GetSrcSize(Op), .Res = Src, }; } void GenerateFlags_RDRAND(FEXCore::X86Tables::DecodedOp Op, Ref Src) { CurrentDeferredFlags = DeferredFlagData { .Type = FlagsGenerationType::TYPE_RDRAND, .SrcSize = GetSrcSize(Op), .Res = Src, }; } Ref AndConst(FEXCore::IR::OpSize Size, Ref Node, uint64_t Const) { uint64_t NodeConst; if (IsValueConstant(WrapNode(Node), &NodeConst)) { return _Constant(NodeConst & Const); } else { return _And(Size, Node, _Constant(Const)); } } /** @} */ /** @} */ Ref GetX87Top(); void SetX87ValidTag(Ref Value, bool Valid); Ref GetX87ValidTag(Ref Value); Ref GetX87Tag(Ref Value, Ref AbridgedFTW); Ref GetX87Tag(Ref Value); void SetX87FTW(Ref FTW); Ref GetX87FTW(); void SetX87Top(Ref Value); bool DestIsLockedMem(FEXCore::X86Tables::DecodedOp Op) const { return DestIsMem(Op) && (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK) != 0; } bool DestIsMem(FEXCore::X86Tables::DecodedOp Op) const { return !Op->Dest.IsGPR(); } void CreateJumpBlocks(const fextl::vector* Blocks); bool BlockSetRIP {false}; bool Multiblock {}; uint64_t Entry; Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref Addr, Ref Value, uint8_t Align = 1) { if (CTX->IsAtomicTSOEnabled()) { return _StoreMemTSO(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1); } else { return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1); } } Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, Ref ssa0, uint8_t Align = 1) { if (CTX->IsAtomicTSOEnabled()) { return _LoadMemTSO(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1); } else { return _LoadMem(Class, Size, ssa0, Invalid(), Align, MEM_OFFSET_SXTX, 1); } } Ref _LoadMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, uint8_t Align = 1) { bool AtomicTSO = CTX->IsAtomicTSOEnabled() && !A.NonTSO; A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size); if (AtomicTSO) { return _LoadMemTSO(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale); } else { return _LoadMem(Class, Size, A.Base, A.Index, Align, A.IndexType, A.IndexScale); } } Ref _StoreMemAutoTSO(FEXCore::IR::RegisterClassType Class, uint8_t Size, AddressMode A, Ref Value, uint8_t Align = 1) { bool AtomicTSO = CTX->IsAtomicTSOEnabled() && !A.NonTSO; A = SelectAddressMode(A, AtomicTSO, Class != GPRClass, Size); if (AtomicTSO) { return _StoreMemTSO(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale); } else { return _StoreMem(Class, Size, Value, A.Base, A.Index, Align, A.IndexType, A.IndexScale); } } Ref Prefetch(bool ForStore, bool Stream, uint8_t CacheLevel, Ref ssa0) { return _Prefetch(ForStore, Stream, CacheLevel, ssa0, Invalid(), MEM_OFFSET_SXTX, 1); } void InstallHostSpecificOpcodeHandlers(); ///< Segment telemetry tracking uint32_t SegmentsNeedReadCheck {~0U}; void CheckLegacySegmentWrite(Ref NewNode, uint32_t SegmentReg); void CheckLegacySegmentRead(Ref NewNode, uint32_t SegmentReg); }; void InstallOpcodeHandlers(Context::OperatingMode Mode); } // namespace FEXCore::IR template<> struct fmt::formatter : fmt::formatter { using Base = fmt::formatter; // Pass-through the underlying value, so IDs can // be formatted like any integral value. template auto format(const FEXCore::IR::OpDispatchBuilder::FlagsGenerationType& ID, FormatContext& ctx) { return Base::format(static_cast(ID), ctx); } };