diff --git a/FEXCore/Source/Interface/Core/Core.cpp b/FEXCore/Source/Interface/Core/Core.cpp index 49ba8a7be..bfc95c45e 100644 --- a/FEXCore/Source/Interface/Core/Core.cpp +++ b/FEXCore/Source/Interface/Core/Core.cpp @@ -949,8 +949,8 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu IR::Ref R = emit->_StoreRegister(emit->Constant(Entrypoint), GPRSize); R->Reg = IR::PhysicalRegister(IR::GPRFixedClass, X86State::REG_R11).Raw; } else { - emit->_StoreContext(GPRSize, IR::FPRClass, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)), - offsetof(Core::CPUState, mm[0][0])); + emit->_StoreContextFPR(GPRSize, emit->_VCastFromGPR(IR::OpSize::i64Bit, IR::OpSize::i64Bit, emit->Constant(Entrypoint)), + offsetof(Core::CPUState, mm[0][0])); } emit->_ExitFunction(IR::OpSize::i64Bit, emit->Constant(GuestThunkEntrypoint), IR::BranchHint::None, emit->Invalid(), emit->Invalid()); }, diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index e3f7b90cd..6b2451614 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -74,7 +74,7 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) { const auto GPRSize = GetGPROpSize(); auto NewRIP = GetRelocatedPC(Op, -Op->InstSize); - _StoreContext(GPRSize, GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip)); + _StoreContextGPR(GPRSize, NewRIP, offsetof(FEXCore::Core::CPUState, rip)); Ref Arguments[SyscallArgs] { InvalidNode, InvalidNode, InvalidNode, InvalidNode, InvalidNode, InvalidNode, InvalidNode, @@ -106,7 +106,7 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) { if (Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_BLOCK_END) { // RIP could have been updated after coming back from the Syscall. - NewRIP = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, rip)); + NewRIP = _LoadContextGPR(GPRSize, offsetof(FEXCore::Core::CPUState, rip)); ExitFunction(NewRIP); } } @@ -139,14 +139,14 @@ void OpDispatchBuilder::LEAOp(OpcodeArgs) { X86Tables::DecodeFlags::GetOpAddr(Op->Flags, 0) == X86Tables::DecodeFlags::FLAG_WIDENING_SIZE_LAST ? OpSize::i64Bit : OpSize::i32Bit; - auto Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], SrcSize, Op->Flags, {.LoadData = false, .AllowUpperGarbage = SrcSize > DstSize}); - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Src, DstSize, OpSize::iInvalid); + auto Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags, {.LoadData = false, .AllowUpperGarbage = SrcSize > DstSize}); + StoreResultGPR_WithOpSize(Op, Op->Dest, Src, DstSize, OpSize::iInvalid); } else { const auto DstSize = X86Tables::DecodeFlags::GetOpAddr(Op->Flags, 0) == X86Tables::DecodeFlags::FLAG_OPERAND_SIZE_LAST ? OpSize::i16Bit : OpSize::i32Bit; - auto Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], SrcSize, Op->Flags, {.LoadData = false, .AllowUpperGarbage = SrcSize > DstSize}); - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Src, DstSize, OpSize::iInvalid); + auto Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags, {.LoadData = false, .AllowUpperGarbage = SrcSize > DstSize}); + StoreResultGPR_WithOpSize(Op, Op->Dest, Src, DstSize, OpSize::iInvalid); } } @@ -165,7 +165,7 @@ void OpDispatchBuilder::RETOp(OpcodeArgs) { Ref NewRIP = Pop(GPRSize, SP); if (Op->OP == 0xC2) { - auto Offset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + auto Offset = LoadSourceGPR(Op, Op->Src[0], Op->Flags); SP = Add(GPRSize, SP, Offset); } @@ -202,7 +202,7 @@ void OpDispatchBuilder::IRETOp(OpcodeArgs) { auto NewRIP = Pop(GPRSize, SP); // CS (lower 16 used) auto NewSegmentCS = Pop(GPRSize, SP); - _StoreContext(OpSize::i16Bit, GPRClass, NewSegmentCS, offsetof(FEXCore::Core::CPUState, cs_idx)); + _StoreContextGPR(OpSize::i16Bit, NewSegmentCS, offsetof(FEXCore::Core::CPUState, cs_idx)); UpdatePrefixFromSegment(NewSegmentCS, FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX); // eflags (lower 16 used) @@ -215,7 +215,7 @@ void OpDispatchBuilder::IRETOp(OpcodeArgs) { // ss auto NewSegmentSS = Pop(GPRSize, SP); - _StoreContext(OpSize::i16Bit, GPRClass, NewSegmentSS, offsetof(FEXCore::Core::CPUState, ss_idx)); + _StoreContextGPR(OpSize::i16Bit, NewSegmentSS, offsetof(FEXCore::Core::CPUState, ss_idx)); UpdatePrefixFromSegment(NewSegmentSS, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX); } else { // Store the stack in 32-bit mode @@ -230,7 +230,7 @@ void OpDispatchBuilder::CallbackReturnOp(OpcodeArgs) { const auto GPRSize = GetGPROpSize(); // Store the new RIP _CallbackReturn(); - auto NewRIP = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, rip)); + auto NewRIP = _LoadContextGPR(GPRSize, offsetof(FEXCore::Core::CPUState, rip)); // This ExitFunction won't actually get hit but needs to exist ExitFunction(NewRIP); BlockSetRIP = true; @@ -286,7 +286,7 @@ void OpDispatchBuilder::ADCOp(OpcodeArgs, uint32_t SrcIndex) { // Calculate flags early. CalculateDeferredFlags(); - Ref Src = LoadSource(GPRClass, Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true}); const auto Size = OpSizeFromDst(Op); const auto OpSize = std::max(OpSize::i32Bit, Size); @@ -298,7 +298,7 @@ void OpDispatchBuilder::ADCOp(OpcodeArgs, uint32_t SrcIndex) { Ref DestMem = MakeSegmentAddress(Op, Op->Dest); Before = _AtomicFetchAdd(Size, ALUOp, DestMem); } else { - Before = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Before = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); } Ref Result; @@ -314,7 +314,7 @@ void OpDispatchBuilder::ADCOp(OpcodeArgs, uint32_t SrcIndex) { } if (!DestIsLockedMem(Op)) { - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } } @@ -322,7 +322,7 @@ void OpDispatchBuilder::SBBOp(OpcodeArgs, uint32_t SrcIndex) { // Calculate flags early. CalculateDeferredFlags(); - Ref Src = LoadSource(GPRClass, Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true}); const auto Size = OpSizeFromDst(Op); const auto OpSize = std::max(OpSize::i32Bit, Size); @@ -335,13 +335,13 @@ void OpDispatchBuilder::SBBOp(OpcodeArgs, uint32_t SrcIndex) { auto SrcPlusCF = IncrementByCarry(OpSize, Src); Before = _AtomicFetchSub(Size, SrcPlusCF, DestMem); } else { - Before = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Before = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); } Result = CalculateFlags_SBB(Size, Before, Src); if (!DestIsLockedMem(Op)) { - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } } @@ -350,19 +350,19 @@ void OpDispatchBuilder::SALCOp(OpcodeArgs) { auto Result = NZCVSelect(OpSize::i32Bit, CondClass::UGE /* CF = 1 */, _InlineConstant(0xffffffff), _InlineConstant(0)); - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PUSHOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Push(Size, LoadSource(GPRClass, Op, Op->Src[0], Op->Flags)); + Push(Size, LoadSourceGPR(Op, Op->Src[0], Op->Flags)); } void OpDispatchBuilder::PUSHREGOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Push(Size, LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true})); + Push(Size, LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true})); } void OpDispatchBuilder::PUSHAOp(OpcodeArgs) { @@ -388,45 +388,51 @@ void OpDispatchBuilder::PUSHSegmentOp(OpcodeArgs, uint32_t SegmentReg) { Ref Src {}; if (!Is64BitMode) { switch (SegmentReg) { - case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX: - Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, es_idx)); + case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX: { + Src = _LoadContextGPR(SrcSize, offsetof(FEXCore::Core::CPUState, es_idx)); break; - case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX: - Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, cs_idx)); + } + case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX: { + Src = _LoadContextGPR(SrcSize, offsetof(FEXCore::Core::CPUState, cs_idx)); break; - case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX: - Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ss_idx)); + } + case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX: { + Src = _LoadContextGPR(SrcSize, offsetof(FEXCore::Core::CPUState, ss_idx)); break; - case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX: - Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ds_idx)); + } + case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX: { + Src = _LoadContextGPR(SrcSize, offsetof(FEXCore::Core::CPUState, ds_idx)); break; - case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX: - Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs_idx)); + } + case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX: { + Src = _LoadContextGPR(SrcSize, offsetof(FEXCore::Core::CPUState, fs_idx)); break; - case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX: - Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_idx)); + } + case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX: { + Src = _LoadContextGPR(SrcSize, offsetof(FEXCore::Core::CPUState, gs_idx)); break; + } default: FEX_UNREACHABLE; } } else { switch (SegmentReg) { case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX: - Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, es_cached)); + Src = _LoadContextGPR(SrcSize, offsetof(FEXCore::Core::CPUState, es_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX: - Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, cs_cached)); + Src = _LoadContextGPR(SrcSize, offsetof(FEXCore::Core::CPUState, cs_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX: - Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ss_cached)); + Src = _LoadContextGPR(SrcSize, offsetof(FEXCore::Core::CPUState, ss_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX: - Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, ds_cached)); + Src = _LoadContextGPR(SrcSize, offsetof(FEXCore::Core::CPUState, ds_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX: - Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs_cached)); + Src = _LoadContextGPR(SrcSize, offsetof(FEXCore::Core::CPUState, fs_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX: - Src = _LoadContext(SrcSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached)); + Src = _LoadContextGPR(SrcSize, offsetof(FEXCore::Core::CPUState, gs_cached)); break; default: FEX_UNREACHABLE; } @@ -440,7 +446,7 @@ void OpDispatchBuilder::PUSHSegmentOp(OpcodeArgs, uint32_t SegmentReg) { void OpDispatchBuilder::POPOp(OpcodeArgs) { Ref Value = Pop(OpSizeFromSrc(Op)); - StoreResult(GPRClass, Op, Value, OpSize::iInvalid); + StoreResultGPR(Op, Value, OpSize::iInvalid); } void OpDispatchBuilder::POPAOp(OpcodeArgs) { @@ -473,24 +479,24 @@ void OpDispatchBuilder::POPSegmentOp(OpcodeArgs, uint32_t SegmentReg) { switch (SegmentReg) { case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX: - _StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, es_idx)); + _StoreContextGPR(DstSize, NewSegment, offsetof(FEXCore::Core::CPUState, es_idx)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX: - _StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, cs_idx)); + _StoreContextGPR(DstSize, NewSegment, offsetof(FEXCore::Core::CPUState, cs_idx)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX: // Unset the 'active' bit in the packed TF, skipping the single step exception after this instruction SetRFLAG(_And(OpSize::i32Bit, GetRFLAG(FEXCore::X86State::RFLAG_TF_RAW_LOC), Constant(1))); - _StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ss_idx)); + _StoreContextGPR(DstSize, NewSegment, offsetof(FEXCore::Core::CPUState, ss_idx)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX: - _StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ds_idx)); + _StoreContextGPR(DstSize, NewSegment, offsetof(FEXCore::Core::CPUState, ds_idx)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX: - _StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, fs_idx)); + _StoreContextGPR(DstSize, NewSegment, offsetof(FEXCore::Core::CPUState, fs_idx)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX: - _StoreContext(DstSize, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, gs_idx)); + _StoreContextGPR(DstSize, NewSegment, offsetof(FEXCore::Core::CPUState, gs_idx)); break; default: break; // Do nothing } @@ -553,7 +559,7 @@ void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) { BlockSetRIP = true; const auto Size = OpSizeFromSrc(Op); - Ref JMPPCOffset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + Ref JMPPCOffset = LoadSourceGPR(Op, Op->Src[0], Op->Flags); // Push the return address. auto ConstantPC = GetRelocatedPC(Op); @@ -651,7 +657,7 @@ void OpDispatchBuilder::SETccOp(OpcodeArgs) { SrcCond = LoadPFRaw(true, ParityJumpIsJP(Op->OP & 0xf)); } - StoreResult(GPRClass, Op, SrcCond, OpSize::iInvalid); + StoreResultGPR(Op, SrcCond, OpSize::iInvalid); } void OpDispatchBuilder::CMOVOp(OpcodeArgs) { @@ -662,12 +668,12 @@ void OpDispatchBuilder::CMOVOp(OpcodeArgs) { CalculateDeferredFlags(); // Destination is always a GPR. - Ref Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags); + Ref Dest = LoadSourceGPR_WithOpSize(Op, Op->Dest, GPRSize, Op->Flags); Ref Src {}, SrcCond {}; if (Op->Src[0].IsGPR()) { - Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], GPRSize, Op->Flags); + Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], GPRSize, Op->Flags); } else { - Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags); } if (auto Cond = DecodeNZCVCondition(OP); Cond) { @@ -684,7 +690,7 @@ void OpDispatchBuilder::CMOVOp(OpcodeArgs) { SrcCond = _NZCVSelect(ResultSize, CondClass::NEQ, Src, Dest); } - StoreResult(GPRClass, Op, SrcCond, OpSize::iInvalid); + StoreResultGPR(Op, SrcCond, OpSize::iInvalid); } void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) { @@ -828,9 +834,9 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) { uint64_t Target = Op->PC + Op->InstSize + Op->Src[1].Literal(); - Ref CondReg = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], SrcSize, Op->Flags); + Ref CondReg = LoadSourceGPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); CondReg = Sub(OpSize, CondReg, 1); - StoreResult(GPRClass, Op, Op->Src[0], CondReg, OpSize::iInvalid); + StoreResultGPR(Op, Op->Src[0], CondReg, OpSize::iInvalid); // If LOOPE then jumps to target if RCX != 0 && ZF == 1 // If LOOPNE then jumps to target if RCX != 0 && ZF == 0 @@ -933,7 +939,7 @@ void OpDispatchBuilder::JUMPAbsoluteOp(OpcodeArgs) { // This is just an unconditional jump // This uses ModRM to determine its location // No way to use this effectively in multiblock - auto RIPOffset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + auto RIPOffset = LoadSourceGPR(Op, Op->Src[0], Op->Flags); // Store the new RIP ExitFunction(RIPOffset); @@ -949,11 +955,11 @@ void OpDispatchBuilder::JUMPFARIndirectOp(OpcodeArgs) { // No way to use this effectively in multiblock Ref Src = MakeSegmentAddress(Op, Op->Dest); AddressMode SrcCS = {.Base = Src, .Offset = 4, .AddrSize = OpSize::i64Bit}; - auto RIPOffset = _LoadMemAutoTSO(GPRClass, OpSize::i32Bit, Src, OpSize::i8Bit); - auto NewSegmentCS = _LoadMemAutoTSO(GPRClass, OpSize::i16Bit, SrcCS, OpSize::i8Bit); + auto RIPOffset = _LoadMemGPRAutoTSO(OpSize::i32Bit, Src, OpSize::i8Bit); + auto NewSegmentCS = _LoadMemGPRAutoTSO(OpSize::i16Bit, SrcCS, OpSize::i8Bit); // Set up the new CSSegment. - _StoreContext(OpSize::i16Bit, GPRClass, NewSegmentCS, offsetof(FEXCore::Core::CPUState, cs_idx)); + _StoreContextGPR(OpSize::i16Bit, NewSegmentCS, offsetof(FEXCore::Core::CPUState, cs_idx)); UpdatePrefixFromSegment(NewSegmentCS, FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX); // Store the new RIP @@ -970,9 +976,9 @@ void OpDispatchBuilder::CALLFARIndirectOp(OpcodeArgs) { Ref Src = MakeSegmentAddress(Op, Op->Dest); AddressMode SrcCS = {.Base = Src, .Offset = 4, .AddrSize = OpSize::i64Bit}; - auto RIPOffset = _LoadMemAutoTSO(GPRClass, OpSize::i32Bit, Src, OpSize::i8Bit); - auto NewSegmentCS = _LoadMemAutoTSO(GPRClass, OpSize::i16Bit, SrcCS, OpSize::i8Bit); - auto CurrentCS = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, cs_idx)); + auto RIPOffset = _LoadMemGPRAutoTSO(OpSize::i32Bit, Src, OpSize::i8Bit); + auto NewSegmentCS = _LoadMemGPRAutoTSO(OpSize::i16Bit, SrcCS, OpSize::i8Bit); + auto CurrentCS = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, cs_idx)); auto NewRIP = GetRelocatedPC(Op); @@ -983,7 +989,7 @@ void OpDispatchBuilder::CALLFARIndirectOp(OpcodeArgs) { Push(SrcSize, NewRIP); // Set up the new CSSegment. - _StoreContext(OpSize::i16Bit, GPRClass, NewSegmentCS, offsetof(FEXCore::Core::CPUState, cs_idx)); + _StoreContextGPR(OpSize::i16Bit, NewSegmentCS, offsetof(FEXCore::Core::CPUState, cs_idx)); UpdatePrefixFromSegment(NewSegmentCS, FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX); // Store the new RIP @@ -1006,7 +1012,7 @@ void OpDispatchBuilder::RETFARIndirectOp(OpcodeArgs) { // Store the new stack pointer StoreGPRRegister(X86State::REG_RSP, SP); - _StoreContext(OpSize::i16Bit, GPRClass, NewSegmentCS, offsetof(FEXCore::Core::CPUState, cs_idx)); + _StoreContextGPR(OpSize::i16Bit, NewSegmentCS, offsetof(FEXCore::Core::CPUState, cs_idx)); UpdatePrefixFromSegment(NewSegmentCS, FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX); // Store the new RIP @@ -1017,8 +1023,8 @@ void OpDispatchBuilder::RETFARIndirectOp(OpcodeArgs) { void OpDispatchBuilder::TESTOp(OpcodeArgs, uint32_t SrcIndex) { // TEST is an instruction that does an AND between the sources // Result isn't stored in result, only writes to flags - Ref Src = LoadSource(GPRClass, Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true}); - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); const auto Size = OpSizeFromDst(Op); LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i64Bit, "Invalid size"); @@ -1064,52 +1070,52 @@ void OpDispatchBuilder::MOVSXDOp(OpcodeArgs) { auto Size = std::min(OpSize::i32Bit, OpSizeFromSrc(Op)); bool Sext = (Size != OpSize::i16Bit) && Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REX_WIDENING; - Ref Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], Size, Op->Flags, {.AllowUpperGarbage = Sext}); + Ref Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], Size, Op->Flags, {.AllowUpperGarbage = Sext}); if (Size == OpSize::i16Bit) { // This'll make sure to insert in to the lower 16bits without modifying upper bits - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Src, Size, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, Src, Size, OpSize::iInvalid); } else if (Sext) { // With REX.W then Sext Src = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(Size), 0, Src); - StoreResult(GPRClass, Op, Src, OpSize::iInvalid); + StoreResultGPR(Op, Src, OpSize::iInvalid); } else { // Without REX.W then Zext (store result implicitly zero extends) - StoreResult(GPRClass, Op, Src, OpSize::iInvalid); + StoreResultGPR(Op, Src, OpSize::iInvalid); } } void OpDispatchBuilder::MOVSXOp(OpcodeArgs) { // Load garbage in upper bits, since we're sign extending anyway const auto Size = OpSizeFromSrc(Op); - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); // Sign-extend to DstSize and zero-extend to the register size, using a fast // path for 32-bit dests where the native 32-bit Sbfe zero extends the top. const auto DstSize = OpSizeFromDst(Op); Src = _Sbfe(DstSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, IR::OpSizeAsBits(Size), 0, Src); - StoreResult(GPRClass, Op, Op->Dest, Src, OpSize::iInvalid); + StoreResultGPR(Op, Op->Dest, Src, OpSize::iInvalid); } void OpDispatchBuilder::MOVZXOp(OpcodeArgs) { - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags); // Store result implicitly zero extends - StoreResult(GPRClass, Op, Src, OpSize::iInvalid); + StoreResultGPR(Op, Src, OpSize::iInvalid); } void OpDispatchBuilder::CMPOp(OpcodeArgs, uint32_t SrcIndex) { // CMP is an instruction that does a SUB between the sources // Result isn't stored in result, only writes to flags - Ref Src = LoadSource(GPRClass, Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true}); - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); CalculateFlags_SUB(OpSizeFromSrc(Op), Dest, Src); } void OpDispatchBuilder::CQOOp(OpcodeArgs) { - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); auto Size = OpSizeFromSrc(Op); Ref Upper = _Sbfe(std::max(OpSize::i32Bit, Size), 1, GetSrcBitSize(Op) - 1, Src); - StoreResult(GPRClass, Op, Upper, OpSize::iInvalid); + StoreResultGPR(Op, Upper, OpSize::iInvalid); } void OpDispatchBuilder::XCHGOp(OpcodeArgs) { @@ -1143,7 +1149,7 @@ void OpDispatchBuilder::XCHGOp(OpcodeArgs) { } // AllowUpperGarbage: OK to allow as it will be overwritten by StoreResult. - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); if (DestIsMem(Op)) { HandledLock = (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK) != 0; @@ -1152,16 +1158,16 @@ void OpDispatchBuilder::XCHGOp(OpcodeArgs) { _MonoBackpatcherWrite(OpSizeFromSrc(Op), Src, Dest); } else { auto Result = _AtomicSwap(OpSizeFromSrc(Op), Src, Dest); - StoreResult(GPRClass, Op, Op->Src[0], Result, OpSize::iInvalid); + StoreResultGPR(Op, Op->Src[0], Result, OpSize::iInvalid); } } else { // AllowUpperGarbage: OK to allow as it will be overwritten by StoreResult. - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); // Swap the contents // Order matters here since we don't want to swap context contents for one that effects the other - StoreResult(GPRClass, Op, Op->Dest, Src, OpSize::iInvalid); - StoreResult(GPRClass, Op, Op->Src[0], Dest, OpSize::iInvalid); + StoreResultGPR(Op, Op->Dest, Src, OpSize::iInvalid); + StoreResultGPR(Op, Op->Src[0], Dest, OpSize::iInvalid); } } @@ -1172,7 +1178,7 @@ void OpDispatchBuilder::CDQOp(OpcodeArgs) { Src = _Sbfe(DstSize <= OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, IR::OpSizeAsBits(SrcSize), 0, Src); - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Src, DstSize, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, Src, DstSize, OpSize::iInvalid); } void OpDispatchBuilder::SAHFOp(OpcodeArgs) { @@ -1238,17 +1244,17 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs, bool ToSeg) { // The loads here also load the selector, NOT the base if (ToSeg) { - Ref Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], OpSize::i16Bit, Op->Flags); + Ref Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], OpSize::i16Bit, Op->Flags); switch (Op->Dest.Data.GPR.GPR) { case FEXCore::X86State::REG_RAX: // ES case FEXCore::X86State::REG_R8: // ES - _StoreContext(OpSize::i16Bit, GPRClass, Src, offsetof(FEXCore::Core::CPUState, es_idx)); + _StoreContextGPR(OpSize::i16Bit, Src, offsetof(FEXCore::Core::CPUState, es_idx)); UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX); break; case FEXCore::X86State::REG_RBX: // DS case FEXCore::X86State::REG_R11: // DS - _StoreContext(OpSize::i16Bit, GPRClass, Src, offsetof(FEXCore::Core::CPUState, ds_idx)); + _StoreContextGPR(OpSize::i16Bit, Src, offsetof(FEXCore::Core::CPUState, ds_idx)); UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX); break; case FEXCore::X86State::REG_RCX: // CS @@ -1263,13 +1269,13 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs, bool ToSeg) { break; case FEXCore::X86State::REG_RDX: // SS case FEXCore::X86State::REG_R10: // SS - _StoreContext(OpSize::i16Bit, GPRClass, Src, offsetof(FEXCore::Core::CPUState, ss_idx)); + _StoreContextGPR(OpSize::i16Bit, Src, offsetof(FEXCore::Core::CPUState, ss_idx)); UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX); break; case FEXCore::X86State::REG_RBP: // GS case FEXCore::X86State::REG_R13: // GS if (!Is64BitMode) { - _StoreContext(OpSize::i16Bit, GPRClass, Src, offsetof(FEXCore::Core::CPUState, gs_idx)); + _StoreContextGPR(OpSize::i16Bit, Src, offsetof(FEXCore::Core::CPUState, gs_idx)); UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX); } else { LogMan::Msg::EFmt("We don't support modifying GS selector in 64bit mode!"); @@ -1279,7 +1285,7 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs, bool ToSeg) { case FEXCore::X86State::REG_RSP: // FS case FEXCore::X86State::REG_R12: // FS if (!Is64BitMode) { - _StoreContext(OpSize::i16Bit, GPRClass, Src, offsetof(FEXCore::Core::CPUState, fs_idx)); + _StoreContextGPR(OpSize::i16Bit, Src, offsetof(FEXCore::Core::CPUState, fs_idx)); UpdatePrefixFromSegment(Src, FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX); } else { LogMan::Msg::EFmt("We don't support modifying FS selector in 64bit mode!"); @@ -1294,26 +1300,26 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs, bool ToSeg) { switch (Op->Src[0].Data.GPR.GPR) { case FEXCore::X86State::REG_RAX: // ES case FEXCore::X86State::REG_R8: // ES - Segment = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, es_idx)); + Segment = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, es_idx)); break; case FEXCore::X86State::REG_RBX: // DS case FEXCore::X86State::REG_R11: // DS - Segment = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, ds_idx)); + Segment = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, ds_idx)); break; case FEXCore::X86State::REG_RCX: // CS case FEXCore::X86State::REG_R9: // CS - Segment = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, cs_idx)); + Segment = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, cs_idx)); break; case FEXCore::X86State::REG_RDX: // SS case FEXCore::X86State::REG_R10: // SS - Segment = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, ss_idx)); + Segment = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, ss_idx)); break; case FEXCore::X86State::REG_RBP: // GS case FEXCore::X86State::REG_R13: // GS if (Is64BitMode) { Segment = Constant(0); } else { - Segment = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, gs_idx)); + Segment = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, gs_idx)); } break; case FEXCore::X86State::REG_RSP: // FS @@ -1321,17 +1327,17 @@ void OpDispatchBuilder::MOVSegOp(OpcodeArgs, bool ToSeg) { if (Is64BitMode) { Segment = Constant(0); } else { - Segment = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, fs_idx)); + Segment = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, fs_idx)); } break; default: UnimplementedOp(Op); return; } if (DestIsMem(Op)) { // If the destination is memory then we always store 16-bits only - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Segment, OpSize::i16Bit, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, Segment, OpSize::i16Bit, OpSize::iInvalid); } else { // If the destination is a GPR then we follow register storing rules - StoreResult(GPRClass, Op, Segment, OpSize::iInvalid); + StoreResultGPR(Op, Segment, OpSize::iInvalid); } } } @@ -1360,29 +1366,29 @@ void OpDispatchBuilder::MOVOffsetOp(OpcodeArgs) { // Dest is GPR Ref Src {}; if (Op->Src[0].Data.Literal.Size <= 4) { - Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.ForceLoad = true}); + Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.ForceLoad = true}); } else { const auto OpSize = OpSizeFromSrc(Op); auto A = GenMemSrcFromOp(0); - Src = _LoadMemAutoTSO(GPRClass, OpSize, A, OpSize::i8Bit); + Src = _LoadMemGPRAutoTSO(OpSize, A, OpSize::i8Bit); } - StoreResult(GPRClass, Op, Op->Dest, Src, OpSize::iInvalid); + StoreResultGPR(Op, Op->Dest, Src, OpSize::iInvalid); break; } case 0xA2: case 0xA3: { // Source is GPR // Dest is memory(literal) - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); // This one is a bit special since the destination is a literal // So the destination gets stored in Src[1] if (Op->Src[1].Data.Literal.Size <= 4) { - StoreResult(GPRClass, Op, Op->Src[1], Src, OpSize::iInvalid); + StoreResultGPR(Op, Op->Src[1], Src, OpSize::iInvalid); } else { const auto OpSize = OpSizeFromSrc(Op); auto A = GenMemSrcFromOp(1); - _StoreMemAutoTSO(GPRClass, OpSize, A, Src, OpSize::i8Bit); + _StoreMemGPRAutoTSO(OpSize, A, Src, OpSize::i8Bit); } break; } @@ -1392,7 +1398,7 @@ void OpDispatchBuilder::MOVOffsetOp(OpcodeArgs) { void OpDispatchBuilder::CPUIDOp(OpcodeArgs) { const auto GPRSize = GetGPROpSize(); - Ref Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], GPRSize, Op->Flags); + Ref Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], GPRSize, Op->Flags); Ref Leaf = LoadGPRRegister(X86State::REG_RCX); Ref RAX = _AllocateGPR(false); @@ -1433,15 +1439,15 @@ void OpDispatchBuilder::XGetBVOp(OpcodeArgs) { void OpDispatchBuilder::SHLOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); - auto Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + auto Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); Ref Result = _Lshl(Size == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, Dest, Src); HandleShift(Op, Result, Dest, ShiftType::LSL, Src); } void OpDispatchBuilder::SHLImmediateOp(OpcodeArgs, bool SHL1Bit) { - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); uint64_t Shift = LoadConstantShift(Op, SHL1Bit); const auto Size = GetSrcBitSize(Op); @@ -1450,13 +1456,13 @@ void OpDispatchBuilder::SHLImmediateOp(OpcodeArgs, bool SHL1Bit) { CalculateFlags_ShiftLeftImmediate(OpSizeFromSrc(Op), Result, Dest, Shift); CalculateDeferredFlags(); - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::SHROp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= OpSize::i32Bit}); - auto Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= OpSize::i32Bit}); + auto Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); auto ALUOp = _Lshr(std::max(OpSize::i32Bit, Size), Dest, Src); HandleShift(Op, ALUOp, Dest, ShiftType::LSR, Src); @@ -1464,14 +1470,14 @@ void OpDispatchBuilder::SHROp(OpcodeArgs) { void OpDispatchBuilder::SHRImmediateOp(OpcodeArgs, bool SHR1Bit) { const auto Size = GetSrcBitSize(Op); - auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32}); + auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32}); uint64_t Shift = LoadConstantShift(Op, SHR1Bit); auto ALUOp = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, Constant(Shift)); CalculateFlags_ShiftRightImmediate(OpSizeFromSrc(Op), ALUOp, Dest, Shift); CalculateDeferredFlags(); - StoreResult(GPRClass, Op, ALUOp, OpSize::iInvalid); + StoreResultGPR(Op, ALUOp, OpSize::iInvalid); } void OpDispatchBuilder::SHLDOp(OpcodeArgs) { @@ -1481,11 +1487,11 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) { const auto Size = GetSrcBitSize(Op); // Allow garbage on the Src if it will be ignored by the Lshr below - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = Size >= 32}); - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = Size >= 32}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags); // Allow garbage on the shift, we're masking it anyway. - Ref Shift = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + Ref Shift = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); // x86 masks the shift by 0x3F or 0x1F depending on size of op. if (Size == 64) { @@ -1521,8 +1527,8 @@ void OpDispatchBuilder::SHLDImmediateOp(OpcodeArgs) { uint64_t Shift = LoadConstantShift(Op, false); const auto Size = GetSrcBitSize(Op); - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = Size >= 32}); - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = Size >= 32}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32}); if (Shift != 0) { Ref Res {}; @@ -1541,10 +1547,10 @@ void OpDispatchBuilder::SHLDImmediateOp(OpcodeArgs) { CalculateFlags_ShiftLeftImmediate(OpSizeFromSrc(Op), Res, Dest, Shift); CalculateDeferredFlags(); - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); } else if (Shift == 0 && Size == 32) { // Ensure Zext still occurs - StoreResult(GPRClass, Op, Dest, OpSize::iInvalid); + StoreResultGPR(Op, Dest, OpSize::iInvalid); } } @@ -1553,8 +1559,8 @@ void OpDispatchBuilder::SHRDOp(OpcodeArgs) { // This instruction conditionally generates flags so we need to insure sane state going in. CalculateDeferredFlags(); - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags); Ref Shift = LoadGPRRegister(X86State::REG_RCX); @@ -1583,8 +1589,8 @@ void OpDispatchBuilder::SHRDOp(OpcodeArgs) { } void OpDispatchBuilder::SHRDImmediateOp(OpcodeArgs) { - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags); uint64_t Shift = LoadConstantShift(Op, false); const auto Size = GetSrcBitSize(Op); @@ -1604,11 +1610,11 @@ void OpDispatchBuilder::SHRDImmediateOp(OpcodeArgs) { Res = _Extr(OpSizeFromSrc(Op), Src, Dest, Shift); } - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); CalculateFlags_ShiftRightDoubleImmediate(OpSizeFromSrc(Op), Res, Dest, Shift); } else if (Shift == 0 && Size == 32) { // Ensure Zext still occurs - StoreResult(GPRClass, Op, Dest, OpSize::iInvalid); + StoreResultGPR(Op, Dest, OpSize::iInvalid); } } @@ -1620,7 +1626,7 @@ void OpDispatchBuilder::ASHROp(OpcodeArgs, bool Immediate, bool SHR1Bit) { // Otherwise, if Size = Opsize, then both are 4 or 8 and match the a64 // semantics directly, so again we can have garbage. The only case where we // need zero-extension here is when the sizes mismatch. - auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = (OpSize == Size) || (Size < OpSize::i32Bit)}); + auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = (OpSize == Size) || (Size < OpSize::i32Bit)}); if (Size < OpSize::i32Bit) { Dest = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(Size), 0, Dest); @@ -1632,9 +1638,9 @@ void OpDispatchBuilder::ASHROp(OpcodeArgs, bool Immediate, bool SHR1Bit) { CalculateFlags_SignShiftRightImmediate(OpSizeFromSrc(Op), Result, Dest, Shift); CalculateDeferredFlags(); - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } else { - auto Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + auto Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); Ref Result = _Ashr(OpSize, Dest, Src); HandleShift(Op, Result, Dest, ShiftType::ASR, Src); @@ -1658,12 +1664,12 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I UnmaskedConst = LoadConstantShift(Op, Is1Bit); UnmaskedSrc = ARef(UnmaskedConst); } else { - UnmaskedSrc = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true})); + UnmaskedSrc = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true})); } auto Src = UnmaskedSrc.And(Mask); // We fill the upper bits so we allow garbage on load. - auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + auto Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); if (Size < 32) { // ARM doesn't support 8/16bit rotates. Emulate with an insert @@ -1673,7 +1679,7 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I // To rotate 64-bits left, right-rotate by (64 - Shift) = -Shift mod 64. auto Res = _Ror(OpSize, Dest, (Left ? Src.Neg() : Src).Ref()); - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); if (Is1Bit || IsImmediate) { if (UnmaskedSrc.C) { @@ -1702,12 +1708,12 @@ void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool I } void OpDispatchBuilder::ANDNBMIOp(OpcodeArgs) { - auto* Src1 = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); - auto* Src2 = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + auto* Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + auto* Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); auto Dest = _Andn(OpSizeFromSrc(Op), Src2, Src1); - StoreResult(GPRClass, Op, Dest, OpSize::iInvalid); + StoreResultGPR(Op, Dest, OpSize::iInvalid); CalculateFlags_Logical(OpSizeFromSrc(Op), Dest); } @@ -1716,8 +1722,8 @@ void OpDispatchBuilder::BEXTRBMIOp(OpcodeArgs) { // along with some edge-case handling and flag setting. LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed"); - auto* Src1 = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); - auto* Src2 = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + auto* Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + auto* Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); const auto Size = OpSizeFromSrc(Op); const auto SrcSize = IR::OpSizeAsBits(Size); @@ -1746,7 +1752,7 @@ void OpDispatchBuilder::BEXTRBMIOp(OpcodeArgs) { auto Dest = _Select(Size, Size, CondClass::ULE, Length, MaxSrcBitOp, Masked, SanitizedShifted); // Finally store the result. - StoreResult(GPRClass, Op, Dest, OpSize::iInvalid); + StoreResultGPR(Op, Dest, OpSize::iInvalid); // ZF is set properly. CF and OF are defined as being set to zero. SF, PF, and // AF are undefined. @@ -1759,11 +1765,11 @@ void OpDispatchBuilder::BLSIBMIOp(OpcodeArgs) { LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed"); const auto Size = OpSizeFromSrc(Op); - auto* Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); auto NegatedSrc = _Neg(Size, Src); auto Result = _And(Size, Src, NegatedSrc); - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); // CF is cleared if Src is zero, otherwise it's set. However, Src is zero iff // Result is zero, so we can test the result instead. So, CF is just the @@ -1780,10 +1786,10 @@ void OpDispatchBuilder::BLSMSKBMIOp(OpcodeArgs) { LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed"); const auto Size = OpSizeFromSrc(Op); - auto* Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); auto Result = _Xor(Size, Sub(Size, Src, 1), Src); - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); InvalidatePF_AF(); // CF set according to the Src @@ -1800,10 +1806,10 @@ void OpDispatchBuilder::BLSRBMIOp(OpcodeArgs) { LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed"); const auto Size = OpSizeFromSrc(Op); - auto* Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); auto Result = _And(Size, Sub(Size, Src, 1), Src); - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); auto CFInv = To01(OpSize::i64Bit, Src); @@ -1820,8 +1826,8 @@ void OpDispatchBuilder::BMI2Shift(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); const auto SrcSize = Op->Src[0].IsGPR() ? GPRSize : Size; - auto* Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], SrcSize, Op->Flags); - auto* Shift = LoadSource_WithOpSize(GPRClass, Op, Op->Src[1], GPRSize, Op->Flags, {.AllowUpperGarbage = true}); + auto* Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); + auto* Shift = LoadSourceGPR_WithOpSize(Op, Op->Src[1], GPRSize, Op->Flags, {.AllowUpperGarbage = true}); Ref Result; if (Op->OP == 0x6F7) { @@ -1835,7 +1841,7 @@ void OpDispatchBuilder::BMI2Shift(OpcodeArgs) { Result = _Lshr(Size, Src, Shift); } - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::BZHI(OpcodeArgs) { @@ -1844,9 +1850,9 @@ void OpDispatchBuilder::BZHI(OpcodeArgs) { // In 32-bit mode we only look at bottom 32-bit, no 8 or 16-bit BZHI so no // need to zero-extend sources - auto* Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); - auto* Index = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + auto* Index = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); // Clear the high bits specified by the index. A64 only considers bottom bits // of the shift, so we don't need to mask bottom 8-bits ourselves. @@ -1862,7 +1868,7 @@ void OpDispatchBuilder::BZHI(OpcodeArgs) { // shenanigans and use the raw versions here. _TestNZ(OpSize::i64Bit, Index, Constant(0xFF & ~(OperandSize - 1))); auto Result = _NZCVSelect(Size, CondClass::NEQ, Src, MaskResult); - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); auto CFInv = _NZCVSelect01(CondClass::EQ); @@ -1891,13 +1897,13 @@ void OpDispatchBuilder::RORX(OpcodeArgs) { return; } - auto* Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); auto* Result = Src; if (DoRotation) [[likely]] { Result = _Ror(OpSizeFromSrc(Op), Src, _InlineConstant(Amount)); } - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::MULX(OpcodeArgs) { @@ -1909,7 +1915,7 @@ void OpDispatchBuilder::MULX(OpcodeArgs) { const auto GPRSize = GetGPROpSize(); const auto Src1Size = Op->Src[1].IsGPR() ? GPRSize : OpSize; - Ref Src1 = LoadSource_WithOpSize(GPRClass, Op, Op->Src[1], Src1Size, Op->Flags); + Ref Src1 = LoadSourceGPR_WithOpSize(Op, Op->Src[1], Src1Size, Op->Flags); Ref Src2 = LoadGPRRegister(X86State::REG_RDX, GPRSize); // As per the Intel Software Development Manual, if the destination and @@ -1917,40 +1923,40 @@ void OpDispatchBuilder::MULX(OpcodeArgs) { // will be the high half of the multiplication result. if (Op->Dest.Data.GPR.GPR == Op->Src[0].Data.GPR.GPR) { Ref ResultHi = _UMulH(OpSize, Src1, Src2); - StoreResult(GPRClass, Op, Op->Dest, ResultHi, OpSize::iInvalid); + StoreResultGPR(Op, Op->Dest, ResultHi, OpSize::iInvalid); } else { Ref ResultLo = _UMul(OpSize, Src1, Src2); Ref ResultHi = _UMulH(OpSize, Src1, Src2); - StoreResult(GPRClass, Op, Op->Src[0], ResultLo, OpSize::iInvalid); - StoreResult(GPRClass, Op, Op->Dest, ResultHi, OpSize::iInvalid); + StoreResultGPR(Op, Op->Src[0], ResultLo, OpSize::iInvalid); + StoreResultGPR(Op, Op->Dest, ResultHi, OpSize::iInvalid); } } void OpDispatchBuilder::PDEP(OpcodeArgs) { LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed"); - auto* Input = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); - auto* Mask = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + auto* Input = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + auto* Mask = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); auto Result = _PDep(OpSizeFromSrc(Op), Input, Mask); - StoreResult(GPRClass, Op, Op->Dest, Result, OpSize::iInvalid); + StoreResultGPR(Op, Op->Dest, Result, OpSize::iInvalid); } void OpDispatchBuilder::PEXT(OpcodeArgs) { LOGMAN_THROW_A_FMT(Op->InstSize >= 4, "No masking needed"); - auto* Input = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); - auto* Mask = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + auto* Input = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + auto* Mask = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); auto Result = _PExt(OpSizeFromSrc(Op), Input, Mask); - StoreResult(GPRClass, Op, Op->Dest, Result, OpSize::iInvalid); + StoreResultGPR(Op, Op->Dest, Result, OpSize::iInvalid); } void OpDispatchBuilder::ADXOp(OpcodeArgs) { const auto OpSize = OpSizeFromSrc(Op); // Only 32/64-bit anyway so allow garbage, we use 32-bit ops. - auto* Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); - auto* Before = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + auto* Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + auto* Before = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); // Handles ADCX and ADOX const bool IsADCX = Op->OP == 0x1F6; @@ -1971,7 +1977,7 @@ void OpDispatchBuilder::ADXOp(OpcodeArgs) { // Do the actual add. HandleNZCV_RMW(); auto Result = _AdcWithFlags(OpSize, Src, Before); - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); // Now restore all flags except the one we're updating. if (CTX->HostFeatures.SupportsFlagM) { @@ -1999,7 +2005,7 @@ void OpDispatchBuilder::RCROp1Bit(OpcodeArgs) { CalculateDeferredFlags(); // We expliclty mask for <32-bit so allow garbage - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); const auto Size = GetSrcBitSize(Op); auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC); Ref Res; @@ -2020,7 +2026,7 @@ void OpDispatchBuilder::RCROp1Bit(OpcodeArgs) { Res = _Orlshl(OpSize::i32Bit, Res, CF, Size - Shift); } - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); // OF is the top two MSBs XOR'd together // Only when Shift == 1, it is undefined otherwise @@ -2031,7 +2037,7 @@ void OpDispatchBuilder::RCROp8x1Bit(OpcodeArgs) { // Calculate flags early. CalculateDeferredFlags(); - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags); const auto SizeBit = GetSrcBitSize(Op); auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC); @@ -2042,7 +2048,7 @@ void OpDispatchBuilder::RCROp8x1Bit(OpcodeArgs) { Ref Res = _Bfe(OpSize::i32Bit, 7, 1, Dest); Res = _Bfi(OpSize::i32Bit, 1, 7, Res, CF); - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); // OF is the top two MSBs XOR'd together SetRFLAG(_XorShift(OpSize::i32Bit, Res, Res, ShiftType::LSR, 1), SizeBit - 2, true); @@ -2062,7 +2068,7 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) { CalculateDeferredFlags(); const auto OpSize = OpSizeFromSrc(Op); - Ref Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); uint64_t Const; if (IsValueConstant(WrapNode(Src), &Const)) { Const &= Mask; @@ -2071,7 +2077,7 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) { return; } - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); // Res = Src >> Shift Ref Res = _Lshr(OpSize, Dest, Src); @@ -2095,7 +2101,7 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) { SetRFLAG(Xor, Size - 2, true); } - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); return; } @@ -2104,8 +2110,8 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) { Op, SrcMasked, [this, Op, Size, OpSize]() { // Rematerialize loads to avoid crossblock liveness - Ref Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); // Res = Src >> Shift Ref Res = _Lshr(OpSize, Dest, Src); @@ -2135,7 +2141,7 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) { auto Xor = _XorShift(OpSize, Res, Res, ShiftType::LSR, 1); SetRFLAG(Xor, Size - 2, true); - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); }, OpSizeFromSrc(Op) == OpSize::i32Bit ? std::make_optional(&OpDispatchBuilder::ZeroShiftResult) : std::nullopt); } @@ -2146,19 +2152,19 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) { const auto Size = GetSrcBitSize(Op); // x86 masks the shift by 0x3F or 0x1F depending on size of op - auto Src = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true})); + auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true})); Src = Src.And(0x1F); // CF only changes if we actually shifted. OF undefined if we didn't shift. // The result is unchanged if we didn't shift. So branch over the whole thing. Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size]() { // Rematerialized to avoid crossblock liveness - auto Src = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true})); + auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true})); Src = Src.And(0x1F); auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC); - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags); Ref Tmp {}; // Insert the incoming value across the temporary 64bit source @@ -2221,7 +2227,7 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) { // rather than zeroes. Ref Res = _Lshr(OpSize::i64Bit, Tmp, Src.Ref()); - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); // Our new CF will be bit (Shift - 1) of the source. 32-bit Lshr masks the // same as x86, but if we constant fold we must mask ourselves. @@ -2245,7 +2251,7 @@ void OpDispatchBuilder::RCLOp1Bit(OpcodeArgs) { // Calculate flags early. CalculateDeferredFlags(); - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags); const auto Size = GetSrcBitSize(Op); const auto OpSize = Size == 64 ? OpSize::i64Bit : OpSize::i32Bit; auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC); @@ -2261,7 +2267,7 @@ void OpDispatchBuilder::RCLOp1Bit(OpcodeArgs) { // Top two MSBs is CF and top bit of result SetRFLAG(_Xor(OpSize, Res, Dest), Size - 1, true); - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); } void OpDispatchBuilder::RCLOp(OpcodeArgs) { @@ -2277,7 +2283,7 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) { // Calculate flags early. CalculateDeferredFlags(); - Ref Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); const auto OpSize = OpSizeFromSrc(Op); uint64_t Const; @@ -2289,7 +2295,7 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) { } // Res = Src << Shift - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); Ref Res = _Lshl(OpSize, Dest, Src); auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC); @@ -2311,7 +2317,7 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) { SetRFLAG(NewOF, Size - 1, true); } - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); return; } @@ -2320,10 +2326,10 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) { Op, SrcMasked, [this, Op, Size, OpSize]() { // Rematerialized to avoid crossblock liveness - Ref Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); // Res = Src << Shift - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); Ref Res = _Lshl(OpSize, Dest, Src); auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC); @@ -2350,7 +2356,7 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) { auto NewOF = _XorShift(OpSize, Res, NewCF, ShiftType::LSL, Size - 1); SetRFLAG(NewOF, Size - 1, true); - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); }, OpSizeFromSrc(Op) == OpSize::i32Bit ? std::make_optional(&OpDispatchBuilder::ZeroShiftResult) : std::nullopt); } @@ -2361,16 +2367,16 @@ void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) { const auto Size = GetSrcBitSize(Op); // x86 masks the shift by 0x3F or 0x1F depending on size of op - auto Src = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true})); + auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true})); Src = Src.And(0x1F); // CF only changes if we actually shifted. OF undefined if we didn't shift. // The result is unchanged if we didn't shift. So branch over the whole thing. Calculate_ShiftVariable(Op, Src.Ref(), [this, Op, Size]() { // Rematerialized to avoid crossblock liveness - auto Src = ARef(LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true})); + auto Src = ARef(LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true})); Src = Src.And(0x1F); - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags); auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_RAW_LOC); @@ -2394,7 +2400,7 @@ void OpDispatchBuilder::RCLSmallerOp(OpcodeArgs) { // Which we emulate with a _Ror Ref Res = _Ror(OpSize::i64Bit, Tmp, Src.Neg().Ref()); - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); // Our new CF is now at the bit position that we are shifting // Either 0 if CF hasn't changed (CF is living in bit 0) @@ -2421,7 +2427,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) { if (IsNonconstant) { // Because we mask explicitly with And/Bfe/Sbfe after, we can allow garbage here. - Src = ARef(LoadSource(GPRClass, Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true})); + Src = ARef(LoadSourceGPR(Op, Op->Src[SrcIndex], Op->Flags, {.AllowUpperGarbage = true})); } else { // Can only be an immediate // Masked by operand size @@ -2431,7 +2437,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) { if (Op->Dest.IsGPR()) { // When the destination is a GPR, we don't care about garbage in the upper bits. // Load the full register. - auto Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GetGPROpSize(), Op->Flags); + auto Dest = LoadSourceGPR_WithOpSize(Op, Op->Dest, GetGPROpSize(), Op->Flags); Value = Dest; // Get the bit selection from the src. We need to mask for 8/16-bit, but @@ -2463,13 +2469,13 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) { case BTAction::BTClear: { Dest = _Andn(LshrOpSize, Dest, BitSelect.MaskBit(LshrOpSize).Ref()); - StoreResult(GPRClass, Op, Dest, OpSize::iInvalid); + StoreResultGPR(Op, Dest, OpSize::iInvalid); break; } case BTAction::BTSet: { Dest = _Or(LshrOpSize, Dest, BitSelect.MaskBit(LshrOpSize).Ref()); - StoreResult(GPRClass, Op, Dest, OpSize::iInvalid); + StoreResultGPR(Op, Dest, OpSize::iInvalid); break; } @@ -2485,7 +2491,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) { SetRFLAG(Value, X86State::RFLAG_CF_RAW_LOC, Src.IsConstant ? Src.C : 0, true); CFInverted = true; - StoreResult(GPRClass, Op, Dest, OpSize::iInvalid); + StoreResultGPR(Op, Dest, OpSize::iInvalid); break; } } @@ -2505,7 +2511,7 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) { switch (Action) { case BTAction::BTNone: { - Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit); + Value = _LoadMemGPRAutoTSO(OpSize::i8Bit, Address, OpSize::i8Bit); break; } @@ -2516,10 +2522,10 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) { HandledLock = true; Value = _AtomicFetchCLR(OpSize::i8Bit, BitMask, LoadEffectiveAddress(this, Address, GetGPROpSize(), true)); } else { - Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit); + Value = _LoadMemGPRAutoTSO(OpSize::i8Bit, Address, OpSize::i8Bit); auto Modified = _Andn(OpSize::i64Bit, Value, BitMask); - _StoreMemAutoTSO(GPRClass, OpSize::i8Bit, Address, Modified, OpSize::i8Bit); + _StoreMemGPRAutoTSO(OpSize::i8Bit, Address, Modified, OpSize::i8Bit); } break; } @@ -2531,10 +2537,10 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) { HandledLock = true; Value = _AtomicFetchOr(OpSize::i8Bit, BitMask, LoadEffectiveAddress(this, Address, GetGPROpSize(), true)); } else { - Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit); + Value = _LoadMemGPRAutoTSO(OpSize::i8Bit, Address, OpSize::i8Bit); auto Modified = _Or(OpSize::i64Bit, Value, BitMask); - _StoreMemAutoTSO(GPRClass, OpSize::i8Bit, Address, Modified, OpSize::i8Bit); + _StoreMemGPRAutoTSO(OpSize::i8Bit, Address, Modified, OpSize::i8Bit); } break; } @@ -2546,10 +2552,10 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) { HandledLock = true; Value = _AtomicFetchXor(OpSize::i8Bit, BitMask, LoadEffectiveAddress(this, Address, GetGPROpSize(), true)); } else { - Value = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, Address, OpSize::i8Bit); + Value = _LoadMemGPRAutoTSO(OpSize::i8Bit, Address, OpSize::i8Bit); auto Modified = _Xor(OpSize::i64Bit, Value, BitMask); - _StoreMemAutoTSO(GPRClass, OpSize::i8Bit, Address, Modified, OpSize::i8Bit); + _StoreMemGPRAutoTSO(OpSize::i8Bit, Address, Modified, OpSize::i8Bit); } break; } @@ -2567,8 +2573,8 @@ void OpDispatchBuilder::BTOp(OpcodeArgs, uint32_t SrcIndex, BTAction Action) { void OpDispatchBuilder::IMUL1SrcOp(OpcodeArgs) { /* We're just going to sign-extend the non-garbage anyway.. */ - Ref Src1 = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); - Ref Src2 = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src1 = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src2 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); const auto Size = OpSizeFromSrc(Op); const auto SizeBits = IR::OpSizeAsBits(Size); @@ -2600,13 +2606,13 @@ void OpDispatchBuilder::IMUL1SrcOp(OpcodeArgs) { default: FEX_UNREACHABLE; } - StoreResult(GPRClass, Op, Dest, OpSize::iInvalid); + StoreResultGPR(Op, Dest, OpSize::iInvalid); CalculateFlags_MUL(Size, Dest, ResultHigh); } void OpDispatchBuilder::IMUL2SrcOp(OpcodeArgs) { - Ref Src1 = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); - Ref Src2 = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); const auto Size = OpSizeFromSrc(Op); const auto SizeBits = IR::OpSizeAsBits(Size); @@ -2639,7 +2645,7 @@ void OpDispatchBuilder::IMUL2SrcOp(OpcodeArgs) { default: FEX_UNREACHABLE; } - StoreResult(GPRClass, Op, Dest, OpSize::iInvalid); + StoreResultGPR(Op, Dest, OpSize::iInvalid); CalculateFlags_MUL(Size, Dest, ResultHigh); } @@ -2647,7 +2653,7 @@ void OpDispatchBuilder::IMULOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); const auto SizeBits = IR::OpSizeAsBits(Size); - Ref Src1 = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src1 = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); Ref Src2 = LoadGPRRegister(X86State::REG_RAX); if (Size != OpSize::i64Bit) { @@ -2699,7 +2705,7 @@ void OpDispatchBuilder::MULOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); const auto SizeBits = IR::OpSizeAsBits(Size); - Ref Src1 = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src1 = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); Ref Src2 = LoadGPRRegister(X86State::REG_RAX); Ref Result {}; @@ -2764,9 +2770,9 @@ void OpDispatchBuilder::NOTOp(OpcodeArgs) { _AtomicXor(Size, MaskConst, DestMem); } else if (!Op->Dest.IsGPR()) { // GPR version plays fast and loose with sizes, be safe for memory tho. - Ref Src = LoadSource(GPRClass, Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceGPR(Op, Op->Dest, Op->Flags); Src = _Xor(OpSize::i64Bit, Src, MaskConst); - StoreResult(GPRClass, Op, Src, OpSize::iInvalid); + StoreResultGPR(Op, Src, OpSize::iInvalid); } else { // Specially handle high bits so we can invert in place with the correct // mask and a larger type. @@ -2780,7 +2786,7 @@ void OpDispatchBuilder::NOTOp(OpcodeArgs) { // Always load full size, we explicitly want the upper bits to get the // insert behaviour for free/implicitly. const auto GPRSize = GetGPROpSize(); - Ref Src = LoadSource_WithOpSize(GPRClass, Op, Dest, GPRSize, Op->Flags); + Ref Src = LoadSourceGPR_WithOpSize(Op, Dest, GPRSize, Op->Flags); // For 8/16-bit, use 64-bit invert so we invert in place, while getting // insert behaviour. For 32-bit, use 32-bit invert to zero the upper bits. @@ -2795,13 +2801,13 @@ void OpDispatchBuilder::NOTOp(OpcodeArgs) { // Always store 64-bit, the Not/Xor correctly handle the upper bits and this // way we can delete the store. - StoreResult_WithOpSize(GPRClass, Op, Dest, Src, GPRSize, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Dest, Src, GPRSize, OpSize::iInvalid); } } void OpDispatchBuilder::XADDOp(OpcodeArgs) { - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false}); - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags); Ref Result; if (Op->Dest.IsGPR()) { @@ -2809,23 +2815,23 @@ void OpDispatchBuilder::XADDOp(OpcodeArgs) { Result = CalculateFlags_ADD(OpSizeFromSrc(Op), Dest, Src); // Previous value in dest gets stored in src - StoreResult(GPRClass, Op, Op->Src[0], Dest, OpSize::iInvalid); + StoreResultGPR(Op, Op->Src[0], Dest, OpSize::iInvalid); // Calculated value gets stored in dst (order is important if dst is same as src) - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } else { HandledLock = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK; Dest = AppendSegmentOffset(Dest, Op->Flags); auto Before = _AtomicFetchAdd(OpSizeFromSrc(Op), Src, Dest); CalculateFlags_ADD(OpSizeFromSrc(Op), Before, Src); - StoreResult(GPRClass, Op, Op->Src[0], Before, OpSize::iInvalid); + StoreResultGPR(Op, Op->Src[0], Before, OpSize::iInvalid); } } void OpDispatchBuilder::PopcountOp(OpcodeArgs) { - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = CTX->HostFeatures.SupportsCSSC || GetSrcSize(Op) >= 4}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = CTX->HostFeatures.SupportsCSSC || GetSrcSize(Op) >= 4}); Src = _Popcount(OpSizeFromSrc(Op), Src); - StoreResult(GPRClass, Op, Src, OpSize::iInvalid); + StoreResultGPR(Op, Src, OpSize::iInvalid); // We need to set ZF while clearing the rest of NZCV. The result of a popcount // is in the range [0, 63]. In particular, it is always positive. So a @@ -2956,7 +2962,7 @@ void OpDispatchBuilder::XLATOp(OpcodeArgs) { Ref Offset = LoadGPRRegister(X86State::REG_RAX, OpSize::i8Bit); AddressMode A = {.Base = Src, .Index = Offset, .AddrSize = OpSize::i64Bit}; - auto Res = _LoadMemAutoTSO(GPRClass, OpSize::i8Bit, A, OpSize::i8Bit); + auto Res = _LoadMemGPRAutoTSO(OpSize::i8Bit, A, OpSize::i8Bit); StoreGPRRegister(X86State::REG_RAX, Res, OpSize::i8Bit); } @@ -2967,23 +2973,23 @@ void OpDispatchBuilder::ReadSegmentReg(OpcodeArgs, OpDispatchBuilder::Segment Se const auto Size = OpSizeFromSrc(Op); Ref Src {}; if (Seg == Segment::FS) { - Src = _LoadContext(Size, GPRClass, offsetof(FEXCore::Core::CPUState, fs_cached)); + Src = _LoadContextGPR(Size, offsetof(FEXCore::Core::CPUState, fs_cached)); } else { - Src = _LoadContext(Size, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached)); + Src = _LoadContextGPR(Size, offsetof(FEXCore::Core::CPUState, gs_cached)); } - StoreResult(GPRClass, Op, Src, OpSize::iInvalid); + StoreResultGPR(Op, Src, OpSize::iInvalid); } void OpDispatchBuilder::WriteSegmentReg(OpcodeArgs, OpDispatchBuilder::Segment Seg) { // Documentation claims that the 32-bit version of this instruction inserts in to the lower 32-bits of the segment // This is incorrect and it instead zero extends the 32-bit value to 64-bit const auto Size = OpSizeFromDst(Op); - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags); if (Seg == Segment::FS) { - _StoreContext(Size, GPRClass, Src, offsetof(FEXCore::Core::CPUState, fs_cached)); + _StoreContextGPR(Size, Src, offsetof(FEXCore::Core::CPUState, fs_cached)); } else { - _StoreContext(Size, GPRClass, Src, offsetof(FEXCore::Core::CPUState, gs_cached)); + _StoreContextGPR(Size, Src, offsetof(FEXCore::Core::CPUState, gs_cached)); } } @@ -3011,7 +3017,7 @@ void OpDispatchBuilder::EnterOp(OpcodeArgs) { if (Level > 0) { for (uint8_t i = 1; i < Level; ++i) { auto MemLoc = Sub(GPRSize, OldBP, i * IR::OpSizeToSize(OperandSize)); - auto Mem = _LoadMem(GPRClass, OperandSize, MemLoc, OperandSize); + auto Mem = _LoadMemGPR(OperandSize, MemLoc, OperandSize); NewSP = PushValue(OperandSize, Mem); } NewSP = PushValue(OperandSize, temp_RBP); @@ -3022,7 +3028,7 @@ void OpDispatchBuilder::EnterOp(OpcodeArgs) { } void OpDispatchBuilder::SGDTOp(OpcodeArgs) { - auto DestAddress = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false}); + auto DestAddress = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false}); // Store an emulated value in the format of: // uint16_t Limit; @@ -3040,12 +3046,12 @@ void OpDispatchBuilder::SGDTOp(OpcodeArgs) { GDTStoreSize = OpSize::i32Bit; } - _StoreMemAutoTSO(GPRClass, OpSize::i16Bit, DestAddress, Constant(0)); - _StoreMemAutoTSO(GPRClass, GDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, Constant(GDTAddress)); + _StoreMemGPRAutoTSO(OpSize::i16Bit, DestAddress, Constant(0)); + _StoreMemGPRAutoTSO(GDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, Constant(GDTAddress)); } void OpDispatchBuilder::SIDTOp(OpcodeArgs) { - auto DestAddress = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false}); + auto DestAddress = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false}); // See SGDTOp, matches Linux in reported values uint64_t IDTAddress = 0xFFFFFE0000000000ULL; @@ -3056,8 +3062,8 @@ void OpDispatchBuilder::SIDTOp(OpcodeArgs) { IDTStoreSize = OpSize::i32Bit; } - _StoreMemAutoTSO(GPRClass, OpSize::i16Bit, DestAddress, Constant(0xfff)); - _StoreMemAutoTSO(GPRClass, IDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, Constant(IDTAddress)); + _StoreMemGPRAutoTSO(OpSize::i16Bit, DestAddress, Constant(0xfff)); + _StoreMemGPRAutoTSO(IDTStoreSize, AddressMode {.Base = DestAddress, .Offset = 2, .AddrSize = OpSize::i64Bit}, Constant(IDTAddress)); } void OpDispatchBuilder::SMSWOp(OpcodeArgs) { @@ -3087,7 +3093,7 @@ void OpDispatchBuilder::SMSWOp(OpcodeArgs) { if (!IsMemDst && DstSize == OpSize::i32Bit) { // Special-case version of `smsw ebx`. This instruction does an insert in to the lower 32-bits on 64-bit hosts. // Override and insert. - auto Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GetGPROpSize(), Op->Flags); + auto Dest = LoadSourceGPR_WithOpSize(Op, Op->Dest, GetGPROpSize(), Op->Flags); Const = _Bfi(OpSize::i64Bit, 32, 0, Dest, Const); DstSize = OpSize::i64Bit; } @@ -3100,7 +3106,7 @@ void OpDispatchBuilder::SMSWOp(OpcodeArgs) { DstSize = OpSize::i16Bit; } - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Const, DstSize, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, Const, DstSize, OpSize::iInvalid); } OpDispatchBuilder::CycleCounterPair OpDispatchBuilder::CycleCounter(bool SelfSynchronizingLoads) { @@ -3139,7 +3145,7 @@ void OpDispatchBuilder::INCOp(OpcodeArgs) { Ref DestAddress = MakeSegmentAddress(Op, Op->Dest); Dest = _AtomicFetchAdd(OpSizeFromSrc(Op), Constant(1), DestAddress); } else { - Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32}); + Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32}); } CalculateDeferredFlags(); @@ -3162,7 +3168,7 @@ void OpDispatchBuilder::INCOp(OpcodeArgs) { } if (!IsLocked) { - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } } @@ -3180,7 +3186,7 @@ void OpDispatchBuilder::DECOp(OpcodeArgs) { // Use Add instead of Sub to avoid a NEG Dest = _AtomicFetchAdd(OpSizeFromSrc(Op), Constant(Size == 64 ? -1 : ((1ULL << Size) - 1)), DestAddress); } else { - Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32}); + Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= 32}); } CalculateDeferredFlags(); @@ -3203,7 +3209,7 @@ void OpDispatchBuilder::DECOp(OpcodeArgs) { } if (!IsLocked) { - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } } @@ -3219,13 +3225,13 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) { if (!Repeat) { // Src is used only for a store of the same size so allow garbage - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); // Only ES prefix Ref Dest = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true); // Store to memory where RDI points - _StoreMemAutoTSO(GPRClass, Size, Dest, Src, Size); + _StoreMemGPRAutoTSO(Size, Dest, Src, Size); // Offset the pointer Ref TailDest = LoadGPRRegister(X86State::REG_RDI); @@ -3234,7 +3240,7 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) { // FEX doesn't support partial faulting REP instructions. // Converting this to a `MemSet` IR op optimizes this quite significantly in our codegen. // If FEX is to gain support for faulting REP instructions, then this implementation needs to change significantly. - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags); Ref Dest = LoadGPRRegister(X86State::REG_RDI); // Only ES prefix @@ -3293,10 +3299,10 @@ void OpDispatchBuilder::MOVSOp(OpcodeArgs) { Ref RSI = MakeSegmentAddress(X86State::REG_RSI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX); Ref RDI = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true); - auto Src = _LoadMemAutoTSO(GPRClass, Size, RSI, Size); + auto Src = _LoadMemGPRAutoTSO(Size, RSI, Size); // Store to memory where RDI points - _StoreMemAutoTSO(GPRClass, Size, RDI, Src, Size); + _StoreMemGPRAutoTSO(Size, RDI, Src, Size); auto PtrDir = LoadDir(IR::OpSizeToSize(Size)); RSI = Add(OpSize::i64Bit, RSI, PtrDir); @@ -3323,8 +3329,8 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) { // Only ES prefix Ref Dest_RDI = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true); - auto Src1 = _LoadMemAutoTSO(GPRClass, Size, Dest_RDI, Size); - auto Src2 = _LoadMemAutoTSO(GPRClass, Size, Dest_RSI, Size); + auto Src1 = _LoadMemGPRAutoTSO(Size, Dest_RDI, Size); + auto Src2 = _LoadMemGPRAutoTSO(Size, Dest_RSI, Size); CalculateFlags_SUB(OpSizeFromSrc(Op), Src2, Src1); @@ -3369,8 +3375,8 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) { // Only ES prefix Ref Dest_RDI = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true); - auto Src1 = _LoadMemAutoTSO(GPRClass, Size, Dest_RDI, Size); - auto Src2 = _LoadMem(GPRClass, Size, Dest_RSI, Size); + auto Src1 = _LoadMemGPRAutoTSO(Size, Dest_RDI, Size); + auto Src2 = _LoadMemGPR(Size, Dest_RSI, Size); // We'll calculate PF/AF after the loop, so use them as temporaries here. StoreRegister(Core::CPUState::PF_AS_GREG, false, Src1); @@ -3440,9 +3446,9 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) { if (!Repeat) { Ref Dest_RSI = MakeSegmentAddress(X86State::REG_RSI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX); - auto Src = _LoadMemAutoTSO(GPRClass, Size, Dest_RSI, Size); + auto Src = _LoadMemGPRAutoTSO(Size, Dest_RSI, Size); - StoreResult(GPRClass, Op, Src, OpSize::iInvalid); + StoreResultGPR(Op, Src, OpSize::iInvalid); // Offset the pointer Ref TailDest_RSI = LoadGPRRegister(X86State::REG_RSI); @@ -3481,9 +3487,9 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) { { Ref Dest_RSI = MakeSegmentAddress(X86State::REG_RSI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX); - auto Src = _LoadMemAutoTSO(GPRClass, Size, Dest_RSI, Size); + auto Src = _LoadMemGPRAutoTSO(Size, Dest_RSI, Size); - StoreResult(GPRClass, Op, Src, OpSize::iInvalid); + StoreResultGPR(Op, Src, OpSize::iInvalid); Ref TailCounter = LoadGPRRegister(X86State::REG_RCX); Ref TailDest_RSI = LoadGPRRegister(X86State::REG_RSI); @@ -3523,8 +3529,8 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) { if (!Repeat) { Ref Dest_RDI = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true); - auto Src1 = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); - auto Src2 = _LoadMemAutoTSO(GPRClass, Size, Dest_RDI, Size); + auto Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + auto Src2 = _LoadMemGPRAutoTSO(Size, Dest_RDI, Size); CalculateFlags_SUB(OpSizeFromSrc(Op), Src1, Src2); @@ -3561,8 +3567,8 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) { { Ref Dest_RDI = MakeSegmentAddress(X86State::REG_RDI, 0, X86Tables::DecodeFlags::FLAG_ES_PREFIX, true); - auto Src1 = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); - auto Src2 = _LoadMemAutoTSO(GPRClass, Size, Dest_RDI, Size); + auto Src1 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + auto Src2 = _LoadMemGPRAutoTSO(Size, Dest_RDI, Size); CalculateFlags_SUB(OpSizeFromSrc(Op), Src1, Src2); @@ -3607,10 +3613,10 @@ void OpDispatchBuilder::BSWAPOp(OpcodeArgs) { // BSWAP of 16bit is undef. ZEN+ causes the lower 16bits to get zero'd Dest = Constant(0); } else { - Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GetGPROpSize(), Op->Flags); + Dest = LoadSourceGPR_WithOpSize(Op, Op->Dest, GetGPROpSize(), Op->Flags); Dest = _Rev(Size, Dest); } - StoreResult(GPRClass, Op, Dest, OpSize::iInvalid); + StoreResultGPR(Op, Dest, OpSize::iInvalid); } void OpDispatchBuilder::PUSHFOp(OpcodeArgs) { @@ -3647,10 +3653,10 @@ void OpDispatchBuilder::NEGOp(OpcodeArgs) { Ref Dest = _AtomicFetchNeg(Size, DestMem); CalculateFlags_SUB(Size, ZeroConst, Dest); } else { - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); Ref Result = CalculateFlags_SUB(Size, ZeroConst, Dest); - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } } @@ -3659,7 +3665,7 @@ void OpDispatchBuilder::DIVOp(OpcodeArgs) { auto Size = OpSizeFromSrc(Op); // This loads the divisor. 32-bit/64-bit paths mask inside the JIT, 8/16 do not. - Ref Divisor = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= OpSize::i32Bit}); + Ref Divisor = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = Size >= OpSize::i32Bit}); if (Size == OpSize::i64Bit && !Is64BitMode) { LogMan::Msg::EFmt("Doesn't exist in 32bit mode"); @@ -3697,7 +3703,7 @@ void OpDispatchBuilder::DIVOp(OpcodeArgs) { void OpDispatchBuilder::IDIVOp(OpcodeArgs) { // This loads the divisor - Ref Divisor = LoadSource(GPRClass, Op, Op->Dest, Op->Flags); + Ref Divisor = LoadSourceGPR(Op, Op->Dest, Op->Flags); const auto GPRSize = GetGPROpSize(); auto Size = OpSizeFromSrc(Op); @@ -3741,8 +3747,8 @@ void OpDispatchBuilder::IDIVOp(OpcodeArgs) { void OpDispatchBuilder::BSFOp(OpcodeArgs) { const auto GPRSize = GetGPROpSize(); const auto DstSize = OpSizeFromDst(Op) == OpSize::i16Bit ? OpSize::i16Bit : GPRSize; - Ref Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, DstSize, Op->Flags, {.AllowUpperGarbage = true}); - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + Ref Dest = LoadSourceGPR_WithOpSize(Op, Op->Dest, DstSize, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); // Find the LSB of this source auto Result = _FindLSB(OpSizeFromSrc(Op), Src); @@ -3757,14 +3763,14 @@ void OpDispatchBuilder::BSFOp(OpcodeArgs) { // hardware satisfies it. We provide the stronger AMD behaviour as // applications might rely on that in the wild. auto SelectOp = NZCVSelect(GPRSize, CondClass::EQ, Dest, Result); - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, SelectOp, DstSize, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, SelectOp, DstSize, OpSize::iInvalid); } void OpDispatchBuilder::BSROp(OpcodeArgs) { const auto GPRSize = GetGPROpSize(); const auto DstSize = OpSizeFromDst(Op) == OpSize::i16Bit ? OpSize::i16Bit : GPRSize; - Ref Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, DstSize, Op->Flags, {.AllowUpperGarbage = true}); - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + Ref Dest = LoadSourceGPR_WithOpSize(Op, Op->Dest, DstSize, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); // Find the MSB of this source auto Result = _FindMSB(OpSizeFromSrc(Op), Src); @@ -3775,7 +3781,7 @@ void OpDispatchBuilder::BSROp(OpcodeArgs) { // If Src was zero then the destination doesn't get modified auto SelectOp = NZCVSelect(GPRSize, CondClass::EQ, Dest, Result); - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, SelectOp, DstSize, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, SelectOp, DstSize, OpSize::iInvalid); } void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) { @@ -3799,7 +3805,7 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) { if (Op->Dest.IsGPR()) { // This is our source register - Ref Src2 = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src2 = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); Ref Src3 = LoadGPRRegister(X86State::REG_RAX); // If the destination is also the accumulator, we get some algebraic @@ -3811,10 +3817,10 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) { Ref Src1Lower {}; if (GPRSize == OpSize::i64Bit && Size == OpSize::i32Bit) { - Src1 = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags, {.AllowUpperGarbage = true}); + Src1 = LoadSourceGPR_WithOpSize(Op, Op->Dest, GPRSize, Op->Flags, {.AllowUpperGarbage = true}); Src1Lower = Trivial ? Src1 : _Bfe(GPRSize, IR::OpSizeAsBits(Size), 0, Src1); } else { - Src1 = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, Size, Op->Flags, {.AllowUpperGarbage = true}); + Src1 = LoadSourceGPR_WithOpSize(Op, Op->Dest, Size, Op->Flags, {.AllowUpperGarbage = true}); Src1Lower = Src1; } @@ -3845,12 +3851,12 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) { // Store in to GPR Dest if (GPRSize == OpSize::i64Bit && Size == OpSize::i32Bit) { - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, DestResult, GPRSize, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, DestResult, GPRSize, OpSize::iInvalid); } else { - StoreResult(GPRClass, Op, DestResult, OpSize::iInvalid); + StoreResultGPR(Op, DestResult, OpSize::iInvalid); } } else { - Ref Src2 = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceGPR(Op, Op->Src[0], Op->Flags); HandledLock = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK; auto Src3 = LoadGPRRegister(X86State::REG_RAX); @@ -4006,9 +4012,9 @@ Ref OpDispatchBuilder::GetSegment(uint32_t Flags, uint32_t DefaultPrefix, bool O if (Is64BitMode) { if (Prefix == FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX) { - return _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs_cached)); + return _LoadContextGPR(GPRSize, offsetof(FEXCore::Core::CPUState, fs_cached)); } else if (Prefix == FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX) { - return _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached)); + return _LoadContextGPR(GPRSize, offsetof(FEXCore::Core::CPUState, gs_cached)); } // If there was any other segment in 64bit then it is ignored } else { @@ -4023,22 +4029,22 @@ Ref OpDispatchBuilder::GetSegment(uint32_t Flags, uint32_t DefaultPrefix, bool O [[likely]] case FEXCore::X86Tables::DecodeFlags::FLAG_NO_PREFIX: return nullptr; case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX: - SegmentResult = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, es_cached)); + SegmentResult = _LoadContextGPR(GPRSize, offsetof(FEXCore::Core::CPUState, es_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX: - SegmentResult = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, cs_cached)); + SegmentResult = _LoadContextGPR(GPRSize, offsetof(FEXCore::Core::CPUState, cs_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX: - SegmentResult = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, ss_cached)); + SegmentResult = _LoadContextGPR(GPRSize, offsetof(FEXCore::Core::CPUState, ss_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX: - SegmentResult = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, ds_cached)); + SegmentResult = _LoadContextGPR(GPRSize, offsetof(FEXCore::Core::CPUState, ds_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX: - SegmentResult = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, fs_cached)); + SegmentResult = _LoadContextGPR(GPRSize, offsetof(FEXCore::Core::CPUState, fs_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX: - SegmentResult = _LoadContext(GPRSize, GPRClass, offsetof(FEXCore::Core::CPUState, gs_cached)); + SegmentResult = _LoadContextGPR(GPRSize, offsetof(FEXCore::Core::CPUState, gs_cached)); break; default: FEX_UNREACHABLE; } @@ -4147,8 +4153,8 @@ void OpDispatchBuilder::UpdatePrefixFromSegment(Ref Segment, uint32_t SegmentReg auto GDT = _Bfe(OpSize::i32Bit, 1, 2, Segment); // Fun quirk, if we mask the selector then it is premultiplied by 8 which we need to do for accessing anyway. auto SegmentOffset = _And(OpSize::i32Bit, Segment, _Constant(0xfff8)); - Ref SegmentBase = _LoadContextIndexed(GDT, OpSize::i64Bit, offsetof(FEXCore::Core::CPUState, segment_arrays[0]), 8, GPRClass); - Ref NewSegment = _LoadMem(GPRClass, OpSize::i64Bit, SegmentBase, SegmentOffset, OpSize::i8Bit, MemOffsetType::UXTW, 1); + Ref SegmentBase = _LoadContextGPRIndexed(GDT, OpSize::i64Bit, offsetof(FEXCore::Core::CPUState, segment_arrays[0]), 8); + Ref NewSegment = _LoadMemGPR(OpSize::i64Bit, SegmentBase, SegmentOffset, OpSize::i8Bit, MemOffsetType::UXTW, 1); CheckLegacySegmentWrite(NewSegment, SegmentReg); // Extract the 32-bit base from the GDT segment. @@ -4159,22 +4165,22 @@ void OpDispatchBuilder::UpdatePrefixFromSegment(Ref Segment, uint32_t SegmentReg switch (SegmentReg) { case FEXCore::X86Tables::DecodeFlags::FLAG_ES_PREFIX: - _StoreContext(OpSize::i32Bit, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, es_cached)); + _StoreContextGPR(OpSize::i32Bit, NewSegment, offsetof(FEXCore::Core::CPUState, es_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_CS_PREFIX: - _StoreContext(OpSize::i32Bit, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, cs_cached)); + _StoreContextGPR(OpSize::i32Bit, NewSegment, offsetof(FEXCore::Core::CPUState, cs_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_SS_PREFIX: - _StoreContext(OpSize::i32Bit, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ss_cached)); + _StoreContextGPR(OpSize::i32Bit, NewSegment, offsetof(FEXCore::Core::CPUState, ss_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX: - _StoreContext(OpSize::i32Bit, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, ds_cached)); + _StoreContextGPR(OpSize::i32Bit, NewSegment, offsetof(FEXCore::Core::CPUState, ds_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_FS_PREFIX: - _StoreContext(OpSize::i32Bit, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, fs_cached)); + _StoreContextGPR(OpSize::i32Bit, NewSegment, offsetof(FEXCore::Core::CPUState, fs_cached)); break; case FEXCore::X86Tables::DecodeFlags::FLAG_GS_PREFIX: - _StoreContext(OpSize::i32Bit, GPRClass, NewSegment, offsetof(FEXCore::Core::CPUState, gs_cached)); + _StoreContextGPR(OpSize::i32Bit, NewSegment, offsetof(FEXCore::Core::CPUState, gs_cached)); break; default: break; // Do nothing } @@ -4414,9 +4420,9 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl _StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, Src, MemStoreDst); } else { // For X87 extended doubles, split before storing - _StoreMem(FPRClass, OpSize::i64Bit, MemStoreDst, Src, Align); + _StoreMemFPR(OpSize::i64Bit, MemStoreDst, Src, Align); auto Upper = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, Src, 1); - _StoreMem(GPRClass, OpSize::i16Bit, Upper, MemStoreDst, Constant(8), std::min(Align, OpSize::i64Bit), MemOffsetType::SXTX, 1); + _StoreMemGPR(OpSize::i16Bit, Upper, MemStoreDst, Constant(8), std::min(Align, OpSize::i64Bit), MemOffsetType::SXTX, 1); } } else { _StoreMemAutoTSO(Class, OpSize, A, Src, Align == OpSize::iInvalid ? OpSize : Align); @@ -4468,14 +4474,14 @@ void OpDispatchBuilder::UnhandledOp(OpcodeArgs) { void OpDispatchBuilder::MOVGPROp(OpcodeArgs, uint32_t SrcIndex) { // StoreResult will store with the same size as the input, so we allow upper // garbage on the input. The zero extension would be pointless. - Ref Src = LoadSource(GPRClass, Op, Op->Src[SrcIndex], Op->Flags, {.Align = OpSize::i8Bit, .AllowUpperGarbage = true}); - StoreResult(GPRClass, Op, Src, OpSize::i8Bit); + Ref Src = LoadSourceGPR(Op, Op->Src[SrcIndex], Op->Flags, {.Align = OpSize::i8Bit, .AllowUpperGarbage = true}); + StoreResultGPR(Op, Src, OpSize::i8Bit); } void OpDispatchBuilder::MOVGPRImmediate(OpcodeArgs) { Ref Src {}; if (Op->Src[0].Data.Literal.Size <= 4) { - Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit, .AllowUpperGarbage = true}); + Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit, .AllowUpperGarbage = true}); } else { // 8-byte literal is special cased. const uint64_t Lower = Op->Src[0].Literal(); @@ -4483,12 +4489,12 @@ void OpDispatchBuilder::MOVGPRImmediate(OpcodeArgs) { const uint64_t Combined = (Upper << 32) | Lower; Src = _Constant(Combined); } - StoreResult(GPRClass, Op, Src, OpSize::i8Bit); + StoreResultGPR(Op, Src, OpSize::i8Bit); } void OpDispatchBuilder::MOVGPRNTOp(OpcodeArgs) { - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit}); - StoreResult(GPRClass, Op, Src, OpSize::i8Bit, MemoryAccessType::STREAM); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit}); + StoreResultGPR(Op, Src, OpSize::i8Bit, MemoryAccessType::STREAM); } void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx) { @@ -4508,7 +4514,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I FlushRegisterCache(); // Move 0 into the register - StoreResult(GPRClass, Op, Constant(0), OpSize::iInvalid); + StoreResultGPR(Op, Constant(0), OpSize::iInvalid); return; } @@ -4521,7 +4527,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I } // X86 basic ALU ops just do the operation between the destination and a single source - Ref Src = LoadSource(GPRClass, Op, Op->Src[SrcIdx], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[SrcIdx], Op->Flags, {.AllowUpperGarbage = true}); // Try to eliminate the masking after 8/16-bit operations with constants, by // promoting to a full size operation that preserves the upper bits. @@ -4557,7 +4563,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I DeriveOp(FetchOp, AtomicFetchOp, _AtomicFetchAdd(Size, Src, DestMem)); Dest = FetchOp; } else { - Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); } const auto OpSize = RoundedSize; @@ -4591,7 +4597,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I } if (!DestIsLockedMem(Op)) { - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, ResultSize, OpSize::iInvalid, MemoryAccessType::DEFAULT); + StoreResultGPR_WithOpSize(Op, Op->Dest, Result, ResultSize, OpSize::iInvalid, MemoryAccessType::DEFAULT); } } @@ -4692,10 +4698,10 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) { // We want to set RIP to the next instruction after INT3/INT1 auto NewRIP = GetRelocatedPC(Op); - _StoreContext(GPRSize, GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip)); + _StoreContextGPR(GPRSize, NewRIP, offsetof(FEXCore::Core::CPUState, rip)); } else if (Op->OP != 0xCE) { auto NewRIP = GetRelocatedPC(Op, -Op->InstSize); - _StoreContext(GPRSize, GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip)); + _StoreContextGPR(GPRSize, NewRIP, offsetof(FEXCore::Core::CPUState, rip)); } if (Op->OP == 0xCE) { // Conditional to only break if Overflow == 1 @@ -4710,7 +4716,7 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) { StartNewBlock(); auto NewRIP = GetRelocatedPC(Op); - _StoreContext(GPRSize, GPRClass, NewRIP, offsetof(FEXCore::Core::CPUState, rip)); + _StoreContextGPR(GPRSize, NewRIP, offsetof(FEXCore::Core::CPUState, rip)); Break(Reason); // Make sure to start a new block after ending this one @@ -4726,20 +4732,20 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) { void OpDispatchBuilder::TZCNT(OpcodeArgs) { // _FindTrailingZeroes ignores upper garbage so we don't need to mask - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); Src = _FindTrailingZeroes(OpSizeFromSrc(Op), Src); - StoreResult(GPRClass, Op, Src, OpSize::iInvalid); + StoreResultGPR(Op, Src, OpSize::iInvalid); CalculateFlags_ZCNT(OpSizeFromSrc(Op), Src); } void OpDispatchBuilder::LZCNT(OpcodeArgs) { // _CountLeadingZeroes clears upper garbage so we don't need to mask - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true}); auto Res = _CountLeadingZeroes(OpSizeFromSrc(Op), Src); - StoreResult(GPRClass, Op, Res, OpSize::iInvalid); + StoreResultGPR(Op, Res, OpSize::iInvalid); CalculateFlags_ZCNT(OpSizeFromSrc(Op), Res); } @@ -4747,19 +4753,19 @@ void OpDispatchBuilder::MOVBEOp(OpcodeArgs) { const auto GPRSize = GetGPROpSize(); const auto SrcSize = OpSizeFromSrc(Op); - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit}); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit}); if (DestIsMem(Op) || SrcSize != OpSize::i16Bit) { Src = _Rev(SrcSize, Src); - StoreResult(GPRClass, Op, Op->Dest, Src, OpSize::iInvalid); + StoreResultGPR(Op, Op->Dest, Src, OpSize::iInvalid); } else { Src = _Rev(std::max(OpSize::i32Bit, SrcSize), Src); // 16-bit does an insert. // Rev of 16-bit value as 32-bit replaces the result in the upper 16-bits of the result. // bfxil the 16-bit result in to the GPR. - Ref Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags); + Ref Dest = LoadSourceGPR_WithOpSize(Op, Op->Dest, GPRSize, Op->Flags); auto Result = _Bfxil(GPRSize, 16, 16, Dest, Src); - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, Result, GPRSize, OpSize::iInvalid); } } @@ -4846,12 +4852,12 @@ void OpDispatchBuilder::CLZeroOp(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref DestMem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false}); + Ref DestMem = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false}); _CacheLineZero(DestMem); } void OpDispatchBuilder::Prefetch(OpcodeArgs, bool ForStore, bool Stream, uint8_t Level) { - Ref DestMem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false}); + Ref DestMem = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false}); _Prefetch(ForStore, Stream, Level, DestMem, Invalid(), MemOffsetType::SXTX, 1); } @@ -4873,7 +4879,7 @@ void OpDispatchBuilder::RDTSCPOp(OpcodeArgs) { } void OpDispatchBuilder::RDPIDOp(OpcodeArgs) { - StoreResult(GPRClass, Op, _ProcessorID(), OpSize::iInvalid); + StoreResultGPR(Op, _ProcessorID(), OpSize::iInvalid); } void OpDispatchBuilder::CRC32(OpcodeArgs) { @@ -4885,17 +4891,17 @@ void OpDispatchBuilder::CRC32(OpcodeArgs) { // Destination GPR size is always 4 or 8 bytes depending on widening const auto DstSize = Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REX_WIDENING ? OpSize::i64Bit : OpSize::i32Bit; - Ref Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags); + Ref Dest = LoadSourceGPR_WithOpSize(Op, Op->Dest, GPRSize, Op->Flags); // Incoming memory is 8, 16, 32, or 64 Ref Src {}; if (Op->Src[0].IsGPR()) { - Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], GPRSize, Op->Flags); + Src = LoadSourceGPR_WithOpSize(Op, Op->Src[0], GPRSize, Op->Flags); } else { - Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit}); + Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit}); } auto Result = _CRC32(Dest, Src, OpSizeFromSrc(Op)); - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template @@ -4905,7 +4911,7 @@ void OpDispatchBuilder::RDRANDOp(OpcodeArgs) { return; } - StoreResult(GPRClass, Op, _RDRAND(Reseed), OpSize::iInvalid); + StoreResultGPR(Op, _RDRAND(Reseed), OpSize::iInvalid); // If the rng number is valid then NZCV is 0b0000, otherwise NZCV is 0b0100 auto CF_inv = GetRFLAG(X86State::RFLAG_ZF_RAW_LOC); @@ -4932,7 +4938,7 @@ void OpDispatchBuilder::BreakOp(OpcodeArgs, FEXCore::IR::BreakDefinition BreakDe // We don't actually support this instruction // Multiblock may hit it though - _StoreContext(GPRSize, GPRClass, GetRelocatedPC(Op, -Op->InstSize), offsetof(FEXCore::Core::CPUState, rip)); + _StoreContextGPR(GPRSize, GetRelocatedPC(Op, -Op->InstSize), offsetof(FEXCore::Core::CPUState, rip)); Break(BreakDefinition); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h index 194149173..79bbae000 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h @@ -1599,8 +1599,7 @@ private: StoreResult(FPRClass, Op, Operand, Src, Align, AccessType); } - void StoreResult(RegisterClassType Class, X86Tables::DecodedOp Op, Ref Src, OpSize Align, - MemoryAccessType AccessType = MemoryAccessType::DEFAULT); + void StoreResult(RegisterClassType Class, X86Tables::DecodedOp Op, Ref Src, OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT); void StoreResultGPR(X86Tables::DecodedOp Op, Ref Src, OpSize Align, MemoryAccessType AccessType = MemoryAccessType::DEFAULT) { StoreResult(GPRClass, Op, Src, Align, AccessType); } diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp index 10e4fb100..ed2223733 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp @@ -46,9 +46,9 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_LoadSource_WithOpSize( } if (NeedsHigh) { - return _LoadMemPairAutoTSO(FPRClass, OpSize::i128Bit, A, OpSize::i8Bit); + return _LoadMemPairFPRAutoTSO(OpSize::i128Bit, A, OpSize::i8Bit); } else { - return {.Low = _LoadMemAutoTSO(FPRClass, OpSize::i128Bit, A, OpSize::i8Bit)}; + return {.Low = _LoadMemFPRAutoTSO(OpSize::i128Bit, A, OpSize::i8Bit)}; } } } @@ -95,9 +95,9 @@ void OpDispatchBuilder::AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::Decode AddressMode A = DecodeAddress(Op, Operand, AccessType, false /* IsLoad */); if (Src.High) { - _StoreMemPairAutoTSO(FPRClass, OpSize::i128Bit, A, Src.Low, Src.High, OpSize::i8Bit); + _StoreMemPairFPRAutoTSO(OpSize::i128Bit, A, Src.Low, Src.High, OpSize::i8Bit); } else { - _StoreMemAutoTSO(FPRClass, OpSize::i128Bit, A, Src.Low, OpSize::i8Bit); + _StoreMemFPRAutoTSO(OpSize::i128Bit, A, Src.Low, OpSize::i8Bit); } } } @@ -151,13 +151,13 @@ void OpDispatchBuilder::AVX128_VMOVScalarImpl(OpcodeArgs, IR::OpSize ElementSize AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Result, .High = High}); } else if (Op->Dest.IsGPR()) { // VMOVSS/SD xmm1, mem32/mem64 - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], ElementSize, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[1], ElementSize, Op->Flags); auto High = LoadZeroVector(OpSize::i128Bit); AVX128_StoreResult_WithOpSize(Op, Op->Dest, RefPair {.Low = Src, .High = High}); } else { // VMOVSS/SD mem32/mem64, xmm1 auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src.Low, ElementSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Src.Low, ElementSize, OpSize::iInvalid); } } @@ -351,7 +351,7 @@ void OpDispatchBuilder::AVX128_MOVVectorNT(OpcodeArgs) { if (Op->Dest.IsGPR()) { ///< MOVNTDQA load non-temporal comes from SSE4.1 and is extended by AVX/AVX2. RefPair Src {}; - Ref SrcAddr = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false}); + Ref SrcAddr = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false}); Src.Low = _VLoadNonTemporal(OpSize::i128Bit, SrcAddr, 0); if (Is128Bit) { @@ -362,7 +362,7 @@ void OpDispatchBuilder::AVX128_MOVVectorNT(OpcodeArgs) { AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src); } else { auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit, MemoryAccessType::STREAM); - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false}); if (Is128Bit) { // Single store non-temporal for 128-bit operations. @@ -379,7 +379,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) { if (Op->Src[0].IsGPR()) { Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false); } else { - Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i64Bit, Op->Flags); + Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i64Bit, Op->Flags); } // This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 256bit @@ -390,7 +390,7 @@ void OpDispatchBuilder::AVX128_MOVQ(OpcodeArgs) { Src.High = ZeroVector; AVX128_StoreResult_WithOpSize(Op, Op->Dest, Src); } else { - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src.Low, OpSize::i64Bit, OpSize::i64Bit); + StoreResultFPR_WithOpSize(Op, Op->Dest, Src.Low, OpSize::i64Bit, OpSize::i64Bit); } } @@ -399,7 +399,7 @@ void OpDispatchBuilder::AVX128_VMOVLP(OpcodeArgs) { if (!Op->Dest.IsGPR()) { ///< VMOVLPS/PD mem64, xmm1 - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src1.Low, OpSize::i64Bit, OpSize::i64Bit); + StoreResultFPR_WithOpSize(Op, Op->Dest, Src1.Low, OpSize::i64Bit, OpSize::i64Bit); } else if (!Op->Src[1].IsGPR()) { ///< VMOVLPS/PD xmm1, xmm2, mem64 // Bits[63:0] come from Src2[63:0] @@ -463,7 +463,7 @@ void OpDispatchBuilder::AVX128_VMOVDDUP(OpcodeArgs) { // 128-bit operation only loads 8-bytes. // 256-bit operation loads a full 32-bytes. if (Is128Bit) { - Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i64Bit, Op->Flags); + Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i64Bit, Op->Flags); } else { Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, true); } @@ -558,18 +558,18 @@ void OpDispatchBuilder::AVX128_InsertCVTGPR_To_FPR(OpcodeArgs, IR::OpSize DstEle if (Op->Src[1].IsGPR()) { // If the source is a GPR then convert directly from the GPR. - auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Op->Src[1], GetGPROpSize(), Op->Flags); + auto Src2 = LoadSourceGPR_WithOpSize(Op, Op->Src[1], GetGPROpSize(), Op->Flags); Result.Low = _VSToFGPRInsert(OpSize::i128Bit, DstElementSize, SrcSize, Src1.Low, Src2, false); } else if (SrcSize != DstElementSize) { // If the source is from memory but the Source size and destination size aren't the same, // then it is more optimal to load in to a GPR and convert between GPR->FPR. // ARM GPR->FPR conversion supports different size source and destinations while FPR->FPR doesn't. - auto Src2 = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags); + auto Src2 = LoadSourceGPR(Op, Op->Src[1], Op->Flags); Result.Low = _VSToFGPRInsert(DstSize, DstElementSize, SrcSize, Src1.Low, Src2, false); } else { // In the case of cvtsi2s{s,d} where the source and destination are the same size, // then it is more optimal to load in to the FPR register directly and convert there. - auto Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + auto Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); // Always signed Result.Low = _VSToFVectorInsert(DstSize, DstElementSize, DstElementSize, Src1.Low, Src2, false, false); } @@ -589,11 +589,11 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSi if (Op->Src[0].IsGPR()) { Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false); } else { - Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcElementSize, Op->Flags); + Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcElementSize, Op->Flags); } Ref Result = CVTFPR_To_GPRImpl(Op, Src.Low, SrcElementSize, HostRoundingMode); - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) { @@ -636,7 +636,7 @@ void OpDispatchBuilder::AVX128_UCOMISx(OpcodeArgs, IR::OpSize ElementSize) { if (Op->Src[0].IsGPR()) { Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false); } else { - Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags); + Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); } Comiss(ElementSize, Src1.Low, Src2.Low); @@ -653,7 +653,7 @@ void OpDispatchBuilder::AVX128_VectorScalarInsertALU(OpcodeArgs, FEXCore::IR::IR if (Op->Src[1].IsGPR()) { Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false); } else { - Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags); + Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags); } // If OpSize == ElementSize then it only does the lower scalar op @@ -690,7 +690,7 @@ void OpDispatchBuilder::AVX128_InsertScalarFCMP(OpcodeArgs, IR::OpSize ElementSi if (Op->Src[1].IsGPR()) { Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false); } else { - Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags); + Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags); } const uint8_t CompType = Op->Src[2].Literal(); @@ -708,12 +708,12 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) { RefPair Result {}; if (Op->Src[0].IsGPR()) { // Loading from GPR and moving to Vector. - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], GetGPROpSize(), Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], GetGPROpSize(), Op->Flags); // zext to 128bit Result.Low = _VCastFromGPR(OpSize::i128Bit, OpSizeFromSrc(Op), Src); } else { // Loading from Memory as a scalar. Zero extend - Result.Low = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Result.Low = LoadSourceFPR(Op, Op->Src[0], Op->Flags); } Result.High = LoadZeroVector(OpSize::i128Bit); @@ -726,11 +726,11 @@ void OpDispatchBuilder::AVX128_MOVBetweenGPR_FPR(OpcodeArgs) { auto ElementSize = OpSizeFromDst(Op); // Extract element from GPR. Zero extending in the process. Src.Low = _VExtractToGPR(OpSizeFromSrc(Op), ElementSize, Src.Low, 0); - StoreResult(GPRClass, Op, Op->Dest, Src.Low, OpSize::iInvalid); + StoreResultGPR(Op, Op->Dest, Src.Low, OpSize::iInvalid); } else { // Storing first element to memory. - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false}); - _StoreMem(FPRClass, OpSizeFromDst(Op), Dest, Src.Low, OpSize::i8Bit); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false}); + _StoreMemFPR(OpSizeFromDst(Op), Dest, Src.Low, OpSize::i8Bit); } } } @@ -758,7 +758,7 @@ void OpDispatchBuilder::AVX128_PExtr(OpcodeArgs, IR::OpSize ElementSize) { const auto GPRSize = GetGPROpSize(); // Extract already zero extends the result. Ref Result = _VExtractToGPR(OpSize::i128Bit, OverridenElementSize, Src.Low, Index); - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, Result, GPRSize, OpSize::iInvalid); return; } @@ -779,7 +779,7 @@ void OpDispatchBuilder::AVX128_ExtendVectorElements(OpcodeArgs, IR::OpSize Eleme const auto SrcSize = OpSizeFromSrc(Op); const auto LoadSize = Is256Bit ? IR::SizeToOpSize(IR::OpSizeToSize(SrcSize) * 2) : SrcSize; - return LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags); + return LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags); } }; @@ -868,7 +868,7 @@ void OpDispatchBuilder::AVX128_MOVMSK(OpcodeArgs, IR::OpSize ElementSize) { auto GPRHigh = Mask8Byte(Src.High); GPR = _Orlshl(OpSize::i64Bit, GPRLow, GPRHigh, 2); } - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, GPR, GetGPROpSize(), OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize(), OpSize::iInvalid); } void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) { @@ -897,7 +897,7 @@ void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) { Result = _Orlshl(OpSize::i64Bit, Result, ResultHigh, 16); } - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, const X86Tables::DecodedOperand& Src1Op, @@ -910,7 +910,7 @@ void OpDispatchBuilder::AVX128_PINSRImpl(OpcodeArgs, IR::OpSize ElementSize, con if (Src2Op.IsGPR()) { // If the source is a GPR then convert directly from the GPR. - auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, GetGPROpSize(), Op->Flags); + auto Src2 = LoadSourceGPR_WithOpSize(Op, Src2Op, GetGPROpSize(), Op->Flags); Result.Low = _VInsGPR(OpSize::i128Bit, ElementSize, Index, Src1.Low, Src2); } else { // If loading from memory then we only load the element size @@ -1047,7 +1047,7 @@ void OpDispatchBuilder::AVX128_InsertScalar_CVT_Float_To_Float(OpcodeArgs, IR::O // Then zero extends the top 128-bit. const auto SrcSize = Op->Src[1].IsGPR() ? OpSize::i128Bit : SrcElementSize; auto Src1 = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, false); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true}); Ref Result = _VFToFScalarInsert(OpSize::i128Bit, DstElementSize, SrcElementSize, Src1.Low, Src2, false); AVX128_StoreResult_WithOpSize(Op, Op->Dest, AVX128_Zext(Result)); @@ -1076,7 +1076,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize } else { // Handle 64-bit memory source. // In the case of cvtps2pd xmm, m64. - Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags); + Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags); } RefPair Result {}; @@ -1154,7 +1154,7 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Int_To_Float(OpcodeArgs, IR::OpSize Sr // unnecessarily zero extend the vector. Otherwise, if // memory, then we want to load the element size exactly. const auto LoadSize = IR::SizeToOpSize(8 * (IR::OpSizeToSize(Size) / 16)); - return RefPair {.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags)}; + return RefPair {.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags)}; } else { return AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128Bit); } @@ -1304,7 +1304,7 @@ void OpDispatchBuilder::AVX128_InsertScalarRound(OpcodeArgs, IR::OpSize ElementS if (Op->Src[1].IsGPR()) { Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false); } else { - Src2.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags); + Src2.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags); } // If OpSize == ElementSize then it only does the lower scalar op @@ -1616,11 +1616,11 @@ void OpDispatchBuilder::AVX128_MASKMOV(OpcodeArgs) { // RDI source (DS prefix by default) auto MemDest = MakeSegmentAddress(X86State::REG_RDI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX); - Ref XMMReg = _LoadMem(FPRClass, Size, MemDest, OpSize::i8Bit); + Ref XMMReg = _LoadMemFPR(Size, MemDest, OpSize::i8Bit); // If the Mask element high bit is set then overwrite the element with the source, else keep the memory variant XMMReg = _VBSL(Size, MaskSrc.Low, VectorSrc.Low, XMMReg); - _StoreMem(FPRClass, Size, MemDest, XMMReg, OpSize::i8Bit); + _StoreMemFPR(Size, MemDest, XMMReg, OpSize::i8Bit); } void OpDispatchBuilder::AVX128_VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize) { @@ -1660,7 +1660,7 @@ void OpDispatchBuilder::AVX128_SaveAVXState(Ref MemBase) { for (uint32_t i = 0; i < NumRegs; i += 2) { RefPair Pair = LoadContextPair(OpSize::i128Bit, AVXHigh0Index + i); - _StoreMemPair(FPRClass, OpSize::i128Bit, Pair.Low, Pair.High, MemBase, i * 16 + 576); + _StoreMemPairFPR(OpSize::i128Bit, Pair.Low, Pair.High, MemBase, i * 16 + 576); } } @@ -1668,7 +1668,7 @@ void OpDispatchBuilder::AVX128_RestoreAVXState(Ref MemBase) { const auto NumRegs = Is64BitMode ? 16U : 8U; for (uint32_t i = 0; i < NumRegs; i += 2) { - auto YMMHRegs = LoadMemPair(FPRClass, OpSize::i128Bit, MemBase, i * 16 + 576); + auto YMMHRegs = LoadMemPairFPR(OpSize::i128Bit, MemBase, i * 16 + 576); AVX128_StoreXMMRegister(i, YMMHRegs.Low, true); AVX128_StoreXMMRegister(i + 1, YMMHRegs.High, true); @@ -1968,7 +1968,7 @@ void OpDispatchBuilder::AVX128_VFMAScalarImpl(OpcodeArgs, IROps IROp, uint8_t Sr if (Op->Src[1].IsGPR()) { Src2 = AVX128_LoadSource_WithOpSize(Op, Op->Src[1], Op->Flags, false).Low; } else { - Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags); + Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags); } Ref Sources[3] = {Dest, Src1, Src2}; @@ -2240,7 +2240,7 @@ void OpDispatchBuilder::AVX128_VCVTPH2PS(OpcodeArgs) { // In the event that a memory operand is used as the source operand, // the access width will always be half the size of the destination vector width // (i.e. 128-bit vector -> 64-bit mem, 256-bit vector -> 128-bit mem) - Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags); + Src.Low = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); } RefPair Result {}; @@ -2295,7 +2295,7 @@ void OpDispatchBuilder::AVX128_VCVTPS2PH(OpcodeArgs) { } if (!Op->Dest.IsGPR()) { - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result.Low, StoreSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result.Low, StoreSize, OpSize::iInvalid); } else { AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result); } diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp index 4980757eb..623b82be9 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp @@ -23,8 +23,8 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); // ARMv8 SHA1 extension provides a `SHA1H` instruction which does a fixed rotate by 30. // This only operates on element 0 rather than element 3. We don't have the luxury of rewriting the x86 SHA algorithm to take advantage of this. @@ -36,7 +36,7 @@ void OpDispatchBuilder::SHA1NEXTEOp(OpcodeArgs) { auto Tmp = _VAdd(OpSize::i128Bit, OpSize::i32Bit, Src, RotatedNode); auto Result = _VInsElement(OpSize::i128Bit, OpSize::i32Bit, 3, 3, Src, Tmp); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) { @@ -44,15 +44,15 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1); // [W0, W1, W2, W3] ^ [W2, W3, W4, W5] Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) { @@ -60,8 +60,8 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); // ARM SHA1 mostly matches x86 semantics, except the input and outputs are both flipped from elements 0,1,2,3 to 3,2,1,0. auto Src1 = SHADataShuffle(Dest); @@ -70,7 +70,7 @@ void OpDispatchBuilder::SHA1MSG2Op(OpcodeArgs) { // The result is swizzled differently than expected auto Result = SHADataShuffle(_VSha1SU1(Src1, Src2)); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) { @@ -79,8 +79,8 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) { return; } const uint64_t Imm8 = Op->Src[1].Literal() & 0b11; - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result {}; Ref ConstantVector {}; @@ -112,7 +112,7 @@ void OpDispatchBuilder::SHA1RNDS4Op(OpcodeArgs) { case 3: Result = SHADataShuffle(_VSha1P(Src1, ZeroRegister, Src2)); break; } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) { @@ -120,12 +120,12 @@ void OpDispatchBuilder::SHA256MSG1Op(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); auto Result = _VSha256U0(Dest, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) { @@ -133,8 +133,8 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); auto Src1 = _VExtr(OpSize::i128Bit, OpSize::i32Bit, Dest, Dest, 3); auto DupDst = _VDupElement(OpSize::i128Bit, OpSize::i32Bit, Dest, 3); @@ -142,7 +142,7 @@ void OpDispatchBuilder::SHA256MSG2Op(OpcodeArgs) { auto Result = _VSha256U1(Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) { @@ -150,8 +150,8 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); // Hardcoded to XMM0 auto XMM0 = LoadXMMRegister(0); @@ -177,7 +177,7 @@ void OpDispatchBuilder::SHA256RNDS2Op(OpcodeArgs) { auto B = _VSha256H2(EFGH, ABCD, Key); auto Result = shuffle_abcd(A, B); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::AESImcOp(OpcodeArgs) { @@ -185,9 +185,9 @@ void OpDispatchBuilder::AESImcOp(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = _VAESImc(Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::AESEncOp(OpcodeArgs) { @@ -195,10 +195,10 @@ void OpDispatchBuilder::AESEncOp(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = _VAESEnc(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit)); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VAESEncOp(OpcodeArgs) { @@ -208,11 +208,11 @@ void OpDispatchBuilder::VAESEncOp(OpcodeArgs) { // TODO: Handle 256-bit VAESENC. LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENC unimplemented"); - Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = _VAESEnc(DstSize, State, Key, LoadZeroVector(DstSize)); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) { @@ -220,10 +220,10 @@ void OpDispatchBuilder::AESEncLastOp(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = _VAESEncLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit)); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) { @@ -233,11 +233,11 @@ void OpDispatchBuilder::VAESEncLastOp(OpcodeArgs) { // TODO: Handle 256-bit VAESENCLAST. LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESENCLAST unimplemented"); - Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = _VAESEncLast(DstSize, State, Key, LoadZeroVector(DstSize)); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::AESDecOp(OpcodeArgs) { @@ -245,10 +245,10 @@ void OpDispatchBuilder::AESDecOp(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = _VAESDec(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit)); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VAESDecOp(OpcodeArgs) { @@ -258,11 +258,11 @@ void OpDispatchBuilder::VAESDecOp(OpcodeArgs) { // TODO: Handle 256-bit VAESDEC. LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDEC unimplemented"); - Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = _VAESDec(DstSize, State, Key, LoadZeroVector(DstSize)); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) { @@ -270,10 +270,10 @@ void OpDispatchBuilder::AESDecLastOp(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = _VAESDecLast(OpSize::i128Bit, Dest, Src, LoadZeroVector(OpSize::i128Bit)); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) { @@ -283,15 +283,15 @@ void OpDispatchBuilder::VAESDecLastOp(OpcodeArgs) { // TODO: Handle 256-bit VAESDECLAST. LOGMAN_THROW_A_FMT(Is128Bit, "256-bit VAESDECLAST unimplemented"); - Ref State = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Key = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref State = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Key = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = _VAESDecLast(DstSize, State, Key, LoadZeroVector(DstSize)); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::AESKeyGenAssistImpl(OpcodeArgs) { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); const uint64_t RCON = Op->Src[1].Literal(); auto KeyGenSwizzle = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, NAMED_VECTOR_AESKEYGENASSIST_SWIZZLE); @@ -305,7 +305,7 @@ void OpDispatchBuilder::AESKeyGenAssist(OpcodeArgs) { } Ref Result = AESKeyGenAssistImpl(Op); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) { @@ -313,12 +313,12 @@ void OpDispatchBuilder::PCLMULQDQOp(OpcodeArgs) { UnimplementedOp(Op); return; } - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); const auto Selector = static_cast(Op->Src[1].Literal()); auto Res = _PCLMUL(OpSize::i128Bit, Dest, Src, Selector & 0b1'0001); - StoreResult(FPRClass, Op, Res, OpSize::iInvalid); + StoreResultFPR(Op, Res, OpSize::iInvalid); } void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) { @@ -328,12 +328,12 @@ void OpDispatchBuilder::VPCLMULQDQOp(OpcodeArgs) { } const auto DstSize = OpSizeFromDst(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); const auto Selector = static_cast(Op->Src[2].Literal()); Ref Res = _PCLMUL(DstSize, Src1, Src2, Selector & 0b1'0001); - StoreResult(FPRClass, Op, Res, OpSize::iInvalid); + StoreResultFPR(Op, Res, OpSize::iInvalid); } } // namespace FEXCore::IR diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp index c107fd8e2..2a45f0235 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp @@ -29,8 +29,8 @@ void OpDispatchBuilder::MOVVectorAlignedOp(OpcodeArgs) { // Nop return; } - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - StoreResult(FPRClass, Op, Src, OpSize::iInvalid); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + StoreResultFPR(Op, Src, OpSize::iInvalid); } void OpDispatchBuilder::MOVVectorUnalignedOp(OpcodeArgs) { @@ -38,8 +38,8 @@ void OpDispatchBuilder::MOVVectorUnalignedOp(OpcodeArgs) { // Nop return; } - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit}); - StoreResult(FPRClass, Op, Src, OpSize::i8Bit); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit}); + StoreResultFPR(Op, Src, OpSize::i8Bit); } void OpDispatchBuilder::MOVVectorNTOp(OpcodeArgs) { @@ -47,23 +47,23 @@ void OpDispatchBuilder::MOVVectorNTOp(OpcodeArgs) { if (Op->Dest.IsGPR() && Size >= OpSize::i128Bit) { ///< MOVNTDQA load non-temporal comes from SSE4.1 and is extended by AVX/AVX2. - Ref SrcAddr = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false}); + Ref SrcAddr = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false}); auto Src = _VLoadNonTemporal(Size, SrcAddr, 0); - StoreResult(FPRClass, Op, Src, OpSize::i8Bit, MemoryAccessType::STREAM); + StoreResultFPR(Op, Src, OpSize::i8Bit, MemoryAccessType::STREAM); } else if (Op->Dest.IsGPR()) { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit, .AccessType = MemoryAccessType::STREAM}); - StoreResult(FPRClass, Op, Src, OpSize::i8Bit, MemoryAccessType::STREAM); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit, .AccessType = MemoryAccessType::STREAM}); + StoreResultFPR(Op, Src, OpSize::i8Bit, MemoryAccessType::STREAM); } else { LOGMAN_THROW_A_FMT(!Op->Dest.IsGPR(), "Destination can't be GPR for non-temporal stores"); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit, .AccessType = MemoryAccessType::STREAM}); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit, .AccessType = MemoryAccessType::STREAM}); if (Size < OpSize::i128Bit) { // Normal streaming store if less than 128-bit // XMM Scalar 32-bit and 64-bit comes from SSE4a MOVNTSS, MOVNTSD // MMX 64-bit comes from MOVNTQ - StoreResult(FPRClass, Op, Src, OpSize::i8Bit, MemoryAccessType::STREAM); + StoreResultFPR(Op, Src, OpSize::i8Bit, MemoryAccessType::STREAM); } else { - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false}); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false}); // Single store non-temporal for larger operations. _VStoreNonTemporal(Size, Src, Dest, 0); @@ -75,46 +75,46 @@ void OpDispatchBuilder::VMOVAPS_VMOVAPDOp(OpcodeArgs) { const auto SrcSize = GetSrcSize(Op); const auto Is128Bit = SrcSize == Core::CPUState::XMM_SSE_REG_SIZE; - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); if (Is128Bit && Op->Dest.IsGPR()) { Src = VZeroExtendOperand(OpSize::i128Bit, Op->Src[0], Src); } - StoreResult(FPRClass, Op, Src, OpSize::iInvalid); + StoreResultFPR(Op, Src, OpSize::iInvalid); } void OpDispatchBuilder::VMOVUPS_VMOVUPDOp(OpcodeArgs) { const auto SrcSize = GetSrcSize(Op); const auto Is128Bit = SrcSize == Core::CPUState::XMM_SSE_REG_SIZE; - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit}); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit}); if (Is128Bit && Op->Dest.IsGPR()) { Src = VZeroExtendOperand(OpSize::i128Bit, Op->Src[0], Src); } - StoreResult(FPRClass, Op, Src, OpSize::i8Bit); + StoreResultFPR(Op, Src, OpSize::i8Bit); } void OpDispatchBuilder::MOVHPDOp(OpcodeArgs) { if (Op->Dest.IsGPR()) { if (Op->Src[0].IsGPR()) { // MOVLHPS between two vector registers. - Ref Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); - Ref Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, OpSize::i128Bit, Op->Flags); + Ref Src = LoadSourceGPR(Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR_WithOpSize(Op, Op->Dest, OpSize::i128Bit, Op->Flags); auto Result = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 1, 0, Dest, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } else { // If the destination is a GPR then the source is memory // xmm1[127:64] = src Ref Src = MakeSegmentAddress(Op, Op->Src[0]); - Ref Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, OpSize::i128Bit, Op->Flags); + Ref Dest = LoadSourceFPR_WithOpSize(Op, Op->Dest, OpSize::i128Bit, Op->Flags); auto Result = _VLoadVectorElement(OpSize::i128Bit, OpSize::i64Bit, Dest, 1, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } } else { // In this case memory is the destination and the high bits of the XMM are source // Mem64 = xmm1[127:64] - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Dest = MakeSegmentAddress(Op, Op->Dest); _VStoreVectorElement(OpSize::i128Bit, OpSize::i64Bit, Src, 1, Dest); } @@ -122,15 +122,15 @@ void OpDispatchBuilder::MOVHPDOp(OpcodeArgs) { void OpDispatchBuilder::VMOVHPOp(OpcodeArgs) { if (Op->Dest.IsGPR()) { - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i128Bit}); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, {.Align = OpSize::i64Bit}); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i128Bit}); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags, {.Align = OpSize::i64Bit}); Ref Result = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 1, 0, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } else { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i128Bit}); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i128Bit}); Ref Result = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 0, 1, Src, Src); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, OpSize::i64Bit, OpSize::i64Bit); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, OpSize::i64Bit, OpSize::i64Bit); } } @@ -138,74 +138,74 @@ void OpDispatchBuilder::MOVLPOp(OpcodeArgs) { if (Op->Dest.IsGPR()) { // xmm, xmm is movhlps special case if (Op->Src[0].IsGPR()) { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i128Bit}); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, {.Align = OpSize::i128Bit}); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i128Bit}); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags, {.Align = OpSize::i128Bit}); auto Result = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 0, 1, Dest, Src); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, OpSize::i128Bit, OpSize::i128Bit); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, OpSize::i128Bit, OpSize::i128Bit); } else { const auto DstSize = OpSizeFromDst(Op); Ref Src = MakeSegmentAddress(Op, Op->Src[0]); - Ref Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags); + Ref Dest = LoadSourceFPR_WithOpSize(Op, Op->Dest, DstSize, Op->Flags); auto Result = _VLoadVectorElement(OpSize::i128Bit, OpSize::i64Bit, Dest, 0, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } } else { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i64Bit}); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, OpSize::i64Bit, OpSize::i64Bit); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i64Bit}); + StoreResultFPR_WithOpSize(Op, Op->Dest, Src, OpSize::i64Bit, OpSize::i64Bit); } } void OpDispatchBuilder::VMOVLPOp(OpcodeArgs) { - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i128Bit}); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i128Bit}); if (!Op->Dest.IsGPR()) { ///< VMOVLPS/PD mem64, xmm1 - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src1, OpSize::i64Bit, OpSize::i64Bit); + StoreResultFPR_WithOpSize(Op, Op->Dest, Src1, OpSize::i64Bit, OpSize::i64Bit); } else if (!Op->Src[1].IsGPR()) { ///< VMOVLPS/PD xmm1, xmm2, mem64 // Bits[63:0] come from Src2[63:0] // Bits[127:64] come from Src1[127:64] - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, {.Align = OpSize::i64Bit}); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags, {.Align = OpSize::i64Bit}); Ref Result = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 1, 1, Src2, Src1); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } else { ///< VMOVHLPS/PD xmm1, xmm2, xmm3 - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags, {.Align = OpSize::i128Bit}); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags, {.Align = OpSize::i128Bit}); Ref Result = _VInsElement(OpSize::i128Bit, OpSize::i64Bit, 0, 1, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } } void OpDispatchBuilder::VMOVSHDUPOp(OpcodeArgs) { const auto SrcSize = OpSizeFromSrc(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = _VTrn2(SrcSize, OpSize::i32Bit, Src, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VMOVSLDUPOp(OpcodeArgs) { const auto SrcSize = OpSizeFromSrc(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = _VTrn(SrcSize, OpSize::i32Bit, Src, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::MOVScalarOpImpl(OpcodeArgs, IR::OpSize ElementSize) { if (Op->Dest.IsGPR() && Op->Src[0].IsGPR()) { // MOVSS/SD xmm1, xmm2 - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); auto Result = _VInsElement(OpSize::i128Bit, ElementSize, 0, 0, Dest, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } else if (Op->Dest.IsGPR()) { // MOVSS/SD xmm1, mem32/mem64 // xmm1[127:0] <- zext(mem32/mem64) - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], ElementSize, Op->Flags); - StoreResult(FPRClass, Op, Src, OpSize::iInvalid); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], ElementSize, Op->Flags); + StoreResultFPR(Op, Src, OpSize::iInvalid); } else { // MOVSS/SD mem32/mem64, xmm1 - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, ElementSize, OpSize::iInvalid); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + StoreResultFPR_WithOpSize(Op, Op->Dest, Src, ElementSize, OpSize::iInvalid); } } @@ -220,18 +220,18 @@ void OpDispatchBuilder::MOVSDOp(OpcodeArgs) { void OpDispatchBuilder::VMOVScalarOpImpl(OpcodeArgs, IR::OpSize ElementSize) { if (Op->Dest.IsGPR() && Op->Src[0].IsGPR() && Op->Src[1].IsGPR()) { // VMOVSS/SD xmm1, xmm2, xmm3 - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = _VInsElement(OpSize::i128Bit, ElementSize, 0, 0, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } else if (Op->Dest.IsGPR()) { // VMOVSS/SD xmm1, mem32/mem64 - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], ElementSize, Op->Flags); - StoreResult(FPRClass, Op, Src, OpSize::iInvalid); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[1], ElementSize, Op->Flags); + StoreResultFPR(Op, Src, OpSize::iInvalid); } else { // VMOVSS/SD mem32/mem64, xmm1 - Ref Src = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, ElementSize, OpSize::iInvalid); + Ref Src = LoadSourceFPR(Op, Op->Src[1], Op->Flags); + StoreResultFPR_WithOpSize(Op, Op->Dest, Src, ElementSize, OpSize::iInvalid); } } @@ -245,12 +245,12 @@ void OpDispatchBuilder::VMOVSSOp(OpcodeArgs) { void OpDispatchBuilder::VectorALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) { const auto Size = OpSizeFromSrc(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); DeriveOp(ALUOp, IROp, _VAdd(Size, ElementSize, Dest, Src)); - StoreResult(FPRClass, Op, ALUOp, OpSize::iInvalid); + StoreResultFPR(Op, ALUOp, OpSize::iInvalid); } void OpDispatchBuilder::VectorXOROp(OpcodeArgs) { @@ -259,7 +259,7 @@ void OpDispatchBuilder::VectorXOROp(OpcodeArgs) { // Special case for vector xor with itself being the optimal way for x86 to zero vector registers. if (Op->Dest.IsGPR() && Op->Src[0].IsGPR() && Op->Dest.Data.GPR.GPR == Op->Src[0].Data.GPR.GPR) { const auto ZeroRegister = LoadZeroVector(Size); - StoreResult(FPRClass, Op, ZeroRegister, OpSize::iInvalid); + StoreResultFPR(Op, ZeroRegister, OpSize::iInvalid); return; } @@ -270,12 +270,12 @@ void OpDispatchBuilder::VectorXOROp(OpcodeArgs) { void OpDispatchBuilder::AVXVectorALUOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) { const auto Size = OpSizeFromSrc(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); DeriveOp(ALUOp, IROp, _VAdd(Size, ElementSize, Src1, Src2)); - StoreResult(FPRClass, Op, ALUOp, OpSize::iInvalid); + StoreResultFPR(Op, ALUOp, OpSize::iInvalid); } void OpDispatchBuilder::AVXVectorXOROp(OpcodeArgs) { @@ -283,7 +283,7 @@ void OpDispatchBuilder::AVXVectorXOROp(OpcodeArgs) { if (Op->Src[0].IsGPR() && Op->Src[1].IsGPR() && Op->Src[0].Data.GPR.GPR == Op->Src[1].Data.GPR.GPR) { const auto DstSize = OpSizeFromDst(Op); const auto ZeroRegister = LoadZeroVector(DstSize); - StoreResult(FPRClass, Op, ZeroRegister, OpSize::iInvalid); + StoreResultFPR(Op, ZeroRegister, OpSize::iInvalid); return; } @@ -293,12 +293,12 @@ void OpDispatchBuilder::AVXVectorXOROp(OpcodeArgs) { void OpDispatchBuilder::VectorALUROp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) { const auto Size = OpSizeFromSrc(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); DeriveOp(ALUOp, IROp, _VAdd(Size, ElementSize, Src, Dest)); - StoreResult(FPRClass, Op, ALUOp, OpSize::iInvalid); + StoreResultFPR(Op, ALUOp, OpSize::iInvalid); } Ref OpDispatchBuilder::VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IROp, IR::OpSize DstSize, IR::OpSize ElementSize, @@ -309,8 +309,8 @@ Ref OpDispatchBuilder::VectorScalarInsertALUOpImpl(OpcodeArgs, IROps IROp, IR::O // element that we're going to operate on. const auto SrcSize = OpSizeFromSrc(Op); - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Src1Op, DstSize, Op->Flags); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Src2Op, SrcSize, Op->Flags, {.AllowUpperGarbage = true}); // If OpSize == ElementSize then it only does the lower scalar op DeriveOp(ALUOp, IROp, _VFAddScalarInsert(DstSize, ElementSize, Src1, Src2, ZeroUpperBits)); @@ -321,7 +321,7 @@ template void OpDispatchBuilder::VectorScalarInsertALUOp(OpcodeArgs) { const auto DstSize = GetGuestVectorLength(); auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Dest, Op->Src[0], false); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::VectorScalarInsertALUOp(OpcodeArgs); @@ -341,7 +341,7 @@ template void OpDispatchBuilder::AVXVectorScalarInsertALUOp(OpcodeArgs) { const auto DstSize = GetGuestVectorLength(); auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Src[0], Op->Src[1], true); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::AVXVectorScalarInsertALUOp(OpcodeArgs); @@ -365,8 +365,8 @@ Ref OpDispatchBuilder::VectorScalarUnaryInsertALUOpImpl(OpcodeArgs, IROps IROp, // element that we're going to operate on. const auto SrcSize = OpSizeFromSrc(Op); - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Src1Op, DstSize, Op->Flags); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Src2Op, SrcSize, Op->Flags, {.AllowUpperGarbage = true}); // If OpSize == ElementSize then it only does the lower scalar op DeriveOp(ALUOp, IROp, _VFSqrtScalarInsert(DstSize, ElementSize, Src1, Src2, ZeroUpperBits)); @@ -377,7 +377,7 @@ template void OpDispatchBuilder::VectorScalarUnaryInsertALUOp(OpcodeArgs) { const auto DstSize = GetGuestVectorLength(); auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Dest, Op->Src[0], false); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::VectorScalarUnaryInsertALUOp(OpcodeArgs); @@ -393,7 +393,7 @@ template void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp(OpcodeArgs) { const auto DstSize = GetGuestVectorLength(); auto Result = VectorScalarInsertALUOpImpl(Op, IROp, DstSize, ElementSize, Op->Src[0], Op->Src[1], true); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::AVXVectorScalarUnaryInsertALUOp(OpcodeArgs); @@ -412,15 +412,15 @@ void OpDispatchBuilder::InsertMMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) { const auto DstSize = GetGuestVectorLength(); const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i64Bit : OpSizeFromSrc(Op); - Ref Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags); - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags); + Ref Dest = LoadSourceFPR_WithOpSize(Op, Op->Dest, DstSize, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); // Always 32-bit. const auto ElementSize = OpSize::i32Bit; // Always signed Dest = _VSToFVectorInsert(DstSize, ElementSize, ElementSize, Dest, Src, true, false); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Dest, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Dest, DstSize, OpSize::iInvalid); } Ref OpDispatchBuilder::InsertCVTGPR_To_FPRImpl(OpcodeArgs, IR::OpSize DstSize, IR::OpSize DstElementSize, const X86Tables::DecodedOperand& Src1Op, @@ -430,23 +430,23 @@ Ref OpDispatchBuilder::InsertCVTGPR_To_FPRImpl(OpcodeArgs, IR::OpSize DstSize, I // element that we're going to operate on. const auto SrcSize = OpSizeFromSrc(Op); - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Src1Op, DstSize, Op->Flags); if (Src2Op.IsGPR()) { // If the source is a GPR then convert directly from the GPR. - auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, GetGPROpSize(), Op->Flags); + auto Src2 = LoadSourceGPR_WithOpSize(Op, Src2Op, GetGPROpSize(), Op->Flags); return _VSToFGPRInsert(DstSize, DstElementSize, SrcSize, Src1, Src2, ZeroUpperBits); } else if (SrcSize != DstElementSize) { // If the source is from memory but the Source size and destination size aren't the same, // then it is more optimal to load in to a GPR and convert between GPR->FPR. // ARM GPR->FPR conversion supports different size source and destinations while FPR->FPR doesn't. - auto Src2 = LoadSource(GPRClass, Op, Src2Op, Op->Flags); + auto Src2 = LoadSourceGPR(Op, Src2Op, Op->Flags); return _VSToFGPRInsert(DstSize, DstElementSize, SrcSize, Src1, Src2, ZeroUpperBits); } // In the case of cvtsi2s{s,d} where the source and destination are the same size, // then it is more optimal to load in to the FPR register directly and convert there. - auto Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags); + auto Src2 = LoadSourceFPR(Op, Src2Op, Op->Flags); // Always signed return _VSToFVectorInsert(DstSize, DstElementSize, DstElementSize, Src1, Src2, false, ZeroUpperBits); } @@ -455,7 +455,7 @@ template void OpDispatchBuilder::InsertCVTGPR_To_FPR(OpcodeArgs) { const auto DstSize = GetGuestVectorLength(); auto Result = InsertCVTGPR_To_FPRImpl(Op, DstSize, DstElementSize, Op->Dest, Op->Src[0], false); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::InsertCVTGPR_To_FPR(OpcodeArgs); @@ -465,7 +465,7 @@ template void OpDispatchBuilder::AVXInsertCVTGPR_To_FPR(OpcodeArgs) { const auto DstSize = GetGuestVectorLength(); Ref Result = InsertCVTGPR_To_FPRImpl(Op, DstSize, DstElementSize, Op->Src[0], Op->Src[1], true); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::AVXInsertCVTGPR_To_FPR(OpcodeArgs); template void OpDispatchBuilder::AVXInsertCVTGPR_To_FPR(OpcodeArgs); @@ -479,8 +479,8 @@ Ref OpDispatchBuilder::InsertScalar_CVT_Float_To_FloatImpl(OpcodeArgs, IR::OpSiz // element that we're going to operate on. const auto SrcSize = Src2Op.IsGPR() ? OpSize::i128Bit : SrcElementSize; - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Src1Op, DstSize, Op->Flags); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Src2Op, SrcSize, Op->Flags, {.AllowUpperGarbage = true}); return _VFToFScalarInsert(DstSize, DstElementSize, SrcElementSize, Src1, Src2, ZeroUpperBits); } @@ -489,7 +489,7 @@ template void OpDispatchBuilder::InsertScalar_CVT_Float_To_Float(OpcodeArgs) { const auto DstSize = GetGuestVectorLength(); Ref Result = InsertScalar_CVT_Float_To_FloatImpl(Op, DstSize, DstElementSize, SrcElementSize, Op->Dest, Op->Src[0], false); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::InsertScalar_CVT_Float_To_Float(OpcodeArgs); @@ -499,7 +499,7 @@ template void OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs) { const auto DstSize = GetGuestVectorLength(); Ref Result = InsertScalar_CVT_Float_To_FloatImpl(Op, DstSize, DstElementSize, SrcElementSize, Op->Src[0], Op->Src[1], true); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::AVXInsertScalar_CVT_Float_To_Float(OpcodeArgs); @@ -526,8 +526,8 @@ Ref OpDispatchBuilder::InsertScalarRoundImpl(OpcodeArgs, IR::OpSize DstSize, IR: // element that we're going to operate on. const auto SrcSize = OpSizeFromSrc(Op); - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, DstSize, Op->Flags); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, SrcSize, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Src1Op, DstSize, Op->Flags); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Src2Op, SrcSize, Op->Flags, {.AllowUpperGarbage = true}); const auto SourceMode = TranslateRoundType(Mode); auto ALUOp = _VFToIScalarInsert(DstSize, ElementSize, Src1, Src2, SourceMode, ZeroUpperBits); @@ -541,7 +541,7 @@ void OpDispatchBuilder::InsertScalarRound(OpcodeArgs) { const auto DstSize = GetGuestVectorLength(); Ref Result = InsertScalarRoundImpl(Op, DstSize, ElementSize, Op->Dest, Op->Src[0], Mode, false); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::InsertScalarRound(OpcodeArgs); @@ -553,7 +553,7 @@ void OpDispatchBuilder::AVXInsertScalarRound(OpcodeArgs) { const auto DstSize = GetGuestVectorLength(); Ref Result = InsertScalarRoundImpl(Op, DstSize, ElementSize, Op->Dest, Op->Src[0], Mode, true); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::AVXInsertScalarRound(OpcodeArgs); @@ -597,11 +597,11 @@ void OpDispatchBuilder::InsertScalarFCMPOp(OpcodeArgs) { const auto DstSize = GetGuestVectorLength(); const auto SrcSize = OpSizeFromSrc(Op); - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Op->Dest, DstSize, Op->Flags); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags, {.AllowUpperGarbage = true}); Ref Result = InsertScalarFCMPOpImpl(DstSize, OpSizeFromDst(Op), ElementSize, Src1, Src2, CompType, false); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::InsertScalarFCMPOp(OpcodeArgs); @@ -616,11 +616,11 @@ void OpDispatchBuilder::AVXInsertScalarFCMPOp(OpcodeArgs) { // We load the full vector width when dealing with a source vector, // so that we don't do any unnecessary zero extension to the scalar // element that we're going to operate on. - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true}); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Op->Src[0], DstSize, Op->Flags); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags, {.AllowUpperGarbage = true}); Ref Result = InsertScalarFCMPOpImpl(DstSize, OpSizeFromDst(Op), ElementSize, Src1, Src2, CompType, true); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::AVXInsertScalarFCMPOp(OpcodeArgs); @@ -630,7 +630,7 @@ void OpDispatchBuilder::RSqrt3DNowOp(OpcodeArgs, bool Duplicate) { const auto Size = OpSizeFromSrc(Op); const auto ElementSize = OpSize::i32Bit; - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Size, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Size, Op->Flags); // For the sqrt reciprocal in 3DNow!, if the source is negative, // then the result has the same sign as the source but the result is always calculated @@ -643,7 +643,7 @@ void OpDispatchBuilder::RSqrt3DNowOp(OpcodeArgs, bool Duplicate) { Result = _VDupElement(Size, ElementSize, Result, 0); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) { @@ -653,10 +653,10 @@ void OpDispatchBuilder::VectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize Element // In the event of a memory operand, we load the exact element size. const auto Size = OpSizeFromSrc(Op); - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Size, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Size, Op->Flags); DeriveOp(ALUOp, IROp, _VFSqrt(Size, ElementSize, Src)); - StoreResult(FPRClass, Op, ALUOp, OpSize::iInvalid); + StoreResultFPR(Op, ALUOp, OpSize::iInvalid); } void OpDispatchBuilder::AVXVectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) { @@ -666,7 +666,7 @@ void OpDispatchBuilder::AVXVectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize Elem // In the event of a memory operand, we load the exact element size. const auto SrcSize = OpSizeFromSrc(Op); - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); DeriveOp(ALUOp, IROp, _VFSqrt(SrcSize, ElementSize, Src)); @@ -675,19 +675,19 @@ void OpDispatchBuilder::AVXVectorUnaryOp(OpcodeArgs, IROps IROp, IR::OpSize Elem // which, on hardware with SVE, zero-extends as part of // storing into the destination. - StoreResult(FPRClass, Op, ALUOp, OpSize::iInvalid); + StoreResultFPR(Op, ALUOp, OpSize::iInvalid); } void OpDispatchBuilder::VectorUnaryDuplicateOpImpl(OpcodeArgs, IROps IROp, IR::OpSize ElementSize) { const auto Size = OpSizeFromSrc(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); DeriveOp(ALUOp, IROp, _VFSqrt(ElementSize, ElementSize, Src)); // Duplicate the lower bits auto Result = _VDupElement(Size, ElementSize, ALUOp, 0); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template @@ -700,7 +700,7 @@ template void OpDispatchBuilder::VectorUnaryDuplicateOpSrc[0].IsGPR() ? OpSize::i128Bit : OpSizeFromSrc(Op); - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); // This instruction is a bit special that if the destination is a register then it'll ZEXT the 64bit source to 128bit if (Op->Dest.IsGPR()) { const auto gpr = Op->Dest.Data.GPR.GPR; @@ -710,7 +710,7 @@ void OpDispatchBuilder::MOVQOp(OpcodeArgs, VectorOpType VectorType) { StoreXMMRegister_WithAVXInsert(VectorType, gprIndex, Reg); } else { // This is simple, just store the result - StoreResult(FPRClass, Op, Src, OpSize::iInvalid); + StoreResultFPR(Op, Src, OpSize::iInvalid); } } @@ -719,15 +719,15 @@ void OpDispatchBuilder::MOVQMMXOp(OpcodeArgs) { if (MMXState == MMXState_X87) { ChgStateX87_MMX(); } - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit}); - StoreResult(FPRClass, Op, Src, OpSize::i8Bit); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags, {.Align = OpSize::i8Bit}); + StoreResultFPR(Op, Src, OpSize::i8Bit); } void OpDispatchBuilder::MOVMSKOp(OpcodeArgs, IR::OpSize ElementSize) { const auto Size = OpSizeFromSrc(Op); const auto NumElements = IR::NumElements(Size, ElementSize); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); if (Size == OpSize::i128Bit && ElementSize == OpSize::i64Bit) { // UnZip2 the 64-bit elements as 32-bit to get the sign bits closer. @@ -741,7 +741,7 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs, IR::OpSize ElementSize) { GPR = _Bfi(OpSize::i64Bit, 32, 31, GPR, GPR); // Shift right to only get the two sign bits we care about. GPR = _Lshr(OpSize::i64Bit, GPR, Constant(62)); - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, GPR, GetGPROpSize(), OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize(), OpSize::iInvalid); } else if (Size == OpSize::i128Bit && ElementSize == OpSize::i32Bit) { // Shift all the sign bits to the bottom of their respective elements. Src = _VUShrI(Size, OpSize::i32Bit, Src, 31); @@ -753,7 +753,7 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs, IR::OpSize ElementSize) { Src = _VAddV(Size, OpSize::i32Bit, Src); // Extract to a GPR. Ref GPR = _VExtractToGPR(Size, OpSize::i32Bit, Src, 0); - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, GPR, GetGPROpSize(), OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, GPR, GetGPROpSize(), OpSize::iInvalid); } else { Ref CurrentVal = Constant(0); @@ -769,7 +769,7 @@ void OpDispatchBuilder::MOVMSKOp(OpcodeArgs, IR::OpSize ElementSize) { CurrentVal = Tmp; } } - StoreResult(GPRClass, Op, CurrentVal, OpSize::iInvalid); + StoreResultGPR(Op, CurrentVal, OpSize::iInvalid); } } @@ -778,7 +778,7 @@ void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) { const auto Is256Bit = SrcSize == OpSize::i256Bit; const auto ExtractSize = Is256Bit ? OpSize::i32Bit : OpSize::i16Bit; - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref VMask = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_MOVMASKB); auto VCMP = _VCMPLTZ(SrcSize, OpSize::i8Bit, Src); @@ -795,25 +795,25 @@ void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) { auto Result = _VExtractToGPR(SrcSize, ExtractSize, VAdd3, 0); - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PUNPCKLOp(OpcodeArgs, IR::OpSize ElementSize) { const auto Size = OpSizeFromSrc(Op); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); auto ALUOp = _VZip(Size, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, ALUOp, OpSize::iInvalid); + StoreResultFPR(Op, ALUOp, OpSize::iInvalid); } void OpDispatchBuilder::VPUNPCKLOp(OpcodeArgs, IR::OpSize ElementSize) { const auto SrcSize = OpSizeFromSrc(Op); const auto Is128Bit = SrcSize == OpSize::i128Bit; - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result {}; if (Is128Bit) { @@ -825,23 +825,23 @@ void OpDispatchBuilder::VPUNPCKLOp(OpcodeArgs, IR::OpSize ElementSize) { Result = _VInsElement(SrcSize, OpSize::i128Bit, 1, 0, ZipLo, ZipHi); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PUNPCKHOp(OpcodeArgs, IR::OpSize ElementSize) { const auto Size = OpSizeFromSrc(Op); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); auto ALUOp = _VZip2(Size, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, ALUOp, OpSize::iInvalid); + StoreResultFPR(Op, ALUOp, OpSize::iInvalid); } void OpDispatchBuilder::VPUNPCKHOp(OpcodeArgs, IR::OpSize ElementSize) { const auto SrcSize = OpSizeFromSrc(Op); const auto Is128Bit = SrcSize == OpSize::i128Bit; - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result {}; if (Is128Bit) { @@ -853,7 +853,7 @@ void OpDispatchBuilder::VPUNPCKHOp(OpcodeArgs, IR::OpSize ElementSize) { Result = _VInsElement(SrcSize, OpSize::i128Bit, 0, 1, ZipHi, ZipLo); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::GeneratePSHUFBMask(IR::OpSize SrcSize) { @@ -891,20 +891,20 @@ Ref OpDispatchBuilder::PSHUFBOpImpl(IR::OpSize SrcSize, Ref Src1, Ref Src2, Ref void OpDispatchBuilder::PSHUFBOp(OpcodeArgs) { const auto SrcSize = OpSizeFromSrc(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PSHUFBOpImpl(SrcSize, Src1, Src2, GeneratePSHUFBMask(SrcSize)); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPSHUFBOp(OpcodeArgs) { const auto SrcSize = OpSizeFromSrc(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = PSHUFBOpImpl(SrcSize, Src1, Src2, GeneratePSHUFBMask(SrcSize)); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::PShufWLane(IR::OpSize Size, FEXCore::IR::IndexNamedVectorConstant IndexConstant, bool LowLane, Ref IncomingLane, @@ -954,21 +954,21 @@ Ref OpDispatchBuilder::PShufWLane(IR::OpSize Size, FEXCore::IR::IndexNamedVector void OpDispatchBuilder::PSHUFW8ByteOp(OpcodeArgs) { uint16_t Shuffle = Op->Src[1].Data.Literal.Value; const auto Size = OpSizeFromSrc(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Dest = PShufWLane(Size, FEXCore::IR::INDEXED_NAMED_VECTOR_PSHUFLW, true, Src, Shuffle); - StoreResult(FPRClass, Op, Dest, OpSize::iInvalid); + StoreResultFPR(Op, Dest, OpSize::iInvalid); } void OpDispatchBuilder::PSHUFWOp(OpcodeArgs, bool Low) { uint16_t Shuffle = Op->Src[1].Data.Literal.Value; const auto Size = OpSizeFromSrc(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); const auto IndexedVectorConstant = Low ? FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFLW : FEXCore::IR::IndexNamedVectorConstant::INDEXED_NAMED_VECTOR_PSHUFHW; Ref Dest = PShufWLane(Size, IndexedVectorConstant, Low, Src, Shuffle); - StoreResult(FPRClass, Op, Dest, OpSize::iInvalid); + StoreResultFPR(Op, Dest, OpSize::iInvalid); } Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle) { @@ -1199,8 +1199,8 @@ Ref OpDispatchBuilder::Single128Bit4ByteVectorShuffle(Ref Src, uint8_t Shuffle) void OpDispatchBuilder::PSHUFDOp(OpcodeArgs) { uint16_t Shuffle = Op->Src[1].Data.Literal.Value; - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - StoreResult(FPRClass, Op, Single128Bit4ByteVectorShuffle(Src, Shuffle), OpSize::iInvalid); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + StoreResultFPR(Op, Single128Bit4ByteVectorShuffle(Src, Shuffle), OpSize::iInvalid); } void OpDispatchBuilder::VPSHUFWOp(OpcodeArgs, IR::OpSize ElementSize, bool Low) { @@ -1208,7 +1208,7 @@ void OpDispatchBuilder::VPSHUFWOp(OpcodeArgs, IR::OpSize ElementSize, bool Low) const auto Is256Bit = SrcSize == OpSize::i256Bit; auto Shuffle = Op->Src[1].Literal(); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); // Note/TODO: With better immediate facilities or vector loading in our IR // much of this can be reduced to setting up a table index register @@ -1249,7 +1249,7 @@ void OpDispatchBuilder::VPSHUFWOp(OpcodeArgs, IR::OpSize ElementSize, bool Low) } } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::SHUFOpImpl(OpcodeArgs, IR::OpSize DstSize, IR::OpSize ElementSize, Ref Src1, Ref Src2, uint8_t Shuffle) { @@ -1444,31 +1444,31 @@ Ref OpDispatchBuilder::SHUFOpImpl(OpcodeArgs, IR::OpSize DstSize, IR::OpSize Ele } void OpDispatchBuilder::SHUFOp(OpcodeArgs, IR::OpSize ElementSize) { - Ref Src1Node = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src2Node = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src1Node = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src2Node = LoadSourceFPR(Op, Op->Src[0], Op->Flags); uint8_t Shuffle = Op->Src[1].Literal(); Ref Result = SHUFOpImpl(Op, OpSizeFromDst(Op), ElementSize, Src1Node, Src2Node, Shuffle); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VSHUFOp(OpcodeArgs, IR::OpSize ElementSize) { - Ref Src1Node = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2Node = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1Node = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2Node = LoadSourceFPR(Op, Op->Src[1], Op->Flags); uint8_t Shuffle = Op->Src[2].Literal(); Ref Result = SHUFOpImpl(Op, OpSizeFromDst(Op), ElementSize, Src1Node, Src2Node, Shuffle); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VANDNOp(OpcodeArgs) { const auto SrcSize = OpSizeFromSrc(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Dest = _VAndn(SrcSize, SrcSize, Src2, Src1); - StoreResult(FPRClass, Op, Dest, OpSize::iInvalid); + StoreResultFPR(Op, Dest, OpSize::iInvalid); } template @@ -1476,8 +1476,8 @@ void OpDispatchBuilder::VHADDPOp(OpcodeArgs) { const auto SrcSize = OpSizeFromSrc(Op); const auto Is256Bit = SrcSize == OpSize::i256Bit; - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); DeriveOp(Res, IROp, _VFAddP(SrcSize, ElementSize, Src1, Src2)); @@ -1487,7 +1487,7 @@ void OpDispatchBuilder::VHADDPOp(OpcodeArgs) { Dest = _VInsElement(SrcSize, OpSize::i64Bit, 2, 1, Dest, Res); } - StoreResult(FPRClass, Op, Dest, OpSize::iInvalid); + StoreResultFPR(Op, Dest, OpSize::iInvalid); } template void OpDispatchBuilder::VHADDPOp(OpcodeArgs); @@ -1500,7 +1500,7 @@ void OpDispatchBuilder::VBROADCASTOp(OpcodeArgs, IR::OpSize ElementSize) { Ref Result {}; if (Op->Src[0].IsGPR()) { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Result = _VDupElement(DstSize, ElementSize, Src, 0); } else { // Get the address to broadcast from into a GPR. @@ -1511,7 +1511,7 @@ void OpDispatchBuilder::VBROADCASTOp(OpcodeArgs, IR::OpSize ElementSize) { // No need to zero-extend result, since implementations // use zero extending AdvSIMD or zeroing SVE loads internally. - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::PINSROpImpl(OpcodeArgs, IR::OpSize ElementSize, const X86Tables::DecodedOperand& Src1Op, @@ -1519,11 +1519,11 @@ Ref OpDispatchBuilder::PINSROpImpl(OpcodeArgs, IR::OpSize ElementSize, const X86 const auto Size = OpSizeFromDst(Op); const auto NumElements = IR::NumElements(Size, ElementSize); const uint64_t Index = Imm.Literal() & (NumElements - 1); - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, Size, Op->Flags); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Src1Op, Size, Op->Flags); if (Src2Op.IsGPR()) { // If the source is a GPR then convert directly from the GPR. - auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, GetGPROpSize(), Op->Flags); + auto Src2 = LoadSourceGPR_WithOpSize(Op, Src2Op, GetGPROpSize(), Op->Flags); return _VInsGPR(Size, ElementSize, Index, Src1, Src2); } @@ -1535,7 +1535,7 @@ Ref OpDispatchBuilder::PINSROpImpl(OpcodeArgs, IR::OpSize ElementSize, const X86 template void OpDispatchBuilder::PINSROp(OpcodeArgs) { Ref Result = PINSROpImpl(Op, ElementSize, Op->Dest, Op->Src[0], Op->Src[1]); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::PINSROp(OpcodeArgs); @@ -1548,7 +1548,7 @@ void OpDispatchBuilder::VPINSRBOp(OpcodeArgs) { if (Op->Dest.Data.GPR.GPR == Op->Src[0].Data.GPR.GPR) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPINSRDQOp(OpcodeArgs) { @@ -1557,7 +1557,7 @@ void OpDispatchBuilder::VPINSRDQOp(OpcodeArgs) { if (Op->Dest.Data.GPR.GPR == Op->Src[0].Data.GPR.GPR) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPINSRWOp(OpcodeArgs) { @@ -1565,7 +1565,7 @@ void OpDispatchBuilder::VPINSRWOp(OpcodeArgs) { if (Op->Dest.Data.GPR.GPR == Op->Src[0].Data.GPR.GPR) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::InsertPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1, const X86Tables::DecodedOperand& Src2, @@ -1580,18 +1580,18 @@ Ref OpDispatchBuilder::InsertPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperan Ref Dest {}; if (ZMask != 0xF) { // Only need to load destination if it isn't a full zero - Dest = LoadSource_WithOpSize(FPRClass, Op, Src1, DstSize, Op->Flags); + Dest = LoadSourceFPR_WithOpSize(Op, Src1, DstSize, Op->Flags); } if ((ZMask & (1 << CountD)) == 0) { // In the case that ZMask overwrites the destination element, then don't even insert Ref Src {}; if (Src2.IsGPR()) { - Src = LoadSource(FPRClass, Op, Src2, Op->Flags); + Src = LoadSourceFPR(Op, Src2, Op->Flags); } else { // If loading from memory then CountS is forced to zero CountS = 0; - Src = LoadSource_WithOpSize(FPRClass, Op, Src2, OpSize::i32Bit, Op->Flags); + Src = LoadSourceFPR_WithOpSize(Op, Src2, OpSize::i32Bit, Op->Flags); } Dest = _VInsElement(DstSize, OpSize::i32Bit, CountD, CountS, Dest, Src); @@ -1616,18 +1616,18 @@ Ref OpDispatchBuilder::InsertPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperan void OpDispatchBuilder::InsertPSOp(OpcodeArgs) { Ref Result = InsertPSOpImpl(Op, Op->Dest, Op->Src[0], Op->Src[1]); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VINSERTPSOp(OpcodeArgs) { Ref Result = InsertPSOpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2]); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PExtrOp(OpcodeArgs, IR::OpSize ElementSize) { const auto DstSize = OpSizeFromDst(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); uint64_t Index = Op->Src[1].Literal(); // Fixup of 32-bit element size. @@ -1647,7 +1647,7 @@ void OpDispatchBuilder::PExtrOp(OpcodeArgs, IR::OpSize ElementSize) { const auto GPRSize = GetGPROpSize(); // Extract already zero extends the result. Ref Result = _VExtractToGPR(OpSize::i128Bit, OverridenElementSize, Src, Index); - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, Result, GPRSize, OpSize::iInvalid); return; } @@ -1661,12 +1661,12 @@ void OpDispatchBuilder::VEXTRACT128Op(OpcodeArgs) { const auto StoreSize = DstIsXMM ? OpSize::i256Bit : OpSize::i128Bit; const auto Selector = Op->Src[1].Literal() & 0b1; - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); // A selector of zero is the same as doing a 128-bit vector move. if (Selector == 0) { Ref Result = DstIsXMM ? _VMov(OpSize::i128Bit, Src) : Src; - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, StoreSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, StoreSize, OpSize::iInvalid); return; } @@ -1675,7 +1675,7 @@ void OpDispatchBuilder::VEXTRACT128Op(OpcodeArgs) { if (DstIsXMM) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, StoreSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, StoreSize, OpSize::iInvalid); } Ref OpDispatchBuilder::PSIGNImpl(OpcodeArgs, IR::OpSize ElementSize, Ref Src1, Ref Src2) { @@ -1688,11 +1688,11 @@ Ref OpDispatchBuilder::PSIGNImpl(OpcodeArgs, IR::OpSize ElementSize, Ref Src1, R template void OpDispatchBuilder::PSIGN(OpcodeArgs) { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); Ref Res = PSIGNImpl(Op, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, Res, OpSize::iInvalid); + StoreResultFPR(Op, Res, OpSize::iInvalid); } template void OpDispatchBuilder::PSIGN(OpcodeArgs); @@ -1701,11 +1701,11 @@ template void OpDispatchBuilder::PSIGN(OpcodeArgs); template void OpDispatchBuilder::VPSIGN(OpcodeArgs) { - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Res = PSIGNImpl(Op, ElementSize, Src1, Src2); - StoreResult(FPRClass, Op, Res, OpSize::iInvalid); + StoreResultFPR(Op, Res, OpSize::iInvalid); } template void OpDispatchBuilder::VPSIGN(OpcodeArgs); @@ -1720,25 +1720,25 @@ Ref OpDispatchBuilder::PSRLDOpImpl(OpcodeArgs, IR::OpSize ElementSize, Ref Src, } void OpDispatchBuilder::PSRLDOp(OpcodeArgs, IR::OpSize ElementSize) { - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PSRLDOpImpl(Op, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPSRLDOp(OpcodeArgs, IR::OpSize ElementSize) { const auto DstSize = GetDstSize(Op); const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE; - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Shift = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Shift = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = PSRLDOpImpl(Op, ElementSize, Src, Shift); if (Is128Bit) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PSRLI(OpcodeArgs, IR::OpSize ElementSize) { @@ -1750,9 +1750,9 @@ void OpDispatchBuilder::PSRLI(OpcodeArgs, IR::OpSize ElementSize) { const auto Size = OpSizeFromSrc(Op); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); Ref Shift = _VUShrI(Size, ElementSize, Dest, ShiftConstant); - StoreResult(FPRClass, Op, Shift, OpSize::iInvalid); + StoreResultFPR(Op, Shift, OpSize::iInvalid); } void OpDispatchBuilder::VPSRLIOp(OpcodeArgs, IR::OpSize ElementSize) { @@ -1760,7 +1760,7 @@ void OpDispatchBuilder::VPSRLIOp(OpcodeArgs, IR::OpSize ElementSize) { const auto Is128Bit = Size == OpSize::i128Bit; const uint64_t ShiftConstant = Op->Src[1].Literal(); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = Src; if (ShiftConstant != 0) [[likely]] { @@ -1771,7 +1771,7 @@ void OpDispatchBuilder::VPSRLIOp(OpcodeArgs, IR::OpSize ElementSize) { } } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::PSLLIImpl(OpcodeArgs, IR::OpSize ElementSize, Ref Src, uint64_t Shift) { @@ -1790,10 +1790,10 @@ void OpDispatchBuilder::PSLLI(OpcodeArgs, IR::OpSize ElementSize) { return; } - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); Ref Result = PSLLIImpl(Op, ElementSize, Dest, ShiftConstant); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPSLLIOp(OpcodeArgs, IR::OpSize ElementSize) { @@ -1801,13 +1801,13 @@ void OpDispatchBuilder::VPSLLIOp(OpcodeArgs, IR::OpSize ElementSize) { const auto DstSize = GetDstSize(Op); const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE; - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PSLLIImpl(Op, ElementSize, Src, ShiftConstant); if (ShiftConstant == 0 && Is128Bit) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::PSLLImpl(OpcodeArgs, IR::OpSize ElementSize, Ref Src, Ref ShiftVec) { @@ -1818,25 +1818,25 @@ Ref OpDispatchBuilder::PSLLImpl(OpcodeArgs, IR::OpSize ElementSize, Ref Src, Ref } void OpDispatchBuilder::PSLL(OpcodeArgs, IR::OpSize ElementSize) { - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PSLLImpl(Op, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPSLLOp(OpcodeArgs, IR::OpSize ElementSize) { const auto DstSize = GetDstSize(Op); const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE; - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], OpSize::i128Bit, Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], OpSize::i128Bit, Op->Flags); Ref Result = PSLLImpl(Op, ElementSize, Src1, Src2); if (Is128Bit) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::PSRAOpImpl(OpcodeArgs, IR::OpSize ElementSize, Ref Src, Ref ShiftVec) { @@ -1847,25 +1847,25 @@ Ref OpDispatchBuilder::PSRAOpImpl(OpcodeArgs, IR::OpSize ElementSize, Ref Src, R } void OpDispatchBuilder::PSRAOp(OpcodeArgs, IR::OpSize ElementSize) { - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PSRAOpImpl(Op, ElementSize, Dest, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPSRAOp(OpcodeArgs, IR::OpSize ElementSize) { const auto DstSize = GetDstSize(Op); const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE; - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = PSRAOpImpl(Op, ElementSize, Src1, Src2); if (Is128Bit) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PSRLDQ(OpcodeArgs) { @@ -1877,13 +1877,13 @@ void OpDispatchBuilder::PSRLDQ(OpcodeArgs) { const auto Size = OpSizeFromDst(Op); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); Ref Result = LoadZeroVector(Size); if (Shift < IR::OpSizeToSize(Size)) { Result = _VExtr(Size, OpSize::i8Bit, Result, Dest, Shift); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPSRLDQOp(OpcodeArgs) { @@ -1891,7 +1891,7 @@ void OpDispatchBuilder::VPSRLDQOp(OpcodeArgs) { const auto Is128Bit = DstSize == OpSize::i128Bit; const uint64_t Shift = Op->Src[1].Literal(); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result {}; if (Shift == 0) [[unlikely]] { @@ -1917,7 +1917,7 @@ void OpDispatchBuilder::VPSRLDQOp(OpcodeArgs) { } } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PSLLDQ(OpcodeArgs) { @@ -1929,13 +1929,13 @@ void OpDispatchBuilder::PSLLDQ(OpcodeArgs) { const auto Size = OpSizeFromDst(Op); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); Ref Result = LoadZeroVector(Size); if (Shift < IR::OpSizeToSize(Size)) { Result = _VExtr(Size, OpSize::i8Bit, Dest, Result, IR::OpSizeToSize(Size) - Shift); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPSLLDQOp(OpcodeArgs) { @@ -1944,7 +1944,7 @@ void OpDispatchBuilder::VPSLLDQOp(OpcodeArgs) { const auto Is128Bit = DstSize == OpSize::i128Bit; const uint64_t Shift = Op->Src[1].Literal(); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = Src; @@ -1967,7 +1967,7 @@ void OpDispatchBuilder::VPSLLDQOp(OpcodeArgs) { } } } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PSRAIOp(OpcodeArgs, IR::OpSize ElementSize) { @@ -1979,9 +1979,9 @@ void OpDispatchBuilder::PSRAIOp(OpcodeArgs, IR::OpSize ElementSize) { const auto Size = OpSizeFromDst(Op); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); Ref Result = _VSShrI(Size, ElementSize, Dest, Shift); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPSRAIOp(OpcodeArgs, IR::OpSize ElementSize) { @@ -1989,7 +1989,7 @@ void OpDispatchBuilder::VPSRAIOp(OpcodeArgs, IR::OpSize ElementSize) { const auto Size = OpSizeFromDst(Op); const auto Is128Bit = Size == OpSize::i128Bit; - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = Src; if (Shift != 0) [[likely]] { @@ -2000,19 +2000,19 @@ void OpDispatchBuilder::VPSRAIOp(OpcodeArgs, IR::OpSize ElementSize) { } } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::AVXVariableShiftImpl(OpcodeArgs, IROps IROp) { const auto DstSize = OpSizeFromDst(Op); const auto SrcSize = OpSizeFromSrc(Op); - Ref Vector = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags); - Ref ShiftVector = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], DstSize, Op->Flags); + Ref Vector = LoadSourceFPR_WithOpSize(Op, Op->Src[0], DstSize, Op->Flags); + Ref ShiftVector = LoadSourceFPR_WithOpSize(Op, Op->Src[1], DstSize, Op->Flags); DeriveOp(Shift, IROp, _VUShr(DstSize, SrcSize, Vector, ShiftVector, true)); - StoreResult(FPRClass, Op, Shift, OpSize::iInvalid); + StoreResultFPR(Op, Shift, OpSize::iInvalid); } void OpDispatchBuilder::VPSLLVOp(OpcodeArgs) { @@ -2032,10 +2032,10 @@ void OpDispatchBuilder::MOVDDUPOp(OpcodeArgs) { // unnecessarily zero extend the vector. Otherwise, if // memory, then we want to load the element size exactly. const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : OpSizeFromSrc(Op); - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); Ref Res = _VDupElement(OpSize::i128Bit, OpSizeFromSrc(Op), Src, 0); - StoreResult(FPRClass, Op, Res, OpSize::iInvalid); + StoreResultFPR(Op, Res, OpSize::iInvalid); } void OpDispatchBuilder::VMOVDDUPOp(OpcodeArgs) { @@ -2044,8 +2044,8 @@ void OpDispatchBuilder::VMOVDDUPOp(OpcodeArgs) { const auto Is256Bit = SrcSize == OpSize::i256Bit; const auto MemSize = Is256Bit ? OpSize::i256Bit : OpSize::i64Bit; - Ref Src = IsSrcGPR ? LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags) : - LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], MemSize, Op->Flags); + const auto LoadSize = IsSrcGPR ? SrcSize : MemSize; + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags); Ref Res {}; if (Is256Bit) { @@ -2054,29 +2054,29 @@ void OpDispatchBuilder::VMOVDDUPOp(OpcodeArgs) { Res = _VDupElement(SrcSize, OpSize::i64Bit, Src, 0); } - StoreResult(FPRClass, Op, Res, OpSize::iInvalid); + StoreResultFPR(Op, Res, OpSize::iInvalid); } Ref OpDispatchBuilder::CVTGPR_To_FPRImpl(OpcodeArgs, IR::OpSize DstElementSize, const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op) { const auto SrcSize = OpSizeFromSrc(Op); - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, OpSize::i128Bit, Op->Flags); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Src1Op, OpSize::i128Bit, Op->Flags); Ref Converted {}; if (Src2Op.IsGPR()) { // If the source is a GPR then convert directly from the GPR. - auto Src2 = LoadSource_WithOpSize(GPRClass, Op, Src2Op, GetGPROpSize(), Op->Flags); + auto Src2 = LoadSourceGPR_WithOpSize(Op, Src2Op, GetGPROpSize(), Op->Flags); Converted = _Float_FromGPR_S(DstElementSize, SrcSize, Src2); } else if (SrcSize != DstElementSize) { // If the source is from memory but the Source size and destination size aren't the same, // then it is more optimal to load in to a GPR and convert between GPR->FPR. // ARM GPR->FPR conversion supports different size source and destinations while FPR->FPR doesn't. - auto Src2 = LoadSource(GPRClass, Op, Src2Op, Op->Flags); + auto Src2 = LoadSourceGPR(Op, Src2Op, Op->Flags); Converted = _Float_FromGPR_S(DstElementSize, SrcSize, Src2); } else { // In the case of cvtsi2s{s,d} where the source and destination are the same size, // then it is more optimal to load in to the FPR register directly and convert there. - auto Src2 = LoadSource(FPRClass, Op, Src2Op, Op->Flags); + auto Src2 = LoadSourceFPR(Op, Src2Op, Op->Flags); Converted = _Vector_SToF(SrcSize, SrcSize, Src2); } @@ -2086,7 +2086,7 @@ Ref OpDispatchBuilder::CVTGPR_To_FPRImpl(OpcodeArgs, IR::OpSize DstElementSize, template void OpDispatchBuilder::CVTGPR_To_FPR(OpcodeArgs) { Ref Result = CVTGPR_To_FPRImpl(Op, DstElementSize, Op->Dest, Op->Src[0]); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::CVTGPR_To_FPR(OpcodeArgs); @@ -2095,7 +2095,7 @@ template void OpDispatchBuilder::CVTGPR_To_FPR(OpcodeArgs); template void OpDispatchBuilder::AVXCVTGPR_To_FPR(OpcodeArgs) { Ref Result = CVTGPR_To_FPRImpl(Op, DstElementSize, Op->Src[0], Op->Src[1]); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::AVXCVTGPR_To_FPR(OpcodeArgs); template void OpDispatchBuilder::AVXCVTGPR_To_FPR(OpcodeArgs); @@ -2135,9 +2135,9 @@ void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs) { // unnecessarily zero extend the vector. Otherwise, if // memory, then we want to load the element size exactly. const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : SrcElementSize; - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); Ref Result = CVTFPR_To_GPRImpl(Op, Src, SrcElementSize, HostRoundingMode); - StoreResult(GPRClass, Op, Result, OpSize::iInvalid); + StoreResultGPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs); @@ -2155,9 +2155,9 @@ Ref OpDispatchBuilder::Vector_CVT_Int_To_FloatImpl(OpcodeArgs, IR::OpSize SrcEle // unnecessarily zero extend the vector. Otherwise, if // memory, then we want to load the element size exactly. const auto LoadSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : IR::SizeToOpSize(8 * (IR::OpSizeToSize(Size) / 16)); - return LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags); + return LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags); } else { - return LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + return LoadSourceFPR(Op, Op->Src[0], Op->Flags); } }(); @@ -2173,7 +2173,7 @@ Ref OpDispatchBuilder::Vector_CVT_Int_To_FloatImpl(OpcodeArgs, IR::OpSize SrcEle template void OpDispatchBuilder::Vector_CVT_Int_To_Float(OpcodeArgs) { Ref Result = Vector_CVT_Int_To_FloatImpl(Op, SrcElementSize, Widen); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::Vector_CVT_Int_To_Float(OpcodeArgs); @@ -2220,9 +2220,9 @@ template void OpDispatchBuilder::Vector_CVT_Float_To_Int(OpcodeArgs) { const auto DstSize = OpSizeFromDst(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = Vector_CVT_Float_To_Int32Impl(Op, DstSize, Src, OpSizeFromSrc(Op), SrcElementSize, HostRoundingMode, true); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::Vector_CVT_Float_To_Int(OpcodeArgs); @@ -2238,8 +2238,8 @@ Ref OpDispatchBuilder::Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, IR::OpSize DstE // Otherwise, if it's a memory load, then we only want to load its exact size. const auto Src2Size = Src2Op.IsGPR() ? OpSize::i128Bit : SrcElementSize; - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Src1Op, OpSize::i128Bit, Op->Flags); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Src2Op, Src2Size, Op->Flags); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Src1Op, OpSize::i128Bit, Op->Flags); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Src2Op, Src2Size, Op->Flags); Ref Converted = _Float_FToF(DstElementSize, SrcElementSize, Src2); @@ -2249,7 +2249,7 @@ Ref OpDispatchBuilder::Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, IR::OpSize DstE template void OpDispatchBuilder::Scalar_CVT_Float_To_Float(OpcodeArgs) { Ref Result = Scalar_CVT_Float_To_FloatImpl(Op, DstElementSize, SrcElementSize, Op->Dest, Op->Src[0]); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::Scalar_CVT_Float_To_Float(OpcodeArgs); @@ -2258,7 +2258,7 @@ template void OpDispatchBuilder::Scalar_CVT_Float_To_Float void OpDispatchBuilder::AVXScalar_CVT_Float_To_Float(OpcodeArgs) { Ref Result = Scalar_CVT_Float_To_FloatImpl(Op, DstElementSize, SrcElementSize, Op->Src[0], Op->Src[1]); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::AVXScalar_CVT_Float_To_Float(OpcodeArgs); @@ -2272,7 +2272,7 @@ void OpDispatchBuilder::Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElem const auto LoadSize = IsFloatSrc && !Op->Src[0].IsGPR() ? (SrcSize >> 1) : SrcSize; - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags); Ref Result {}; if (DstElementSize > SrcElementSize) { @@ -2290,11 +2290,11 @@ void OpDispatchBuilder::Vector_CVT_Float_To_Float(OpcodeArgs, IR::OpSize DstElem Result = _VMov(OpSize::i128Bit, Result); } } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); // Always 32-bit. auto ElementSize = OpSize::i32Bit; @@ -2306,7 +2306,7 @@ void OpDispatchBuilder::MMX_To_XMM_Vector_CVT_Int_To_Float(OpcodeArgs) { // Always signed Src = _Vector_SToF(DstSize, ElementSize, Src); - StoreResult(FPRClass, Op, Src, OpSize::iInvalid); + StoreResultFPR(Op, Src, OpSize::iInvalid); } template @@ -2321,9 +2321,9 @@ void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs) { // memory, then we want to load the element size exactly. const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : OpSizeFromSrc(Op); const auto DstSize = OpSizeFromDst(Op); - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); Ref Result = Vector_CVT_Float_To_Int32Impl(Op, DstSize, Src, SrcSize, SrcElementSize, HostRoundingMode, false /* TODO? */); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, DstSize, OpSize::iInvalid); } template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs); @@ -2334,21 +2334,21 @@ template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_IntSrc[0], Op->Flags); + Ref MaskSrc = LoadSourceGPR(Op, Op->Src[0], Op->Flags); // Mask only cares about the top bit of each byte MaskSrc = _VCMPLTZ(Size, OpSize::i8Bit, MaskSrc); // Vector that will overwrite byte elements. - Ref VectorSrc = LoadSource(GPRClass, Op, Op->Dest, Op->Flags); + Ref VectorSrc = LoadSourceGPR(Op, Op->Dest, Op->Flags); // RDI source (DS prefix by default) auto MemDest = MakeSegmentAddress(X86State::REG_RDI, Op->Flags, X86Tables::DecodeFlags::FLAG_DS_PREFIX); - Ref XMMReg = _LoadMem(FPRClass, Size, MemDest, OpSize::i8Bit); + Ref XMMReg = _LoadMemFPR(Size, MemDest, OpSize::i8Bit); // If the Mask element high bit is set then overwrite the element with the source, else keep the memory variant XMMReg = _VBSL(Size, MaskSrc, VectorSrc, XMMReg); - _StoreMem(FPRClass, Size, MemDest, XMMReg, OpSize::i8Bit); + _StoreMemFPR(Size, MemDest, XMMReg, OpSize::i8Bit); } void OpDispatchBuilder::VMASKMOVOpImpl(OpcodeArgs, IR::OpSize ElementSize, IR::OpSize DataSize, bool IsStore, @@ -2358,10 +2358,10 @@ void OpDispatchBuilder::VMASKMOVOpImpl(OpcodeArgs, IR::OpSize ElementSize, IR::O return MakeSegmentAddress(Op, Data, GetGPROpSize()); }; - Ref Mask = LoadSource_WithOpSize(FPRClass, Op, MaskOp, DataSize, Op->Flags); + Ref Mask = LoadSourceFPR_WithOpSize(Op, MaskOp, DataSize, Op->Flags); if (IsStore) { - Ref Data = LoadSource_WithOpSize(FPRClass, Op, DataOp, DataSize, Op->Flags); + Ref Data = LoadSourceFPR_WithOpSize(Op, DataOp, DataSize, Op->Flags); Ref Address = MakeAddress(Op->Dest); _VStoreVectorMasked(DataSize, ElementSize, Mask, Data, Address, Invalid(), MemOffsetType::SXTX, 1); } else { @@ -2373,7 +2373,7 @@ void OpDispatchBuilder::VMASKMOVOpImpl(OpcodeArgs, IR::OpSize ElementSize, IR::O if (Is128Bit) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } } @@ -2398,27 +2398,27 @@ void OpDispatchBuilder::MOVBetweenGPR_FPR(OpcodeArgs, VectorOpType VectorType) { Ref Result {}; if (Op->Src[0].IsGPR()) { // Loading from GPR and moving to Vector. - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], GetGPROpSize(), Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], GetGPROpSize(), Op->Flags); // zext to 128bit Result = _VCastFromGPR(OpSize::i128Bit, OpSizeFromSrc(Op), Src); } else { // Loading from Memory as a scalar. Zero extend - Result = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Result = LoadSourceFPR(Op, Op->Src[0], Op->Flags); } StoreResult_WithAVXInsert(VectorType, FPRClass, Op, Result, OpSize::iInvalid); } else { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); if (Op->Dest.IsGPR()) { const auto ElementSize = OpSizeFromDst(Op); // Extract element from GPR. Zero extending in the process. Src = _VExtractToGPR(OpSizeFromSrc(Op), ElementSize, Src, 0); - StoreResult(GPRClass, Op, Op->Dest, Src, OpSize::iInvalid); + StoreResultGPR(Op, Op->Dest, Src, OpSize::iInvalid); } else { // Storing first element to memory. - Ref Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false}); - _StoreMem(FPRClass, OpSizeFromDst(Op), Dest, Src, OpSize::i8Bit); + Ref Dest = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false}); + _StoreMemFPR(OpSizeFromDst(Op), Dest, Src, OpSize::i8Bit); } } } @@ -2455,13 +2455,13 @@ void OpDispatchBuilder::VFCMPOp(OpcodeArgs) { const auto SrcSize = OpSizeFromSrc(Op); const auto DstSize = OpSizeFromDst(Op); - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags); - Ref Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, DstSize, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); + Ref Dest = LoadSourceFPR_WithOpSize(Op, Op->Dest, DstSize, Op->Flags); const uint8_t CompType = Op->Src[1].Data.Literal.Value; Ref Result = VFCMPOpImpl(OpSizeFromSrc(Op), ElementSize, Dest, Src, CompType); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::VFCMPOp(OpcodeArgs); @@ -2475,11 +2475,11 @@ void OpDispatchBuilder::AVXVFCMPOp(OpcodeArgs) { const auto DstSize = OpSizeFromDst(Op); const uint8_t CompType = Op->Src[2].Literal(); - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], SrcSize, Op->Flags); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Op->Src[0], DstSize, Op->Flags); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], SrcSize, Op->Flags); Ref Result = VFCMPOpImpl(OpSizeFromSrc(Op), ElementSize, Src1, Src2, CompType); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::AVXVFCMPOp(OpcodeArgs); @@ -2552,7 +2552,7 @@ void OpDispatchBuilder::XSaveOpImpl(OpcodeArgs) { // XSTATE_BV section of the header is 8 bytes in size, but we only really // care about setting at most 3 bits in the first byte. We zero out the rest. - _StoreMem(GPRClass, OpSize::i64Bit, RequestedFeatures, Base, Constant(512), OpSize::i8Bit, MemOffsetType::SXTX, 1); + _StoreMemGPR(OpSize::i64Bit, RequestedFeatures, Base, Constant(512), OpSize::i8Bit, MemOffsetType::SXTX, 1); } } @@ -2574,16 +2574,16 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) { } { - auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW)); - _StoreMem(GPRClass, OpSize::i16Bit, MemBase, FCW, OpSize::i16Bit); + auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW)); + _StoreMemGPR(OpSize::i16Bit, MemBase, FCW, OpSize::i16Bit); } - { _StoreMem(GPRClass, OpSize::i16Bit, ReconstructFSW_Helper(), MemBase, Constant(2), OpSize::i16Bit, MemOffsetType::SXTX, 1); } + { _StoreMemGPR(OpSize::i16Bit, ReconstructFSW_Helper(), MemBase, Constant(2), OpSize::i16Bit, MemOffsetType::SXTX, 1); } { // Abridged FTW - auto FTW = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); - _StoreMem(GPRClass, OpSize::i8Bit, FTW, MemBase, Constant(4), OpSize::i8Bit, MemOffsetType::SXTX, 1); + auto FTW = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); + _StoreMemGPR(OpSize::i8Bit, FTW, MemBase, Constant(4), OpSize::i8Bit, MemOffsetType::SXTX, 1); } // BYTE | 0 1 | 2 3 | 4 | 5 | 6 7 | 8 9 | a b | c d | e f | @@ -2637,11 +2637,11 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) { const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit; for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) { - Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass); + Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit)); if (ReducedPrecisionMode) { data = _F80CVTTo(data, OpSize::i64Bit); } - _StoreMem(FPRClass, OpSize::i128Bit, data, MemBase, Constant(16 * i + 32), OpSize::i8Bit, MemOffsetType::SXTX, 1); + _StoreMemFPR(OpSize::i128Bit, data, MemBase, Constant(16 * i + 32), OpSize::i8Bit, MemOffsetType::SXTX, 1); Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst); } } @@ -2650,13 +2650,13 @@ void OpDispatchBuilder::SaveSSEState(Ref MemBase) { const auto NumRegs = Is64BitMode ? 16U : 8U; for (uint32_t i = 0; i < NumRegs; i += 2) { - _StoreMemPair(FPRClass, OpSize::i128Bit, LoadXMMRegister(i), LoadXMMRegister(i + 1), MemBase, i * 16 + 160); + _StoreMemPairFPR(OpSize::i128Bit, LoadXMMRegister(i), LoadXMMRegister(i + 1), MemBase, i * 16 + 160); } } void OpDispatchBuilder::SaveMXCSRState(Ref MemBase) { // Store MXCSR and the mask for all bits. - _StoreMemPair(GPRClass, OpSize::i32Bit, GetMXCSR(), Constant(0xFFFF), MemBase, 24); + _StoreMemPairGPR(OpSize::i32Bit, GetMXCSR(), Constant(0xFFFF), MemBase, 24); } void OpDispatchBuilder::SaveAVXState(Ref MemBase) { @@ -2666,12 +2666,12 @@ void OpDispatchBuilder::SaveAVXState(Ref MemBase) { Ref Upper0 = _VDupElement(OpSize::i256Bit, OpSize::i128Bit, LoadXMMRegister(i + 0), 1); Ref Upper1 = _VDupElement(OpSize::i256Bit, OpSize::i128Bit, LoadXMMRegister(i + 1), 1); - _StoreMemPair(FPRClass, OpSize::i128Bit, Upper0, Upper1, MemBase, i * 16 + 576); + _StoreMemPairFPR(OpSize::i128Bit, Upper0, Upper1, MemBase, i * 16 + 576); } } Ref OpDispatchBuilder::GetMXCSR() { - Ref MXCSR = _LoadContext(OpSize::i32Bit, GPRClass, offsetof(FEXCore::Core::CPUState, mxcsr)); + Ref MXCSR = _LoadContextGPR(OpSize::i32Bit, offsetof(FEXCore::Core::CPUState, mxcsr)); // Mask out unsupported bits // Keeps FZ, RC, exception masks, and DAZ MXCSR = _And(OpSize::i32Bit, MXCSR, Constant(0xFFC0)); @@ -2684,7 +2684,7 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) { RestoreX87State(Mem); RestoreSSEState(Mem); - Ref MXCSR = _LoadMem(GPRClass, OpSize::i32Bit, Mem, Constant(24), OpSize::i32Bit, MemOffsetType::SXTX, 1); + Ref MXCSR = _LoadMemGPR(OpSize::i32Bit, Mem, Constant(24), OpSize::i32Bit, MemOffsetType::SXTX, 1); RestoreMXCSRState(MXCSR); } @@ -2701,7 +2701,7 @@ void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) { // Note: we rematerialize Base/Mask in each block to avoid crossblock // liveness. Ref Base = XSaveBase(Op); - Ref Mask = _LoadMem(GPRClass, OpSize::i64Bit, Base, Constant(512), OpSize::i64Bit, MemOffsetType::SXTX, 1); + Ref Mask = _LoadMemGPR(OpSize::i64Bit, Base, Constant(512), OpSize::i64Bit, MemOffsetType::SXTX, 1); Ref BitFlag = _Bfe(OpSize, FieldSize, BitIndex, Mask); auto CondJump_ = CondJump(BitFlag, CondClass::NEQ); @@ -2745,7 +2745,7 @@ void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) { 1, [this, Op] { Ref Base = XSaveBase(Op); - Ref MXCSR = _LoadMem(GPRClass, OpSize::i32Bit, Base, Constant(24), OpSize::i32Bit, MemOffsetType::SXTX, 1); + Ref MXCSR = _LoadMemGPR(OpSize::i32Bit, Base, Constant(24), OpSize::i32Bit, MemOffsetType::SXTX, 1); RestoreMXCSRState(MXCSR); }, [] { /* Intentionally do nothing*/ }, 2); @@ -2755,24 +2755,24 @@ void OpDispatchBuilder::XRstorOpImpl(OpcodeArgs) { void OpDispatchBuilder::RestoreX87State(Ref MemBase) { _StackForceSlow(); - auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, MemBase, OpSize::i16Bit); - _StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); + auto NewFCW = _LoadMemGPR(OpSize::i16Bit, MemBase, OpSize::i16Bit); + _StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); { - auto NewFSW = _LoadMem(GPRClass, OpSize::i16Bit, MemBase, Constant(2), OpSize::i16Bit, MemOffsetType::SXTX, 1); + auto NewFSW = _LoadMemGPR(OpSize::i16Bit, MemBase, Constant(2), OpSize::i16Bit, MemOffsetType::SXTX, 1); ReconstructX87StateFromFSW_Helper(NewFSW); } { // Abridged FTW - auto NewFTW = _LoadMem(GPRClass, OpSize::i8Bit, MemBase, Constant(4), OpSize::i8Bit, MemOffsetType::SXTX, 1); - _StoreContext(OpSize::i8Bit, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); + auto NewFTW = _LoadMemGPR(OpSize::i8Bit, MemBase, Constant(4), OpSize::i8Bit, MemOffsetType::SXTX, 1); + _StoreContextGPR(OpSize::i8Bit, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); } for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; i += 2) { - auto MMRegs = LoadMemPair(FPRClass, OpSize::i128Bit, MemBase, i * 16 + 32); - _StoreContext(OpSize::i128Bit, FPRClass, MMRegs.Low, MMBaseOffset() + i * 16); - _StoreContext(OpSize::i128Bit, FPRClass, MMRegs.High, MMBaseOffset() + (i + 1) * 16); + auto MMRegs = LoadMemPairFPR(OpSize::i128Bit, MemBase, i * 16 + 32); + _StoreContextFPR(OpSize::i128Bit, MMRegs.Low, MMBaseOffset() + i * 16); + _StoreContextFPR(OpSize::i128Bit, MMRegs.High, MMBaseOffset() + (i + 1) * 16); } } @@ -2780,7 +2780,7 @@ void OpDispatchBuilder::RestoreSSEState(Ref MemBase) { const auto NumRegs = Is64BitMode ? 16U : 8U; for (uint32_t i = 0; i < NumRegs; i += 2) { - auto XMMRegs = LoadMemPair(FPRClass, OpSize::i128Bit, MemBase, i * 16 + 160); + auto XMMRegs = LoadMemPairFPR(OpSize::i128Bit, MemBase, i * 16 + 160); StoreXMMRegister(i, XMMRegs.Low); StoreXMMRegister(i + 1, XMMRegs.High); @@ -2791,7 +2791,7 @@ void OpDispatchBuilder::RestoreMXCSRState(Ref MXCSR) { // Mask out unsupported bits MXCSR = _And(OpSize::i32Bit, MXCSR, Constant(0xFFC0)); - _StoreContext(OpSize::i32Bit, GPRClass, MXCSR, offsetof(FEXCore::Core::CPUState, mxcsr)); + _StoreContextGPR(OpSize::i32Bit, MXCSR, offsetof(FEXCore::Core::CPUState, mxcsr)); // We only support the rounding mode and FTZ bit being set Ref RoundingMode = _Bfe(OpSize::i32Bit, 3, 13, MXCSR); _SetRoundingMode(RoundingMode, true, MXCSR); @@ -2803,7 +2803,7 @@ void OpDispatchBuilder::RestoreAVXState(Ref MemBase) { for (uint32_t i = 0; i < NumRegs; i += 2) { Ref XMMReg0 = LoadXMMRegister(i + 0); Ref XMMReg1 = LoadXMMRegister(i + 1); - auto YMMHRegs = LoadMemPair(FPRClass, OpSize::i128Bit, MemBase, i * 16 + 576); + auto YMMHRegs = LoadMemPairFPR(OpSize::i128Bit, MemBase, i * 16 + 576); StoreXMMRegister(i + 0, _VInsElement(OpSize::i256Bit, OpSize::i128Bit, 1, 0, XMMReg0, YMMHRegs.Low)); StoreXMMRegister(i + 1, _VInsElement(OpSize::i256Bit, OpSize::i128Bit, 1, 0, XMMReg1, YMMHRegs.High)); } @@ -2818,7 +2818,7 @@ void OpDispatchBuilder::DefaultX87State(OpcodeArgs) { // all of the ST0-7/MM0-7 registers to zero. Ref ZeroVector = LoadZeroVector(OpSize::i64Bit); for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) { - _StoreContext(OpSize::i128Bit, FPRClass, ZeroVector, MMBaseOffset() + i * 16); + _StoreContextFPR(OpSize::i128Bit, ZeroVector, MMBaseOffset() + i * 16); } } @@ -2850,7 +2850,7 @@ Ref OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand const auto Is256Bit = DstSize == OpSize::i256Bit; const auto Index = Imm.Literal(); - Ref Src2Node = LoadSource(FPRClass, Op, Src2, Op->Flags); + Ref Src2Node = LoadSourceFPR(Op, Src2, Op->Flags); if (Index == 0) { if (IsAVX && !Is256Bit) { // 128-bit AVX needs to zero the upper bits. @@ -2859,7 +2859,7 @@ Ref OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand return Src2Node; } } - Ref Src1Node = LoadSource(FPRClass, Op, Src1, Op->Flags); + Ref Src1Node = LoadSourceFPR(Op, Src1, Op->Flags); if (Index >= (IR::OpSizeToSize(SanitizedDstSize) * 2)) { // If the immediate is greater than both vectors combined then it zeroes the vector @@ -2879,19 +2879,19 @@ Ref OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand void OpDispatchBuilder::PAlignrOp(OpcodeArgs) { Ref Result = PALIGNROpImpl(Op, Op->Dest, Op->Src[0], Op->Src[1], false); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPALIGNROp(OpcodeArgs) { Ref Result = PALIGNROpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2], true); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::UCOMISxOp(OpcodeArgs) { const auto SrcSize = Op->Src[0].IsGPR() ? GetGuestVectorLength() : ElementSize; - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, GetGuestVectorLength(), Op->Flags); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Op->Dest, GetGuestVectorLength(), Op->Flags); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); Comiss(ElementSize, Src1, Src2); } @@ -2900,21 +2900,21 @@ template void OpDispatchBuilder::UCOMISxOp(OpcodeArgs); template void OpDispatchBuilder::UCOMISxOp(OpcodeArgs); void OpDispatchBuilder::LDMXCSR(OpcodeArgs) { - Ref Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, OpSize::i32Bit, Op->Flags); + Ref Dest = LoadSourceGPR_WithOpSize(Op, Op->Dest, OpSize::i32Bit, Op->Flags); RestoreMXCSRState(Dest); } void OpDispatchBuilder::STMXCSR(OpcodeArgs) { - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, GetMXCSR(), OpSize::i32Bit, OpSize::iInvalid); + StoreResultGPR_WithOpSize(Op, Op->Dest, GetMXCSR(), OpSize::i32Bit, OpSize::iInvalid); } template void OpDispatchBuilder::PACKUSOp(OpcodeArgs) { - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = _VSQXTUNPair(OpSizeFromSrc(Op), ElementSize, Dest, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::PACKUSOp(OpcodeArgs); @@ -2924,8 +2924,8 @@ void OpDispatchBuilder::VPACKUSOp(OpcodeArgs, IR::OpSize ElementSize) { const auto DstSize = OpSizeFromDst(Op); const auto Is256Bit = DstSize == OpSize::i256Bit; - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = _VSQXTUNPair(OpSizeFromSrc(Op), ElementSize, Src1, Src2); if (Is256Bit) { @@ -2933,16 +2933,16 @@ void OpDispatchBuilder::VPACKUSOp(OpcodeArgs, IR::OpSize ElementSize) { Ref Swapped = _VInsElement(DstSize, OpSize::i64Bit, 2, 1, Result, Result); Result = _VInsElement(DstSize, OpSize::i64Bit, 1, 2, Swapped, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::PACKSSOp(OpcodeArgs) { - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = _VSQXTNPair(OpSizeFromSrc(Op), ElementSize, Dest, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::PACKSSOp(OpcodeArgs); @@ -2952,8 +2952,8 @@ void OpDispatchBuilder::VPACKSSOp(OpcodeArgs, IR::OpSize ElementSize) { const auto DstSize = OpSizeFromDst(Op); const auto Is256Bit = DstSize == OpSize::i256Bit; - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = _VSQXTNPair(OpSizeFromSrc(Op), ElementSize, Src1, Src2); if (Is256Bit) { @@ -2961,7 +2961,7 @@ void OpDispatchBuilder::VPACKSSOp(OpcodeArgs, IR::OpSize ElementSize) { Ref Swapped = _VInsElement(DstSize, OpSize::i64Bit, 2, 1, Result, Result); Result = _VInsElement(DstSize, OpSize::i64Bit, 1, 2, Swapped, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::PMULLOpImpl(OpSize Size, IR::OpSize ElementSize, bool Signed, Ref Src1, Ref Src2) { @@ -2987,11 +2987,11 @@ template void OpDispatchBuilder::PMULLOp(OpcodeArgs) { static_assert(ElementSize == OpSize::i32Bit, "Currently only handles 32-bit -> 64-bit"); - Ref Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Res = PMULLOpImpl(OpSizeFromSrc(Op), ElementSize, Signed, Src1, Src2); - StoreResult(FPRClass, Op, Res, OpSize::iInvalid); + StoreResultFPR(Op, Res, OpSize::iInvalid); } template void OpDispatchBuilder::PMULLOp(OpcodeArgs); @@ -3001,11 +3001,11 @@ template void OpDispatchBuilder::VPMULLOp(OpcodeArgs) { static_assert(ElementSize == OpSize::i32Bit, "Currently only handles 32-bit -> 64-bit"); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = PMULLOpImpl(OpSizeFromSrc(Op), ElementSize, Signed, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::VPMULLOp(OpcodeArgs); @@ -3013,7 +3013,7 @@ template void OpDispatchBuilder::VPMULLOp(OpcodeArgs); template void OpDispatchBuilder::MOVQ2DQ(OpcodeArgs) { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); // This instruction is a bit special in that if the source is MMX then it zexts to 128bit if constexpr (ToXMM) { @@ -3023,7 +3023,7 @@ void OpDispatchBuilder::MOVQ2DQ(OpcodeArgs) { StoreXMMRegister(Index, Src); } else { // This is simple, just store the result - StoreResult(FPRClass, Op, Src, OpSize::iInvalid); + StoreResultFPR(Op, Src, OpSize::iInvalid); } } @@ -3049,11 +3049,11 @@ Ref OpDispatchBuilder::ADDSUBPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Sr template void OpDispatchBuilder::ADDSUBPOp(OpcodeArgs) { - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = ADDSUBPOpImpl(OpSizeFromSrc(Op), ElementSize, Dest, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::ADDSUBPOp(OpcodeArgs); @@ -3061,11 +3061,11 @@ template void OpDispatchBuilder::ADDSUBPOp(OpcodeArgs); template void OpDispatchBuilder::VADDSUBPOp(OpcodeArgs) { - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = ADDSUBPOpImpl(OpSizeFromSrc(Op), ElementSize, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::VADDSUBPOp(OpcodeArgs); @@ -3074,21 +3074,21 @@ template void OpDispatchBuilder::VADDSUBPOp(OpcodeArgs); void OpDispatchBuilder::PFNACCOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); auto DestUnzip = _VUnZip(Size, OpSize::i32Bit, Dest, Src); auto SrcUnzip = _VUnZip2(Size, OpSize::i32Bit, Dest, Src); auto Result = _VFSub(Size, OpSize::i32Bit, DestUnzip, SrcUnzip); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PFPNACCOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref ResAdd {}; Ref ResSub {}; @@ -3099,19 +3099,19 @@ void OpDispatchBuilder::PFPNACCOp(OpcodeArgs) { auto Result = _VInsElement(OpSize::i64Bit, OpSize::i32Bit, 1, 0, ResSub, ResAdd); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PSWAPDOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); auto Result = _VRev64(Size, OpSize::i32Bit, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PI2FWOp(OpcodeArgs) { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); const auto Size = OpSizeFromDst(Op); @@ -3125,11 +3125,11 @@ void OpDispatchBuilder::PI2FWOp(OpcodeArgs) { // int32_t to float Src = _Vector_SToF(Size, OpSize::i32Bit, Src); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Src, Size, OpSize::iInvalid); } void OpDispatchBuilder::PF2IWOp(OpcodeArgs) { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); const auto Size = OpSizeFromDst(Op); @@ -3142,14 +3142,14 @@ void OpDispatchBuilder::PF2IWOp(OpcodeArgs) { // Now we need to sign extend the 16bit value to 32-bit Src = _VSXTL(Size, OpSize::i16Bit, Src); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Src, Size, OpSize::iInvalid); } void OpDispatchBuilder::PMULHRWOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Res {}; @@ -3165,14 +3165,14 @@ void OpDispatchBuilder::PMULHRWOp(OpcodeArgs) { // Now shift and narrow to convert 32-bit values to 16bit, storing the top 16bits Res = _VUShrNI(Size << 1, OpSize::i32Bit, Res, 16); - StoreResult(FPRClass, Op, Res, OpSize::iInvalid); + StoreResultFPR(Op, Res, OpSize::iInvalid); } template void OpDispatchBuilder::VPFCMPOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, OpSizeFromDst(Op), Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR_WithOpSize(Op, Op->Dest, OpSizeFromDst(Op), Op->Flags); Ref Result {}; // This maps 1:1 to an AArch64 NEON Op @@ -3190,7 +3190,7 @@ void OpDispatchBuilder::VPFCMPOp(OpcodeArgs) { default: LOGMAN_MSG_A_FMT("Unknown Comparison type: {}", CompType); break; } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::VPFCMPOp<0>(OpcodeArgs); @@ -3223,21 +3223,21 @@ Ref OpDispatchBuilder::PMADDWDOpImpl(IR::OpSize Size, Ref Src1, Ref Src2) { void OpDispatchBuilder::PMADDWD(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PMADDWDOpImpl(Size, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPMADDWDOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = PMADDWDOpImpl(Size, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::PMADDUBSWOpImpl(IR::OpSize Size, Ref Src1, Ref Src2) { @@ -3285,21 +3285,21 @@ Ref OpDispatchBuilder::PMADDUBSWOpImpl(IR::OpSize Size, Ref Src1, Ref Src2) { void OpDispatchBuilder::PMADDUBSW(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PMADDUBSWOpImpl(Size, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPMADDUBSWOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = PMADDUBSWOpImpl(Size, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2) { @@ -3313,11 +3313,11 @@ Ref OpDispatchBuilder::PMULHWOpImpl(OpcodeArgs, bool Signed, Ref Src1, Ref Src2) template void OpDispatchBuilder::PMULHW(OpcodeArgs) { - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PMULHWOpImpl(Op, Signed, Dest, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::PMULHW(OpcodeArgs); @@ -3328,14 +3328,14 @@ void OpDispatchBuilder::VPMULHWOp(OpcodeArgs) { const auto DstSize = GetDstSize(Op); const auto Is128Bit = DstSize == Core::CPUState::XMM_SSE_REG_SIZE; - Ref Dest = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = PMULHWOpImpl(Op, Signed, Dest, Src); if (Is128Bit) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::VPMULHWOp(OpcodeArgs); @@ -3372,19 +3372,19 @@ Ref OpDispatchBuilder::PMULHRSWOpImpl(OpSize Size, Ref Src1, Ref Src2) { } void OpDispatchBuilder::PMULHRSW(OpcodeArgs) { - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PMULHRSWOpImpl(OpSizeFromSrc(Op), Dest, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPMULHRSWOp(OpcodeArgs) { - Ref Dest = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = PMULHRSWOpImpl(OpSizeFromSrc(Op), Dest, Src); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::HSUBPOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref Src1, Ref Src2) { @@ -3395,10 +3395,10 @@ Ref OpDispatchBuilder::HSUBPOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref S template void OpDispatchBuilder::HSUBP(OpcodeArgs) { - Ref Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = HSUBPOpImpl(OpSizeFromSrc(Op), ElementSize, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::HSUBP(OpcodeArgs); @@ -3408,8 +3408,8 @@ void OpDispatchBuilder::VHSUBPOp(OpcodeArgs, IR::OpSize ElementSize) { const auto DstSize = OpSizeFromDst(Op); const auto Is256Bit = DstSize == OpSize::i256Bit; - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = HSUBPOpImpl(OpSizeFromSrc(Op), ElementSize, Src1, Src2); Ref Dest = Result; @@ -3418,7 +3418,7 @@ void OpDispatchBuilder::VHSUBPOp(OpcodeArgs, IR::OpSize ElementSize) { Dest = _VInsElement(DstSize, OpSize::i64Bit, 2, 1, Dest, Result); } - StoreResult(FPRClass, Op, Dest, OpSize::iInvalid); + StoreResultFPR(Op, Dest, OpSize::iInvalid); } Ref OpDispatchBuilder::PHSUBOpImpl(OpSize Size, Ref Src1, Ref Src2, IR::OpSize ElementSize) { @@ -3429,10 +3429,10 @@ Ref OpDispatchBuilder::PHSUBOpImpl(OpSize Size, Ref Src1, Ref Src2, IR::OpSize E template void OpDispatchBuilder::PHSUB(OpcodeArgs) { - Ref Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PHSUBOpImpl(OpSizeFromSrc(Op), Src1, Src2, ElementSize); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::PHSUB(OpcodeArgs); @@ -3442,14 +3442,14 @@ void OpDispatchBuilder::VPHSUBOp(OpcodeArgs, IR::OpSize ElementSize) { const auto DstSize = OpSizeFromDst(Op); const auto Is256Bit = DstSize == OpSize::i256Bit; - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = PHSUBOpImpl(OpSizeFromSrc(Op), Src1, Src2, ElementSize); if (Is256Bit) { Ref Inserted = _VInsElement(DstSize, OpSize::i64Bit, 1, 2, Result, Result); Result = _VInsElement(DstSize, OpSize::i64Bit, 2, 1, Inserted, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::PHADDSOpImpl(OpSize Size, Ref Src1, Ref Src2) { @@ -3463,19 +3463,19 @@ Ref OpDispatchBuilder::PHADDSOpImpl(OpSize Size, Ref Src1, Ref Src2) { } void OpDispatchBuilder::PHADDS(OpcodeArgs) { - Ref Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PHADDSOpImpl(OpSizeFromSrc(Op), Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPHADDSWOp(OpcodeArgs) { const auto SrcSize = OpSizeFromSrc(Op); const auto Is256Bit = SrcSize == OpSize::i256Bit; - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = PHADDSOpImpl(OpSizeFromSrc(Op), Src1, Src2); Ref Dest = Result; @@ -3485,7 +3485,7 @@ void OpDispatchBuilder::VPHADDSWOp(OpcodeArgs) { Dest = _VInsElement(SrcSize, OpSize::i64Bit, 2, 1, Dest, Result); } - StoreResult(FPRClass, Op, Dest, OpSize::iInvalid); + StoreResultFPR(Op, Dest, OpSize::iInvalid); } Ref OpDispatchBuilder::PHSUBSOpImpl(OpSize Size, Ref Src1, Ref Src2) { @@ -3499,18 +3499,18 @@ Ref OpDispatchBuilder::PHSUBSOpImpl(OpSize Size, Ref Src1, Ref Src2) { } void OpDispatchBuilder::PHSUBS(OpcodeArgs) { - Ref Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PHSUBSOpImpl(OpSizeFromSrc(Op), Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPHSUBSWOp(OpcodeArgs) { const auto DstSize = OpSizeFromDst(Op); const auto Is256Bit = DstSize == OpSize::i256Bit; - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = PHSUBSOpImpl(OpSizeFromSrc(Op), Src1, Src2); Ref Dest = Result; @@ -3519,7 +3519,7 @@ void OpDispatchBuilder::VPHSUBSWOp(OpcodeArgs) { Dest = _VInsElement(DstSize, OpSize::i64Bit, 2, 1, Dest, Result); } - StoreResult(FPRClass, Op, Dest, OpSize::iInvalid); + StoreResultFPR(Op, Dest, OpSize::iInvalid); } Ref OpDispatchBuilder::PSADBWOpImpl(IR::OpSize Size, Ref Src1, Ref Src2) { @@ -3564,21 +3564,21 @@ Ref OpDispatchBuilder::PSADBWOpImpl(IR::OpSize Size, Ref Src1, Ref Src2) { void OpDispatchBuilder::PSADBW(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = PSADBWOpImpl(Size, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPSADBWOp(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = PSADBWOpImpl(Size, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::ExtendVectorElementsImpl(OpcodeArgs, IR::OpSize ElementSize, IR::OpSize DstElementSize, bool Signed) { @@ -3586,14 +3586,14 @@ Ref OpDispatchBuilder::ExtendVectorElementsImpl(OpcodeArgs, IR::OpSize ElementSi const auto GetSrc = [&] { if (Op->Src[0].IsGPR()) { - return LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], DstSize, Op->Flags); + return LoadSourceFPR_WithOpSize(Op, Op->Src[0], DstSize, Op->Flags); } else { // For memory operands the 256-bit variant loads twice the size specified in the table. const auto Is256Bit = DstSize == OpSize::i256Bit; const auto SrcSize = OpSizeFromSrc(Op); const auto LoadSize = Is256Bit ? IR::SizeToOpSize(IR::OpSizeToSize(SrcSize) * 2) : SrcSize; - return LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], LoadSize, Op->Flags); + return LoadSourceFPR_WithOpSize(Op, Op->Src[0], LoadSize, Op->Flags); } }; @@ -3614,7 +3614,7 @@ Ref OpDispatchBuilder::ExtendVectorElementsImpl(OpcodeArgs, IR::OpSize ElementSi template void OpDispatchBuilder::ExtendVectorElements(OpcodeArgs) { Ref Result = ExtendVectorElementsImpl(Op, ElementSize, DstElementSize, Signed); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::ExtendVectorElements(OpcodeArgs); @@ -3640,12 +3640,12 @@ void OpDispatchBuilder::VectorRound(OpcodeArgs) { // No need to zero extend the vector in the event we have a // scalar source, especially since it's only inserted into another vector. const auto SrcSize = OpSizeFromSrc(Op); - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); const uint64_t Mode = Op->Src[1].Literal(); Src = VectorRoundImpl(OpSizeFromDst(Op), ElementSize, Src, Mode); - StoreResult(FPRClass, Op, Src, OpSize::iInvalid); + StoreResultFPR(Op, Src, OpSize::iInvalid); } template void OpDispatchBuilder::VectorRound(OpcodeArgs); @@ -3659,10 +3659,10 @@ void OpDispatchBuilder::AVXVectorRound(OpcodeArgs) { // scalar source, especially since it's only inserted into another vector. const auto SrcSize = OpSizeFromSrc(Op); - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcSize, Op->Flags); Ref Result = VectorRoundImpl(OpSizeFromDst(Op), ElementSize, Src, Mode); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::AVXVectorRound(OpcodeArgs); @@ -3908,10 +3908,10 @@ template void OpDispatchBuilder::VectorBlend(OpcodeArgs) { uint8_t Select = Op->Src[1].Literal(); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Dest = VectorBlend(OpSize::i128Bit, ElementSize, Dest, Src, Select); - StoreResult(FPRClass, Op, Dest, OpSize::iInvalid); + StoreResultFPR(Op, Dest, OpSize::iInvalid); } template void OpDispatchBuilder::VectorBlend(OpcodeArgs); @@ -3921,8 +3921,8 @@ template void OpDispatchBuilder::VectorBlend(OpcodeArgs); void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize) { const auto Size = OpSizeFromSrc(Op); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); auto Mask = LoadXMMRegister(0); @@ -3935,15 +3935,15 @@ void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize) auto Result = _VBSL(Size, Mask, Src, Dest); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::AVXVectorVariableBlend(OpcodeArgs, IR::OpSize ElementSize) { const auto SrcSize = OpSizeFromSrc(Op); const auto ElementSizeBits = IR::OpSizeAsBits(ElementSize); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); // Mask register is encoded within bits [7:4] of the selector const auto Src3Selector = Op->Src[2].Literal(); @@ -3951,7 +3951,7 @@ void OpDispatchBuilder::AVXVectorVariableBlend(OpcodeArgs, IR::OpSize ElementSiz Ref Shifted = _VSShrI(SrcSize, ElementSize, Mask, ElementSizeBits - 1); Ref Result = _VBSL(SrcSize, Shifted, Src2, Src1); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::PTestOpImpl(OpSize Size, Ref Dest, Ref Src) { @@ -3977,8 +3977,8 @@ void OpDispatchBuilder::PTestOpImpl(OpSize Size, Ref Dest, Ref Src) { } void OpDispatchBuilder::PTestOp(OpcodeArgs) { - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); PTestOpImpl(OpSizeFromSrc(Op), Dest, Src); } @@ -4012,8 +4012,8 @@ void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref template void OpDispatchBuilder::VTESTPOp(OpcodeArgs) { - Ref Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); VTESTOpImpl(OpSizeFromSrc(Op), ElementSize, Src1, Src2); } @@ -4023,7 +4023,7 @@ template void OpDispatchBuilder::VTESTPOp(OpcodeArgs); Ref OpDispatchBuilder::PHMINPOSUWOpImpl(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); // Setup a vector swizzle // Initially load a 64-bit mask of immediates @@ -4061,7 +4061,7 @@ Ref OpDispatchBuilder::PHMINPOSUWOpImpl(OpcodeArgs) { void OpDispatchBuilder::PHMINPOSUWOp(OpcodeArgs) { Ref Result = PHMINPOSUWOpImpl(Op); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::DPPOpImpl(IR::OpSize DstSize, Ref Src1, Ref Src2, uint8_t Mask, IR::OpSize ElementSize) { @@ -4243,11 +4243,11 @@ Ref OpDispatchBuilder::DPPOpImpl(IR::OpSize DstSize, Ref Src1, Ref Src2, uint8_t template void OpDispatchBuilder::DPPOp(OpcodeArgs) { - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = DPPOpImpl(OpSizeFromDst(Op), Dest, Src, Op->Src[1].Literal(), ElementSize); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::DPPOp(OpcodeArgs); @@ -4262,8 +4262,8 @@ Ref OpDispatchBuilder::VDPPSOpImpl(OpcodeArgs, const X86Tables::DecodedOperand& const auto DstSize = OpSizeFromDst(Op); - Ref Src1V = LoadSource(FPRClass, Op, Src1, Op->Flags); - Ref Src2V = LoadSource(FPRClass, Op, Src2, Op->Flags); + Ref Src1V = LoadSourceFPR(Op, Src1, Op->Flags); + Ref Src2V = LoadSourceFPR(Op, Src2, Op->Flags); Ref ZeroVec = LoadZeroVector(DstSize); @@ -4312,15 +4312,15 @@ void OpDispatchBuilder::VDPPOp(OpcodeArgs) { // 256-bit DPPS isn't handled by the 128-bit solution. Result = VDPPSOpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2]); } else { - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Result = DPPOpImpl(DstSize, Src1, Src2, Op->Src[2].Literal(), ElementSize); } // We don't need to emit a _VMov to clear the upper lane, since DPPOpImpl uses a zero vector // to construct the results, so the upper lane will always be cleared for the 128-bit version. - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::VDPPOp(OpcodeArgs); @@ -4416,32 +4416,32 @@ Ref OpDispatchBuilder::MPSADBWOpImpl(IR::OpSize SrcSize, Ref Src1, Ref Src2, uin void OpDispatchBuilder::MPSADBWOp(OpcodeArgs) { const uint8_t Select = Op->Src[1].Literal(); const auto SrcSize = OpSizeFromSrc(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = MPSADBWOpImpl(SrcSize, Src1, Src2, Select); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VMPSADBWOp(OpcodeArgs) { const uint8_t Select = Op->Src[2].Literal(); const auto SrcSize = OpSizeFromSrc(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = MPSADBWOpImpl(SrcSize, Src1, Src2, Select); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VINSERTOp(OpcodeArgs) { const auto DstSize = OpSizeFromDst(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], OpSize::i128Bit, Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], OpSize::i128Bit, Op->Flags); const auto Selector = Op->Src[2].Literal() & 1; Ref Result = _VInsElement(DstSize, OpSize::i128Bit, Selector, 0, Src1, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VCVTPH2PSOp(OpcodeArgs) { @@ -4451,10 +4451,10 @@ void OpDispatchBuilder::VCVTPH2PSOp(OpcodeArgs) { const auto DstSize = OpSizeFromDst(Op); const auto SrcLoadSize = Op->Src[0].IsGPR() ? DstSize : IR::SizeToOpSize(IR::OpSizeToSize(DstSize) / 2); - Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcLoadSize, Op->Flags); + Ref Src = LoadSourceFPR_WithOpSize(Op, Op->Src[0], SrcLoadSize, Op->Flags); Ref Result = _Vector_FToF(DstSize, OpSize::i32Bit, Src, OpSize::i16Bit); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VCVTPS2PHOp(OpcodeArgs) { @@ -4464,7 +4464,7 @@ void OpDispatchBuilder::VCVTPS2PHOp(OpcodeArgs) { const auto Imm8 = Op->Src[1].Literal(); const auto UseMXCSR = (Imm8 & 0b100) != 0; - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = nullptr; if (UseMXCSR) { @@ -4487,13 +4487,13 @@ void OpDispatchBuilder::VCVTPS2PHOp(OpcodeArgs) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, StoreSize, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Dest, Result, StoreSize, OpSize::iInvalid); } void OpDispatchBuilder::VPERM2Op(OpcodeArgs) { const auto DstSize = OpSizeFromDst(Op); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); const auto Selector = Op->Src[2].Literal(); Ref Result = LoadZeroVector(DstSize); @@ -4515,7 +4515,7 @@ void OpDispatchBuilder::VPERM2Op(OpcodeArgs) { Result = SelectElement(1, (Selector >> 4) & 0b11); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::VPERMDIndices(OpSize DstSize, Ref Indices, Ref IndexMask, Ref Repeating3210) { @@ -4583,8 +4583,8 @@ Ref OpDispatchBuilder::VPERMDIndices(OpSize DstSize, Ref Indices, Ref IndexMask, void OpDispatchBuilder::VPERMDOp(OpcodeArgs) { const auto DstSize = OpSizeFromDst(Op); - Ref Indices = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Indices = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[1], Op->Flags); // Get rid of any junk unrelated to the relevant selector index bits (bits [2:0]) Ref IndexMask = _VectorImm(DstSize, OpSize::i32Bit, 0b111); @@ -4596,12 +4596,12 @@ void OpDispatchBuilder::VPERMDOp(OpcodeArgs) { // Now lets finally shuffle this bad boy around. Ref Result = _VTBL1(DstSize, Src, FinalIndices); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPERMQOp(OpcodeArgs) { const auto DstSize = OpSizeFromDst(Op); - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); const auto Selector = Op->Src[1].Literal(); Ref Result {}; @@ -4618,7 +4618,7 @@ void OpDispatchBuilder::VPERMQOp(OpcodeArgs) { Result = _VInsElement(DstSize, OpSize::i64Bit, i, SrcIndex, Result, Src); } } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::VBLENDOpImpl(IR::OpSize VecSize, IR::OpSize ElementSize, Ref Src1, Ref Src2, Ref ZeroRegister, uint64_t Selector) { @@ -4640,24 +4640,24 @@ void OpDispatchBuilder::VBLENDPDOp(OpcodeArgs) { const auto Is256Bit = DstSize == OpSize::i256Bit; const auto Selector = Op->Src[2].Literal(); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); if (Selector == 0) { Ref Result = Is256Bit ? Src1 : _VMov(OpSize::i128Bit, Src1); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); return; } // Only the first four bits of the 8-bit immediate are used, so only check them. if (((Selector & 0b11) == 0b11 && !Is256Bit) || (Selector & 0b1111) == 0b1111) { Ref Result = Is256Bit ? Src2 : _VMov(OpSize::i128Bit, Src2); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); return; } const auto ZeroRegister = LoadZeroVector(DstSize); Ref Result = VBLENDOpImpl(DstSize, OpSize::i64Bit, Src1, Src2, ZeroRegister, Selector); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPBLENDDOp(OpcodeArgs) { @@ -4665,8 +4665,8 @@ void OpDispatchBuilder::VPBLENDDOp(OpcodeArgs) { const auto Is256Bit = DstSize == OpSize::i256Bit; const auto Selector = Op->Src[2].Literal(); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); // Each bit in the selector chooses between Src1 and Src2. // If a bit is set, then we select it's corresponding 32-bit element from Src2 @@ -4679,11 +4679,11 @@ void OpDispatchBuilder::VPBLENDDOp(OpcodeArgs) { if (Selector == 0) { Ref Result = Is256Bit ? Src1 : _VMov(OpSize::i128Bit, Src1); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); return; } if (Selector == 0xFF && Is256Bit) { - StoreResult(FPRClass, Op, Src2, OpSize::iInvalid); + StoreResultFPR(Op, Src2, OpSize::iInvalid); return; } // The only bits we care about from the 8-bit immediate for 128-bit operations @@ -4691,7 +4691,7 @@ void OpDispatchBuilder::VPBLENDDOp(OpcodeArgs) { // silliness is going on and the upper bits are being set even when they'll // be ignored if ((Selector & 0xF) == 0xF && !Is256Bit) { - StoreResult(FPRClass, Op, _VMov(OpSize::i128Bit, Src2), OpSize::iInvalid); + StoreResultFPR(Op, _VMov(OpSize::i128Bit, Src2), OpSize::iInvalid); return; } @@ -4700,7 +4700,7 @@ void OpDispatchBuilder::VPBLENDDOp(OpcodeArgs) { if (!Is256Bit) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VPBLENDWOp(OpcodeArgs) { @@ -4708,17 +4708,17 @@ void OpDispatchBuilder::VPBLENDWOp(OpcodeArgs) { const auto Is128Bit = DstSize == OpSize::i128Bit; const auto Selector = Op->Src[2].Literal(); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); if (Selector == 0) { Ref Result = Is128Bit ? _VMov(OpSize::i128Bit, Src1) : Src1; - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); return; } if (Selector == 0xFF) { Ref Result = Is128Bit ? _VMov(OpSize::i128Bit, Src2) : Src2; - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); return; } @@ -4732,7 +4732,7 @@ void OpDispatchBuilder::VPBLENDWOp(OpcodeArgs) { if (Is128Bit) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VZEROOp(OpcodeArgs) { @@ -4765,7 +4765,7 @@ void OpDispatchBuilder::VPERMILImmOp(OpcodeArgs, IR::OpSize ElementSize) { const auto Is256Bit = DstSize == OpSize::i256Bit; const auto Selector = Op->Src[1].Literal() & 0xFF; - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Result = LoadZeroVector(DstSize); if (ElementSize == OpSize::i64Bit) { @@ -4790,7 +4790,7 @@ void OpDispatchBuilder::VPERMILImmOp(OpcodeArgs, IR::OpSize ElementSize) { } } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } Ref OpDispatchBuilder::VPERMILRegOpImpl(OpSize DstSize, IR::OpSize ElementSize, Ref Src, Ref Indices) { @@ -4842,11 +4842,11 @@ Ref OpDispatchBuilder::VPERMILRegOpImpl(OpSize DstSize, IR::OpSize ElementSize, template void OpDispatchBuilder::VPERMILRegOp(OpcodeArgs) { - Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Indices = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Indices = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result = VPERMILRegOpImpl(OpSizeFromDst(Op), ElementSize, Src, Indices); - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } template void OpDispatchBuilder::VPERMILRegOp(OpcodeArgs); @@ -4861,8 +4861,8 @@ void OpDispatchBuilder::PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask // instructions in the Intel Software Development Manual). // // So, we specify Src2 as having an alignment of 1 to indicate this. - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, OpSize::i128Bit, Op->Flags); - Ref Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::i128Bit, Op->Flags, {.Align = OpSize::i8Bit}); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Op->Dest, OpSize::i128Bit, Op->Flags); + Ref Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::i128Bit, Op->Flags, {.Align = OpSize::i8Bit}); Ref IntermediateResult {}; if (IsExplicit) { @@ -4953,13 +4953,13 @@ void OpDispatchBuilder::VFMAImpl(OpcodeArgs, IROps IROp, bool Scalar, uint8_t Sr const OpSize ElementSize = Op->Flags & X86Tables::DecodeFlags::FLAG_OPTION_AVX_W ? OpSize::i64Bit : OpSize::i32Bit; - Ref Dest = LoadSource_WithOpSize(FPRClass, Op, Op->Dest, Size, Op->Flags); - Ref Src1 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Size, Op->Flags); + Ref Dest = LoadSourceFPR_WithOpSize(Op, Op->Dest, Size, Op->Flags); + Ref Src1 = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Size, Op->Flags); Ref Src2 {}; if (Op->Src[1].IsGPR()) { - Src2 = LoadSource_WithOpSize(FPRClass, Op, Op->Src[1], Size, Op->Flags); + Src2 = LoadSourceFPR_WithOpSize(Op, Op->Src[1], Size, Op->Flags); } else { - Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); } Ref Sources[3] = { @@ -4978,7 +4978,7 @@ void OpDispatchBuilder::VFMAImpl(OpcodeArgs, IROps IROp, bool Scalar, uint8_t Sr if (!Is256Bit) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } void OpDispatchBuilder::VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, uint8_t Src2Idx, uint8_t AddendIdx) { @@ -4987,9 +4987,9 @@ void OpDispatchBuilder::VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, const OpSize ElementSize = Op->Flags & X86Tables::DecodeFlags::FLAG_OPTION_AVX_W ? OpSize::i64Bit : OpSize::i32Bit; - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Src1 = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); - Ref Src2 = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Sources[3] = { Dest, @@ -5012,7 +5012,7 @@ void OpDispatchBuilder::VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, if (!Is256Bit) { Result = _VMov(OpSize::i128Bit, Result); } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); } OpDispatchBuilder::RefVSIB OpDispatchBuilder::LoadVSIB(const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand, uint32_t Flags) { @@ -5059,8 +5059,8 @@ void OpDispatchBuilder::VPGATHER(OpcodeArgs) { const bool SupportsSVELoad = (VSIB.Scale == 1 || VSIB.Scale == IR::OpSizeToSize(AddrElementSize)) && (AddrElementSize == ElementLoadSize); - Ref Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags); - Ref Mask = LoadSource(FPRClass, Op, Op->Src[1], Op->Flags); + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Mask = LoadSourceFPR(Op, Op->Src[1], Op->Flags); Ref Result {}; if (!SupportsSVELoad) { @@ -5122,11 +5122,11 @@ void OpDispatchBuilder::VPGATHER(OpcodeArgs) { } } - StoreResult(FPRClass, Op, Result, OpSize::iInvalid); + StoreResultFPR(Op, Result, OpSize::iInvalid); ///< Assume non-faulting behaviour and clear the mask register. auto Zero = LoadZeroVector(Size); - StoreResult_WithOpSize(FPRClass, Op, Op->Src[1], Zero, Size, OpSize::iInvalid); + StoreResultFPR_WithOpSize(Op, Op->Src[1], Zero, Size, OpSize::iInvalid); } template void OpDispatchBuilder::VPGATHER(OpcodeArgs); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp index 1c134713b..8c5aea74c 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp @@ -28,7 +28,7 @@ class OrderedNode; Ref OpDispatchBuilder::GetX87Top() { // Yes, we are storing 3 bits in a single flag register. // Deal with it - return _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC); + return _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC); } void OpDispatchBuilder::SetX87FTW(Ref FTW) { @@ -52,18 +52,18 @@ void OpDispatchBuilder::SetX87FTW(Ref FTW) { FTW = _Orlshr(OpSize::i32Bit, FTW, FTW, 4); // ...and that's it. StoreContext implicitly does the final masking. - _StoreContext(OpSize::i8Bit, GPRClass, FTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); + _StoreContextGPR(OpSize::i8Bit, FTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); } void OpDispatchBuilder::SetX87Top(Ref Value) { - _StoreContext(OpSize::i8Bit, GPRClass, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC); + _StoreContextGPR(OpSize::i8Bit, Value, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC); } // Float LoaD operation with memory operand void OpDispatchBuilder::FLD(OpcodeArgs, IR::OpSize Width) { const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width; - Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags); + Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags); Ref ConvertedData = Data; // Convert to 80bit float if (Width == OpSize::i32Bit || Width == OpSize::i64Bit) { @@ -79,14 +79,14 @@ void OpDispatchBuilder::FLDFromStack(OpcodeArgs) { void OpDispatchBuilder::FBLD(OpcodeArgs) { // Read from memory - Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags); + Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags); Ref ConvertedData = _F80BCDLoad(Data); _PushStack(ConvertedData, Data, OpSize::i128Bit, true); } void OpDispatchBuilder::FBSTP(OpcodeArgs) { Ref converted = _F80BCDStore(_ReadStackValue(0)); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit); + StoreResultFPR_WithOpSize(Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit); _PopStackDestroy(); } @@ -99,7 +99,7 @@ void OpDispatchBuilder::FLD_Const(OpcodeArgs, NamedVectorConstant K) { void OpDispatchBuilder::FILD(OpcodeArgs) { const auto ReadWidth = OpSizeFromSrc(Op); // Read from memory - Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags); + Ref Data = LoadSourceGPR_WithOpSize(Op, Op->Src[0], ReadWidth, Op->Flags); // Sign extend to 64bits if (ReadWidth != OpSize::i64Bit) { @@ -180,7 +180,7 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) { Data = _F80CVTInt(Size, Data, Truncate); - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Data, Size, OpSize::i8Bit); + StoreResultGPR_WithOpSize(Op, Op->Dest, Data, Size, OpSize::i8Bit); if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { _PopStackDestroy(); @@ -206,10 +206,10 @@ void OpDispatchBuilder::FADD(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa // We have one memory argument Ref Arg {}; if (Integer) { - Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags); Arg = _F80CVTToInt(Arg, Width); } else { - Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Arg = _F80CVTTo(Arg, Width); } @@ -236,10 +236,10 @@ void OpDispatchBuilder::FMUL(OpcodeArgs, IR::OpSize Width, bool Integer, OpDispa // We have one memory argument Ref arg {}; if (Integer) { - arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags); arg = _F80CVTToInt(arg, Width); } else { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); arg = _F80CVTTo(arg, Width); } @@ -273,10 +273,10 @@ void OpDispatchBuilder::FDIV(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re // We have one memory argument Ref arg {}; if (Integer) { - arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags); arg = _F80CVTToInt(arg, Width); } else { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); arg = _F80CVTTo(arg, Width); } @@ -314,10 +314,10 @@ void OpDispatchBuilder::FSUB(OpcodeArgs, IR::OpSize Width, bool Integer, bool Re // We have one memory argument Ref Arg {}; if (Integer) { - Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags); Arg = _F80CVTToInt(Arg, Width); } else { - Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Arg = _F80CVTTo(Arg, Width); } @@ -340,7 +340,7 @@ Ref OpDispatchBuilder::GetX87FTW_Helper() { // bytes, we use the well-known bit twiddling algorithm: // // https://graphics.stanford.edu/~seander/bithacks.html#InterleaveBMN - Ref X = _LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); + Ref X = _LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); X = _Orlshl(OpSize::i32Bit, X, X, 4); X = _And(OpSize::i32Bit, X, Constant(0x0f0f0f0f)); X = _Orlshl(OpSize::i32Bit, X, X, 2); @@ -381,41 +381,41 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) { _SyncStackToSlow(); const auto Size = OpSizeFromSrc(Op); - Ref Mem = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.LoadData = false}); + Ref Mem = LoadSourceGPR(Op, Op->Dest, Op->Flags, {.LoadData = false}); Mem = AppendSegmentOffset(Mem, Op->Flags); { - auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW)); - _StoreMem(GPRClass, Size, Mem, FCW, Size); + auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW)); + _StoreMemGPR(Size, Mem, FCW, Size); } - { _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); } + { _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); } auto ZeroConst = Constant(0); { // FTW - _StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1); + _StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1); } { // Instruction Offset - _StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1); + _StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1); } { // Instruction CS selector (+ Opcode) - _StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1); + _StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1); } { // Data pointer offset - _StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1); + _StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1); } { // Data pointer selector - _StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1); + _StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1); } } @@ -441,20 +441,20 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) { _StackForceSlow(); const auto Size = OpSizeFromSrc(Op); - Ref Mem = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.LoadData = false}); + Ref Mem = LoadSourceGPR(Op, Op->Src[0], Op->Flags, {.LoadData = false}); Mem = AppendSegmentOffset(Mem, Op->Flags); - auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit); - _StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); + auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit); + _StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 1); - auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size); + auto NewFSW = _LoadMemGPR(Size, MemLocation, Size); ReconstructX87StateFromFSW_Helper(NewFSW); { // FTW Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 2); - SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size)); + SetX87FTW(_LoadMemGPR(Size, MemLocation, Size)); } } @@ -483,61 +483,61 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) { Ref Mem = MakeSegmentAddress(Op, Op->Dest); Ref Top = GetX87Top(); { - auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW)); - _StoreMem(GPRClass, Size, Mem, FCW, Size); + auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW)); + _StoreMemGPR(Size, Mem, FCW, Size); } - { _StoreMem(GPRClass, Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); } + { _StoreMemGPR(Size, ReconstructFSW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); } auto ZeroConst = Constant(0); { // FTW - _StoreMem(GPRClass, Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1); + _StoreMemGPR(Size, GetX87FTW_Helper(), Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1); } { // Instruction Offset - _StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1); + _StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 3), Size, MemOffsetType::SXTX, 1); } { // Instruction CS selector (+ Opcode) - _StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1); + _StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 4), Size, MemOffsetType::SXTX, 1); } { // Data pointer offset - _StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1); + _StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 5), Size, MemOffsetType::SXTX, 1); } { // Data pointer selector - _StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1); + _StoreMemGPR(Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MemOffsetType::SXTX, 1); } auto SevenConst = Constant(7); const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit; for (int i = 0; i < 7; ++i) { - Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass); + Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit)); if (ReducedPrecisionMode) { data = _F80CVTTo(data, OpSize::i64Bit); } - _StoreMem(FPRClass, OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1); + _StoreMemFPR(OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1); Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst); } // The final st(7) needs a bit of special handling here - Ref data = _LoadContextIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass); + Ref data = _LoadContextFPRIndexed(Top, LoadSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit)); if (ReducedPrecisionMode) { data = _F80CVTTo(data, OpSize::i64Bit); } // ST7 broken in to two parts // Lower 64bits [63:0] // upper 16 bits [79:64] - _StoreMem(FPRClass, OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MemOffsetType::SXTX, 1); + _StoreMemFPR(OpSize::i64Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10)), OpSize::i8Bit, MemOffsetType::SXTX, 1); auto topBytes = _VDupElement(OpSize::i128Bit, OpSize::i16Bit, data, 4); - _StoreMem(FPRClass, OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1); + _StoreMemFPR(OpSize::i16Bit, topBytes, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (7 * 10) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1); // reset to default FNINIT(Op); @@ -548,8 +548,8 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); Ref Mem = MakeSegmentAddress(Op, Op->Src[0]); - auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit); - _StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); + auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit); + _StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); if (ReducedPrecisionMode) { // ignore the rounding precision, we're always 64-bit in F64. // extract rounding mode @@ -561,11 +561,11 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) { _SetRoundingMode(roundingMode, false, roundingMode); } - auto NewFSW = _LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); + auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 1), Size, MemOffsetType::SXTX, 1); Ref Top = ReconstructX87StateFromFSW_Helper(NewFSW); { // FTW - SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1)); + SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1)); } auto SevenConst = Constant(7); @@ -574,14 +574,14 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) { Ref Mask = _VLoadTwoGPRs(low, high); const auto StoreSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit; for (int i = 0; i < 7; ++i) { - Ref Reg = _LoadMem(FPRClass, OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1); + Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1); // Mask off the top bits Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask); if (ReducedPrecisionMode) { // Convert to double precision Reg = _F80CVT(OpSize::i64Bit, Reg); } - _StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass); + _StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit)); Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst); } @@ -590,20 +590,19 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) { // ST7 broken in to two parts // Lower 64bits [63:0] // upper 16 bits [79:64] - Ref Reg = _LoadMem(FPRClass, OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MemOffsetType::SXTX, 1); - Ref RegHigh = - _LoadMem(FPRClass, OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1); + Ref Reg = _LoadMemFPR(OpSize::i64Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7)), OpSize::i8Bit, MemOffsetType::SXTX, 1); + Ref RegHigh = _LoadMemFPR(OpSize::i16Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * 7) + 8), OpSize::i8Bit, MemOffsetType::SXTX, 1); Reg = _VInsElement(OpSize::i128Bit, OpSize::i16Bit, 4, 0, Reg, RegHigh); if (ReducedPrecisionMode) { Reg = _F80CVT(OpSize::i64Bit, Reg); // Convert to double precision } - _StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass); + _StoreContextFPRIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit)); } // Load / Store Control Word void OpDispatchBuilder::X87FSTCW(OpcodeArgs) { - auto FCW = _LoadContext(OpSize::i16Bit, GPRClass, offsetof(FEXCore::Core::CPUState, FCW)); - StoreResult(GPRClass, Op, FCW, OpSize::iInvalid); + auto FCW = _LoadContextGPR(OpSize::i16Bit, offsetof(FEXCore::Core::CPUState, FCW)); + StoreResultGPR(Op, FCW, OpSize::iInvalid); } void OpDispatchBuilder::X87FLDCW(OpcodeArgs) { @@ -611,8 +610,8 @@ void OpDispatchBuilder::X87FLDCW(OpcodeArgs) { // to switch for now to slow mode whenever these are manually changed. // Remove the next line and try DF_04.asm in fast path. _StackForceSlow(); - Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); - _StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); + Ref NewFCW = LoadSourceGPR(Op, Op->Src[0], Op->Flags); + _StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); } void OpDispatchBuilder::FXCH(OpcodeArgs) { @@ -648,10 +647,10 @@ void OpDispatchBuilder::FCOMI(OpcodeArgs, IR::OpSize Width, bool Integer, OpDisp if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) { // Memory arg if (Integer) { - arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags); b = _F80CVTToInt(arg, Width); } else { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); b = _F80CVTTo(arg, Width); } } else { @@ -767,7 +766,7 @@ Ref OpDispatchBuilder::ReconstructFSW_Helper(Ref T) { void OpDispatchBuilder::X87FNSTSW(OpcodeArgs) { Ref TopValue = _SyncStackToSlow(); Ref StatusWord = ReconstructFSW_Helper(TopValue); - StoreResult(GPRClass, Op, StatusWord, OpSize::iInvalid); + StoreResultGPR(Op, StatusWord, OpSize::iInvalid); } void OpDispatchBuilder::FNCLEX(OpcodeArgs) { @@ -786,12 +785,12 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) { // Init FCW to 0x037F auto NewFCW = Constant(0x037F); - _StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); + _StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); // Set top to zero SetX87Top(Zero); // Tags all get marked as invalid - _StoreContext(OpSize::i8Bit, GPRClass, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); + _StoreContextGPR(OpSize::i8Bit, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); // Reinits the simulated stack _InitStack(); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87F64.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87F64.cpp index d105a6e6b..b3162eab6 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87F64.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87F64.cpp @@ -29,38 +29,38 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) { const auto Size = OpSizeFromSrc(Op); Ref Mem = MakeSegmentAddress(Op, Op->Src[0]); - auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit); + auto NewFCW = _LoadMemGPR(OpSize::i16Bit, Mem, OpSize::i16Bit); // ignore the rounding precision, we're always 64-bit in F64. // extract rounding mode Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW); _SetRoundingMode(roundingMode, false, roundingMode); - _StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); + _StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); - auto NewFSW = _LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MemOffsetType::SXTX, 1); + auto NewFSW = _LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size)), Size, MemOffsetType::SXTX, 1); ReconstructX87StateFromFSW_Helper(NewFSW); { // FTW - SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1)); + SetX87FTW(_LoadMemGPR(Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MemOffsetType::SXTX, 1)); } } void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) { _StackForceSlow(); - Ref NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + Ref NewFCW = LoadSourceGPR(Op, Op->Src[0], Op->Flags); // ignore the rounding precision, we're always 64-bit in F64. // extract rounding mode Ref roundingMode = _Bfe(OpSize::i32Bit, 3, 10, NewFCW); _SetRoundingMode(roundingMode, false, roundingMode); - _StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); + _StoreContextGPR(OpSize::i16Bit, NewFCW, offsetof(FEXCore::Core::CPUState, FCW)); } // F64 ops // Float load op with memory operand void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) { const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width; - Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], Width, Op->Flags); + Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags); // Convert to 64bit float Ref ConvertedData = Data; if (Width == OpSize::i32Bit) { @@ -73,7 +73,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) { void OpDispatchBuilder::FBLDF64(OpcodeArgs) { // Read from memory - Ref Data = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSize::f80Bit, Op->Flags); + Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags); Ref ConvertedData = _F80BCDLoad(Data); ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData); _PushStack(ConvertedData, Data, OpSize::i64Bit, true); @@ -82,7 +82,7 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) { void OpDispatchBuilder::FBSTPF64(OpcodeArgs) { Ref converted = _F80CVTTo(_ReadStackValue(0), OpSize::i64Bit); converted = _F80BCDStore(converted); - StoreResult_WithOpSize(FPRClass, Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit); + StoreResultFPR_WithOpSize(Op, Op->Dest, converted, OpSize::f80Bit, OpSize::i8Bit); _PopStackDestroy(); } @@ -95,7 +95,7 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) { const auto ReadWidth = OpSizeFromSrc(Op); // Read from memory - Ref Data = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], ReadWidth, Op->Flags); + Ref Data = LoadSourceGPR_WithOpSize(Op, Op->Src[0], ReadWidth, Op->Flags); if (ReadWidth == OpSize::i16Bit) { Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data); } @@ -112,7 +112,7 @@ void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) { } else { data = _Float_ToGPR_S(Size == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, OpSize::i64Bit, data); } - StoreResult_WithOpSize(GPRClass, Op, Op->Dest, data, Size, OpSize::i8Bit); + StoreResultGPR_WithOpSize(Op, Op->Dest, data, Size, OpSize::i8Bit); if ((Op->TableInfo->Flags & X86Tables::InstFlags::FLAGS_POP) != 0) { _PopStackDestroy(); @@ -138,16 +138,16 @@ void OpDispatchBuilder::FADDF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi Ref arg {}; if (Integer) { - arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags); if (Width == OpSize::i16Bit) { arg = _Sbfe(OpSize::i64Bit, 16, 0, arg); } arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg); } else if (Width == OpSize::i32Bit) { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg); } else if (Width == OpSize::i64Bit) { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); } else { FEX_UNREACHABLE; } @@ -176,16 +176,16 @@ void OpDispatchBuilder::FMULF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpDi Ref arg {}; if (Integer) { - arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags); if (Width == OpSize::i16Bit) { arg = _Sbfe(OpSize::i64Bit, 16, 0, arg); } arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg); } else if (Width == OpSize::i32Bit) { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg); } else if (Width == OpSize::i64Bit) { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); } else { FEX_UNREACHABLE; } @@ -228,16 +228,16 @@ void OpDispatchBuilder::FDIVF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) { if (Integer) { - Arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + Arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags); if (Width == OpSize::i16Bit) { Arg = _Sbfe(OpSize::i64Bit, 16, 0, Arg); } Arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, Arg); } else if (Width == OpSize::i32Bit) { - Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, Arg); } else if (Width == OpSize::i64Bit) { - Arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + Arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); } } else { FEX_UNREACHABLE; @@ -285,16 +285,16 @@ void OpDispatchBuilder::FSUBF64(OpcodeArgs, IR::OpSize Width, bool Integer, bool if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) { if (Integer) { - arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags); if (Width == OpSize::i16Bit) { arg = _Sbfe(OpSize::i64Bit, 16, 0, arg); } arg = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg); } else if (Width == OpSize::i32Bit) { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); arg = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg); } else if (Width == OpSize::i64Bit) { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); } } else { FEX_UNREACHABLE; @@ -332,16 +332,16 @@ void OpDispatchBuilder::FCOMIF64(OpcodeArgs, IR::OpSize Width, bool Integer, OpD } else if (Width == OpSize::i16Bit || Width == OpSize::i32Bit || Width == OpSize::i64Bit) { // Memory arg if (Integer) { - arg = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceGPR(Op, Op->Src[0], Op->Flags); if (Width == OpSize::i16Bit) { arg = _Sbfe(OpSize::i64Bit, 16, 0, arg); } b = _Float_FromGPR_S(OpSize::i64Bit, Width == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit, arg); } else if (Width == OpSize::i32Bit) { - arg = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + arg = LoadSourceFPR(Op, Op->Src[0], Op->Flags); b = _Float_FToF(OpSize::i64Bit, OpSize::i32Bit, arg); } else if (Width == OpSize::i64Bit) { - b = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags); + b = LoadSourceFPR(Op, Op->Src[0], Op->Flags); } } else { FEX_UNREACHABLE; diff --git a/FEXCore/Source/Interface/IR/IREmitter.h b/FEXCore/Source/Interface/IR/IREmitter.h index b9ae6b6fd..472253f32 100644 --- a/FEXCore/Source/Interface/IR/IREmitter.h +++ b/FEXCore/Source/Interface/IR/IREmitter.h @@ -180,6 +180,9 @@ public: return _StoreMem(FPRClass, Size, Value, Addr, Offset, Align, OffsetType, OffsetScale); } + IRPair _StoreMemPairGPR(OpSize Size, Ref Value1, Ref Value2, Ref Addr, uint32_t Offset) { + return _StoreMemPair(GPRClass, Size, Value1, Value2, Addr, Offset); + } IRPair _StoreMemPairFPR(OpSize Size, Ref Value1, Ref Value2, Ref Addr, uint32_t Offset) { return _StoreMemPair(FPRClass, Size, Value1, Value2, Addr, Offset); } diff --git a/FEXCore/Source/Interface/IR/Passes/x87StackOptimizationPass.cpp b/FEXCore/Source/Interface/IR/Passes/x87StackOptimizationPass.cpp index 4fa53060f..dc3e3c392 100644 --- a/FEXCore/Source/Interface/IR/Passes/x87StackOptimizationPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/x87StackOptimizationPass.cpp @@ -182,7 +182,7 @@ private: MemOffsetType OffsetType = Op->OffsetType; uint8_t OffsetScale = Op->OffsetScale; - IREmit->_StoreMem(FPRClass, OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale); + IREmit->_StoreMemFPR(OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale); auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1); // Store the Upper part of the register (the remaining 2 bytes) into memory. @@ -193,7 +193,7 @@ private: .Offset = 8, .AddrSize = OpSize::i64Bit}; A = SelectAddressMode(IREmit, A, GPROpSize, Features.SupportsTSOImm9, false, false, OpSize::i16Bit); - IREmit->_StoreMem(GPRClass, OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MemOffsetType::SXTX, A.IndexScale); + IREmit->_StoreMemGPR(OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MemOffsetType::SXTX, A.IndexScale); } void StoreStackMem_Helper(const IROp_StoreStackMem* Op, Ref StackNode) { @@ -211,7 +211,7 @@ private: if (!ReducedPrecisionMode || StrictReducedPrecisionMode) { StackNode = SilenceNaN(StackNode); } - IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale); + IREmit->_StoreMemFPR(Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale); break; } @@ -252,7 +252,7 @@ private: [[fallthrough]]; } case OpSize::i64Bit: { - IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale); + IREmit->_StoreMemFPR(Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale); break; } @@ -412,8 +412,7 @@ inline void X87StackOptimization::MigrateToSlowPathIf(bool ShouldMigrate) { inline Ref X87StackOptimization::GetTopWithCache_Slow() { if (!TopOffsetCache[0]) { - TopOffsetCache[0] = - IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC); + TopOffsetCache[0] = IREmit->_LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC); } return TopOffsetCache[0]; } @@ -458,7 +457,7 @@ inline void X87StackOptimization::SetTopWithCache_Slow(Ref Value) { inline Ref X87StackOptimization::GetFTW() { if (!FTWCached) { - FTWCached = IREmit->_LoadContext(OpSize::i8Bit, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); + FTWCached = IREmit->_LoadContextGPR(OpSize::i8Bit, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); } return FTWCached; } @@ -481,8 +480,7 @@ inline Ref X87StackOptimization::LoadStackValueAtOffset_Slow(uint8_t Offset) { OrderedNode* TopOffsetAddress = GetOffsetTopAddressWithCache_Slow(Offset); auto Size = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit; if (!TopValueCache[Offset]) { - TopValueCache[Offset] = - IREmit->_LoadMem(FPRClass, Size, TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1); + TopValueCache[Offset] = IREmit->_LoadMemFPR(Size, TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1); } return TopValueCache[Offset]; } @@ -616,7 +614,7 @@ inline void X87StackOptimization::UpdateTopForPush_Slow() { void X87StackOptimization::FlushCachedRegs() { if (FlushTopPending) { - IREmit->_StoreContext(OpSize::i8Bit, GPRClass, TopOffsetCache[0], offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC); + IREmit->_StoreContextGPR(OpSize::i8Bit, TopOffsetCache[0], offsetof(FEXCore::Core::CPUState, flags) + FEXCore::X86State::X87FLAG_TOP_LOC); FlushTopPending = false; } @@ -624,7 +622,7 @@ void X87StackOptimization::FlushCachedRegs() { for (size_t i = 0; i < FlushValuesPending.size(); i++) { if (FlushValuesPending[i]) { OrderedNode* TopOffsetAddress = GetOffsetTopAddressWithCache_Slow(i); - IREmit->_StoreMem(FPRClass, Size, TopValueCache[i], TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1); + IREmit->_StoreMemFPR(Size, TopValueCache[i], TopOffsetAddress, IREmit->_InlineConstant(MMBaseOffset()), Size, MemOffsetType::SXTX, 1); // store FlushValuesPending[i] = false; } @@ -675,7 +673,7 @@ void X87StackOptimization::FlushCachedRegs() { } }(); - IREmit->_StoreContext(OpSize::i8Bit, GPRClass, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); + IREmit->_StoreContextGPR(OpSize::i8Bit, NewFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW)); FTWCached = NewFTW; }