diff --git a/FEXCore/Source/Interface/Config/Config.cpp b/FEXCore/Source/Interface/Config/Config.cpp index 4686fe0a4..f93fe1128 100644 --- a/FEXCore/Source/Interface/Config/Config.cpp +++ b/FEXCore/Source/Interface/Config/Config.cpp @@ -339,8 +339,7 @@ namespace DefaultValues { #else constexpr uint32_t MaxCoreNumber = 1; #endif - constexpr uint32_t MinCoreNumber = 1; - if (Core > MaxCoreNumber || Core < MinCoreNumber) { + if (Core > MaxCoreNumber) { // Sanitize the core option by setting the core to the JIT if invalid FEXCore::Config::EraseSet(FEXCore::Config::CONFIG_CORE, fextl::fmt::format("{}", static_cast(FEXCore::Config::CONFIG_IRJIT))); } diff --git a/FEXCore/Source/Interface/Context/Context.h b/FEXCore/Source/Interface/Context/Context.h index f666b0a2c..148cc3a52 100644 --- a/FEXCore/Source/Interface/Context/Context.h +++ b/FEXCore/Source/Interface/Context/Context.h @@ -46,7 +46,6 @@ namespace CodeSerialize { namespace CPU { class Arm64JITCore; class X86JITCore; - class InterpreterCore; class Dispatcher; } namespace HLE { @@ -205,7 +204,6 @@ namespace FEXCore::Context { friend class FEXCore::CPU::X86JITCore; #endif - friend class FEXCore::CPU::InterpreterCore; friend class FEXCore::IR::Validation::IRValidation; struct { diff --git a/FEXCore/Source/Interface/Core/Core.cpp b/FEXCore/Source/Interface/Core/Core.cpp index 3383379b5..091455cb2 100644 --- a/FEXCore/Source/Interface/Core/Core.cpp +++ b/FEXCore/Source/Interface/Core/Core.cpp @@ -17,7 +17,6 @@ $end_info$ #include "Interface/Core/GdbServer.h" #include "Interface/Core/ObjectCache/ObjectCacheService.h" #include "Interface/Core/OpcodeDispatcher.h" -#include "Interface/Core/Interpreter/InterpreterCore.h" #include "Interface/Core/JIT/JITCore.h" #include "Interface/Core/Dispatcher/Dispatcher.h" #include "Interface/Core/X86Tables/X86Tables.h" diff --git a/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp b/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp index 8d052aa26..7413309cb 100644 --- a/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp +++ b/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp @@ -3,7 +3,6 @@ #include "Interface/Core/LookupCache.h" #include "Interface/Core/Dispatcher/Arm64Dispatcher.h" -#include "Interface/Core/Interpreter/InterpreterClass.h" #include "Interface/Context/Context.h" #include "Interface/Core/X86HelperGen.h" @@ -554,28 +553,6 @@ size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t Gues return UsedBytes; } -size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) { - LOGMAN_THROW_AA_FMT(!config.StaticRegisterAllocation, "GenerateInterpreterTrampoline dispatcher does not support SRA"); - - FEXCore::ARMEmitter::Emitter emit{CodeBuffer, MaxInterpreterTrampolineSize}; - ARMEmitter::ForwardLabel InlineIRData; - - emit.mov(ARMEmitter::XReg::x0, STATE); - emit.adr(ARMEmitter::Reg::r1, &InlineIRData); - - emit.ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter)); - emit.blr(ARMEmitter::Reg::r3); - - emit.ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop)); - emit.br(ARMEmitter::Reg::r0); - - emit.Bind(&InlineIRData); - - auto UsedBytes = emit.GetCursorOffset(); - emit.ClearICache(CodeBuffer, UsedBytes); - return UsedBytes; -} - void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) { // Setup dispatcher specific pointers that need to be accessed from JIT code { diff --git a/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.h b/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.h index a6301506a..e5653e5e9 100644 --- a/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.h +++ b/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.h @@ -22,7 +22,6 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter { Arm64Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config); void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override; size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override; - size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override; #ifdef VIXL_SIMULATOR void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) override; diff --git a/FEXCore/Source/Interface/Core/Dispatcher/Dispatcher.h b/FEXCore/Source/Interface/Core/Dispatcher/Dispatcher.h index fdc835fe4..d9c969d7c 100644 --- a/FEXCore/Source/Interface/Core/Dispatcher/Dispatcher.h +++ b/FEXCore/Source/Interface/Core/Dispatcher/Dispatcher.h @@ -61,10 +61,8 @@ public: // These are across all arches for now static constexpr size_t MaxGDBPauseCheckSize = 128; - static constexpr size_t MaxInterpreterTrampolineSize = 128; virtual size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) = 0; - virtual size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) = 0; static fextl::unique_ptr CreateX86(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config); static fextl::unique_ptr CreateArm64(FEXCore::Context::ContextImpl *CTX, const DispatcherConfig &Config); diff --git a/FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.cpp b/FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.cpp index a8d4f741a..3f4f5756c 100644 --- a/FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.cpp +++ b/FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.cpp @@ -4,7 +4,6 @@ #include "Interface/Core/Dispatcher/X86Dispatcher.h" -#include "Interface/Core/Interpreter/InterpreterClass.h" #include "Interface/Core/X86HelperGen.h" #include "Interface/Context/Context.h" @@ -381,29 +380,6 @@ size_t X86Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestR return emit.getSize(); } - -size_t X86Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) { - using namespace Xbyak; - using namespace Xbyak::util; - - Xbyak::CodeGenerator emit(1, &emit); // actual emit target set with setNewBuffer - emit.setNewBuffer(CodeBuffer, MaxInterpreterTrampolineSize); - - Label InlineIRData; - - emit.mov(rdi, STATE); - emit.lea(rsi, ptr[rip + InlineIRData]); - emit.call(qword STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter)); - - emit.jmp(qword STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop)); - - emit.L(InlineIRData); - - emit.ready(); - - return emit.getSize(); -} - X86Dispatcher::~X86Dispatcher() { FEXCore::Allocator::VirtualFree(top_, MAX_DISPATCHER_CODE_SIZE); } @@ -423,9 +399,6 @@ void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Threa Common.GuestSignal_SIGSEGV = GuestSignal_SIGSEGV; Common.SignalReturnHandler = SignalHandlerReturnAddress; Common.SignalReturnHandlerRT = SignalHandlerReturnAddressRT; - - auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter; - (uintptr_t&)Interpreter.CallbackReturn = IntCallbackReturnAddress; } } diff --git a/FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.h b/FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.h index 189889594..eb6ba2af1 100644 --- a/FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.h +++ b/FEXCore/Source/Interface/Core/Dispatcher/X86Dispatcher.h @@ -35,7 +35,6 @@ class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator { X86Dispatcher(FEXCore::Context::ContextImpl *ctx, const DispatcherConfig &config); void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override; size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override; - size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override; virtual ~X86Dispatcher() override; }; diff --git a/FEXCore/Source/Interface/Core/Interpreter/InterpreterClass.h b/FEXCore/Source/Interface/Core/Interpreter/InterpreterClass.h deleted file mode 100644 index a6b1ad0f7..000000000 --- a/FEXCore/Source/Interface/Core/Interpreter/InterpreterClass.h +++ /dev/null @@ -1,53 +0,0 @@ -// SPDX-License-Identifier: MIT -#pragma once - -#include "Interface/Core/InternalThreadState.h" -#include "Interface/Core/Dispatcher/Dispatcher.h" - -#include -#include -#include -#include -#include - -namespace FEXCore::CPU { -class Dispatcher; -class X86DispatchGenerator; -class Arm64DispatchGenerator; - -using DestMapType = fextl::vector; - -class InterpreterCore final : public CPUBackend { -public: - explicit InterpreterCore(Dispatcher *Dispatch, - FEXCore::Core::InternalThreadState *Thread); - - [[nodiscard]] fextl::string GetName() override { return "Interpreter"; } - - [[nodiscard]] CPUBackend::CompiledCode CompileCode(uint64_t Entry, - FEXCore::IR::IRListView const *IR, - FEXCore::Core::DebugData *DebugData, - FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override; - - [[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; } - - [[nodiscard]] bool NeedsOpDispatch() override { return true; } - - static void InitializeSignalHandlers(FEXCore::Context::ContextImpl *CTX); - - void ClearCache() override; - -private: - size_t BufferUsed; - Dispatcher *Dispatch; -}; - -template -T AtomicCompareAndSwap(T expected, T desired, T *addr); - -uint8_t AtomicFetchNeg(uint8_t *Addr); -uint16_t AtomicFetchNeg(uint16_t *Addr); -uint32_t AtomicFetchNeg(uint32_t *Addr); -uint64_t AtomicFetchNeg(uint64_t *Addr); - -} // namespace FEXCore::CPU diff --git a/FEXCore/Source/Interface/Core/Interpreter/InterpreterCore.cpp b/FEXCore/Source/Interface/Core/Interpreter/InterpreterCore.cpp deleted file mode 100644 index 941c464ef..000000000 --- a/FEXCore/Source/Interface/Core/Interpreter/InterpreterCore.cpp +++ /dev/null @@ -1,101 +0,0 @@ -// SPDX-License-Identifier: MIT -#include "Interface/Context/Context.h" - -#include "Interface/Core/Dispatcher/Dispatcher.h" -#include "Interface/Core/Interpreter/InterpreterClass.h" -#include -#include -#include -#include -#include -#include - -#include -#include -#include - -#include "InterpreterOps.h" - -#if defined(_M_X86_64) - #include "Interface/Core/Dispatcher/X86Dispatcher.h" -#elif defined(_M_ARM_64) - #include "Interface/Core/Dispatcher/Arm64Dispatcher.h" -#else - #error missing arch -#endif - -static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16; -static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128; - -namespace FEXCore::IR { - class IRListView; - class RegisterAllocationData; -} - - -namespace FEXCore::CPU { - -InterpreterCore::InterpreterCore(Dispatcher *Dispatcher, FEXCore::Core::InternalThreadState *Thread) - : CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE) - , Dispatch(Dispatcher) - { - - auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter; - - Interpreter.FragmentExecuter = reinterpret_cast(&InterpreterOps::InterpretIR); - - ClearCache(); -} - -CPUBackend::CompiledCode InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) { - - const auto IRSize = AlignUp(IR->GetInlineSize(), 16); - const auto MaxSize = IRSize + Dispatcher::MaxInterpreterTrampolineSize + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize; - - if ((BufferUsed + MaxSize) > CurrentCodeBuffer->Size) { - static_cast(ThreadState->CTX)->ClearCodeCache(ThreadState); - } - - CPUBackend::CompiledCode CodeData{}; - - const auto BufferStartOffset = BufferUsed; - CodeData.BlockBegin = CodeData.BlockEntry = CurrentCodeBuffer->Ptr + BufferStartOffset; - - auto DestBuffer = CodeData.BlockBegin; - - if (GDBEnabled) { - const auto GDBSize = Dispatch->GenerateGDBPauseCheck(DestBuffer, Entry); - DestBuffer += GDBSize; - BufferUsed += GDBSize; - } - - const auto TrampolineSize = Dispatch->GenerateInterpreterTrampoline(DestBuffer); - DestBuffer += TrampolineSize; - BufferUsed += TrampolineSize; - - - IR->Serialize(DestBuffer); - DestBuffer += IRSize; - BufferUsed += IRSize; - - CodeData.Size = BufferUsed - BufferStartOffset; - - return CodeData; -} - -void InterpreterCore::ClearCache() { - // Calling this one is needed to setup the initial CurrentCodeBuffer - [[maybe_unused]] auto CodeBuffer = GetEmptyCodeBuffer(); - BufferUsed = 0; -} - -fextl::unique_ptr CreateInterpreterCore(FEXCore::Context::ContextImpl *ctx, FEXCore::Core::InternalThreadState *Thread) { - return fextl::make_unique(ctx->Dispatcher.get(), Thread); -} - -CPUBackendFeatures GetInterpreterBackendFeatures() { - return CPUBackendFeatures { - .SupportsVTBL2 = true, - }; -} -} diff --git a/FEXCore/Source/Interface/Core/Interpreter/InterpreterCore.h b/FEXCore/Source/Interface/Core/Interpreter/InterpreterCore.h deleted file mode 100644 index 862966921..000000000 --- a/FEXCore/Source/Interface/Core/Interpreter/InterpreterCore.h +++ /dev/null @@ -1,24 +0,0 @@ -// SPDX-License-Identifier: MIT -#pragma once - -#include -#include - -namespace FEXCore::Context { -class ContextImpl; -} - -namespace FEXCore::Core { - struct InternalThreadState; -} - -namespace FEXCore::CPU { -class CPUBackend; -struct DispatcherConfig; - -[[nodiscard]] fextl::unique_ptr CreateInterpreterCore(FEXCore::Context::ContextImpl *ctx, - FEXCore::Core::InternalThreadState *Thread); -void InitializeInterpreterSignalHandlers(FEXCore::Context::ContextImpl *CTX); -CPUBackendFeatures GetInterpreterBackendFeatures(); - -} // namespace FEXCore::CPU diff --git a/FEXCore/Source/Interface/Core/Interpreter/InterpreterDefines.h b/FEXCore/Source/Interface/Core/Interpreter/InterpreterDefines.h deleted file mode 100644 index 9967bbe96..000000000 --- a/FEXCore/Source/Interface/Core/Interpreter/InterpreterDefines.h +++ /dev/null @@ -1,227 +0,0 @@ -// SPDX-License-Identifier: MIT -#pragma once - -#include -#include - -#define GD *GetDest(Data->SSAData, Node) -#define GDP GetDest(Data->SSAData, Node) - -#define DO_OP(size, type, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(GDP); \ - auto *Src1_d = reinterpret_cast(Src1); \ - auto *Src2_d = reinterpret_cast(Src2); \ - *Dst_d = func(*Src1_d, *Src2_d); \ - break; \ - } -#define DO_SCALAR_COMPARE_OP(size, type, type2, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src1_d = reinterpret_cast(Src1); \ - auto *Src2_d = reinterpret_cast(Src2); \ - Dst_d[0] = func(Src1_d[0], Src2_d[0]); \ - break; \ - } - -#define DO_VECTOR_COMPARE_OP(size, type, type2, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src1_d = reinterpret_cast(Src1); \ - auto *Src2_d = reinterpret_cast(Src2); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = func(Src1_d[i], Src2_d[i]); \ - } \ - break; \ - } -#define DO_VECTOR_OP(size, type, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src1_d = reinterpret_cast(Src1); \ - auto *Src2_d = reinterpret_cast(Src2); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = func(Src1_d[i], Src2_d[i]); \ - } \ - break; \ - } -#define DO_VECTOR_OP_WIDE(size, type, type2, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src1_d = reinterpret_cast(Src1); \ - auto *Src2_d = reinterpret_cast(Src2); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = func((type2)Src1_d[i], (type2)Src2_d[i]); \ - } \ - break; \ - } - -#define DO_VECTOR_PAIR_OP(size, type, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src1_d = reinterpret_cast(Src1); \ - auto *Src2_d = reinterpret_cast(Src2); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = func(Src1_d[i*2], Src1_d[i*2 + 1]); \ - Dst_d[i+Elements] = func(Src2_d[i*2], Src2_d[i*2 + 1]); \ - } \ - break; \ - } -#define DO_VECTOR_FCADD_PAIR_OP(size, type, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src1_d = reinterpret_cast(Src1); \ - auto *Src2_d = reinterpret_cast(Src2); \ - for (uint8_t i = 0; i < Elements; i += 2) { \ - func(&Dst_d[i], &Src1_d[i], &Src2_d[i]); \ - } \ - break; \ - } - -#define DO_VECTOR_SCALAR_OP(size, type, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src1_d = reinterpret_cast(Src1); \ - auto *Src2_d = reinterpret_cast(Src2); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = func(Src1_d[i], *Src2_d); \ - } \ - break; \ - } -#define DO_VECTOR_SCALAR_WIDE_OP(size, type, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src1_d = reinterpret_cast(Src1); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = func(Src1_d[i], Src2); \ - } \ - break; \ - } - -#define DO_VECTOR_0SRC_OP(size, type, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = func(); \ - } \ - break; \ - } -#define DO_VECTOR_1SRC_OP(size, type, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src_d = reinterpret_cast(Src); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = func(Src_d[i]); \ - } \ - break; \ - } -#define DO_VECTOR_REDUCE_1SRC_OP(size, type, func, start_val) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src_d = reinterpret_cast(Src); \ - type begin = start_val; \ - for (uint8_t i = 0; i < Elements; ++i) { \ - begin = func(begin, Src_d[i]); \ - } \ - Dst_d[0] = begin; \ - break; \ - } -#define DO_VECTOR_SAT_OP(size, type, func, min, max) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src1_d = reinterpret_cast(Src1); \ - auto *Src2_d = reinterpret_cast(Src2); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = func(Src1_d[i], Src2_d[i], min, max); \ - } \ - break; \ - } - -#define DO_VECTOR_1SRC_2TYPE_OP(size, type, type2, func, min, max) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src_d = reinterpret_cast(Src); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = (type)func(Src_d[i], min, max); \ - } \ - break; \ - } - -#define DO_VECTOR_1SRC_2TYPE_OP_NOSIZE(type, type2, func, min, max) \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src_d = reinterpret_cast(Src); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = (type)func(Src_d[i], min, max); \ - } -#define DO_VECTOR_1SRC_2TYPE_OP_TOP(size, type, type2, func, min, max) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src_d = reinterpret_cast(Src2); \ - memcpy(Dst_d, Src1, Elements * sizeof(type2)); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i+Elements] = (type)func(Src_d[i], min, max); \ - } \ - break; \ - } - -#define DO_VECTOR_1SRC_2TYPE_OP_TOP_SRC(size, type, type2, func, min, max) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src_d = reinterpret_cast(Src); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = (type)func(Src_d[i+Elements], min, max); \ - } \ - break; \ - } -#define DO_VECTOR_1SRC_2TYPE_OP_TOP_DST(size, type, type2, func, min, max) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src_d = reinterpret_cast(Src); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i+Elements] = (type)func(Src_d[i], min, max); \ - } \ - break; \ - } -#define DO_VECTOR_2SRC_2TYPE_OP(size, type, type2, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src1_d = reinterpret_cast(Src1); \ - auto *Src2_d = reinterpret_cast(Src2); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = (type)func((type)Src1_d[i], (type)Src2_d[i]); \ - } \ - break; \ - } -#define DO_VECTOR_2SRC_2TYPE_OP_TOP_SRC(size, type, type2, func) \ - case size: { \ - auto *Dst_d = reinterpret_cast(std::data(Tmp)); \ - auto *Src1_d = reinterpret_cast(Src1); \ - auto *Src2_d = reinterpret_cast(Src2); \ - for (uint8_t i = 0; i < Elements; ++i) { \ - Dst_d[i] = (type)func((type)Src1_d[i+Elements], (type)Src2_d[i+Elements]); \ - } \ - break; \ - } - -struct InterpVector256 { - __uint128_t Lower; - __uint128_t Upper; -}; - -template -Res GetDest(void* SSAData, FEXCore::IR::OrderedNodeWrapper Op) { - auto DstPtr = &reinterpret_cast(SSAData)[Op.ID().Value]; - return reinterpret_cast(DstPtr); -} - -template -Res GetDest(void* SSAData, FEXCore::IR::NodeID Op) { - auto DstPtr = &reinterpret_cast(SSAData)[Op.Value]; - return reinterpret_cast(DstPtr); -} - - -template -Res GetSrc(void* SSAData, FEXCore::IR::OrderedNodeWrapper Src) { - auto DstPtr = &reinterpret_cast(SSAData)[Src.ID().Value]; - return reinterpret_cast(DstPtr); -} diff --git a/FEXCore/Source/Interface/Core/Interpreter/InterpreterOps.h b/FEXCore/Source/Interface/Core/Interpreter/InterpreterOps.h index d0569c383..1c48e2189 100644 --- a/FEXCore/Source/Interface/Core/Interpreter/InterpreterOps.h +++ b/FEXCore/Source/Interface/Core/Interpreter/InterpreterOps.h @@ -9,19 +9,11 @@ #include #include -namespace FEXCore::Core { - struct InternalThreadState; -} - namespace FEXCore::IR { class IRListView; struct IROp_Header; } -namespace FEXCore::Core{ - struct DebugData; -} - namespace FEXCore::CPU { enum FallbackABI { FABI_UNKNOWN, @@ -52,412 +44,7 @@ namespace FEXCore::CPU { class InterpreterOps { public: - static void InterpretIR(FEXCore::Core::CpuStateFrame *Frame, FEXCore::IR::IRListView const *IR); static void FillFallbackIndexPointers(uint64_t *Info); static bool GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info); - - struct IROpData { - FEXCore::Core::InternalThreadState *State{}; - uint64_t CurrentEntry{}; - FEXCore::IR::IRListView const *CurrentIR{}; - volatile void *StackEntry{}; - void *SSAData{}; - struct { - bool Quit; - bool Redo; - } BlockResults{}; - - IR::NodeIterator BlockIterator{0, 0}; - }; - -#define DEF_OP(x) static void Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node) - - ///< Unhandled handler - DEF_OP(Unhandled); - - ///< No-op Handler - DEF_OP(NoOp); - - ///< ALU Ops - DEF_OP(TruncElementPair); - DEF_OP(Constant); - DEF_OP(EntrypointOffset); - DEF_OP(InlineConstant); - DEF_OP(InlineEntrypointOffset); - DEF_OP(CycleCounter); - DEF_OP(Add); - DEF_OP(AddNZCV); - DEF_OP(TestNZ); - DEF_OP(Sub); - DEF_OP(SubNZCV); - DEF_OP(Neg); - DEF_OP(Abs); - DEF_OP(Mul); - DEF_OP(UMul); - DEF_OP(Div); - DEF_OP(UDiv); - DEF_OP(Rem); - DEF_OP(URem); - DEF_OP(MulH); - DEF_OP(UMulH); - DEF_OP(Or); - DEF_OP(Orlshl); - DEF_OP(Orlshr); - DEF_OP(And); - DEF_OP(Andn); - DEF_OP(Xor); - DEF_OP(Lshl); - DEF_OP(Lshr); - DEF_OP(Ashr); - DEF_OP(Rol); - DEF_OP(Ror); - DEF_OP(Extr); - DEF_OP(PDep); - DEF_OP(PExt); - DEF_OP(LDiv); - DEF_OP(LUDiv); - DEF_OP(LRem); - DEF_OP(LURem); - DEF_OP(Zext); - DEF_OP(Not); - DEF_OP(Popcount); - DEF_OP(FindLSB); - DEF_OP(FindMSB); - DEF_OP(FindTrailingZeroes); - DEF_OP(CountLeadingZeroes); - DEF_OP(Rev); - DEF_OP(Bfi); - DEF_OP(Bfxil); - DEF_OP(Bfe); - DEF_OP(Sbfe); - DEF_OP(Select); - DEF_OP(VExtractToGPR); - DEF_OP(Float_ToGPR_ZU); - DEF_OP(Float_ToGPR_ZS); - DEF_OP(Float_ToGPR_S); - DEF_OP(FCmp); - - ///< Atomic ops - DEF_OP(CASPair); - DEF_OP(CAS); - DEF_OP(AtomicAdd); - DEF_OP(AtomicSub); - DEF_OP(AtomicAnd); - DEF_OP(AtomicOr); - DEF_OP(AtomicXor); - DEF_OP(AtomicSwap); - DEF_OP(AtomicFetchAdd); - DEF_OP(AtomicFetchSub); - DEF_OP(AtomicFetchAnd); - DEF_OP(AtomicFetchOr); - DEF_OP(AtomicFetchXor); - DEF_OP(AtomicFetchNeg); - DEF_OP(TelemetrySetValue); - - ///< Branch ops - DEF_OP(CallbackReturn); - DEF_OP(ExitFunction); - DEF_OP(Jump); - DEF_OP(CondJump); - DEF_OP(Syscall); - DEF_OP(InlineSyscall); - DEF_OP(Thunk); - DEF_OP(ValidateCode); - DEF_OP(ThreadRemoveCodeEntry); - DEF_OP(CPUID); - DEF_OP(XGETBV); - - ///< Conversion ops - DEF_OP(VInsGPR); - DEF_OP(VCastFromGPR); - DEF_OP(VDupFromGPR); - DEF_OP(Float_FromGPR_S); - DEF_OP(Float_FToF); - DEF_OP(Vector_SToF); - DEF_OP(Vector_FToZS); - DEF_OP(Vector_FToS); - DEF_OP(Vector_FToF); - DEF_OP(Vector_FToI); - - ///< Flag ops - DEF_OP(GetHostFlag); - - ///< Memory ops - DEF_OP(LoadContext); - DEF_OP(StoreContext); - DEF_OP(LoadRegister); - DEF_OP(StoreRegister); - DEF_OP(LoadContextIndexed); - DEF_OP(StoreContextIndexed); - DEF_OP(SpillRegister); - DEF_OP(FillRegister); - DEF_OP(LoadFlag); - DEF_OP(StoreFlag); - DEF_OP(LoadMem); - DEF_OP(StoreMem); - DEF_OP(VLoadVectorMasked); - DEF_OP(VStoreVectorMasked); - DEF_OP(VLoadVectorElement); - DEF_OP(VStoreVectorElement); - DEF_OP(VBroadcastFromMem); - DEF_OP(Push); - DEF_OP(MemSet); - DEF_OP(MemCpy); - DEF_OP(CacheLineClear); - DEF_OP(CacheLineClean); - DEF_OP(CacheLineZero); - - ///< Misc ops - DEF_OP(EndBlock); - DEF_OP(Fence); - DEF_OP(Break); - DEF_OP(Print); - DEF_OP(GetRoundingMode); - DEF_OP(SetRoundingMode); - DEF_OP(ProcessorID); - DEF_OP(RDRAND); - DEF_OP(Yield); - - ///< Move ops - DEF_OP(ExtractElementPair); - DEF_OP(CreateElementPair); - DEF_OP(Mov); - - ///< Vector ops - DEF_OP(VectorZero); - DEF_OP(VectorImm); - DEF_OP(LoadNamedVectorConstant); - DEF_OP(LoadNamedVectorIndexedConstant); - DEF_OP(VMov); - DEF_OP(VAnd); - DEF_OP(VBic); - DEF_OP(VOr); - DEF_OP(VXor); - DEF_OP(VAdd); - DEF_OP(VSub); - DEF_OP(VUQAdd); - DEF_OP(VUQSub); - DEF_OP(VSQAdd); - DEF_OP(VSQSub); - DEF_OP(VAddP); - DEF_OP(VAddV); - DEF_OP(VUMinV); - DEF_OP(VURAvg); - DEF_OP(VAbs); - DEF_OP(VPopcount); - DEF_OP(VFAdd); - DEF_OP(VFAddP); - DEF_OP(VFSub); - DEF_OP(VFMul); - DEF_OP(VFDiv); - DEF_OP(VFMin); - DEF_OP(VFMax); - DEF_OP(VFRecp); - DEF_OP(VFSqrt); - DEF_OP(VFRSqrt); - DEF_OP(VNeg); - DEF_OP(VFNeg); - DEF_OP(VNot); - DEF_OP(VUMin); - DEF_OP(VSMin); - DEF_OP(VUMax); - DEF_OP(VSMax); - DEF_OP(VZip); - DEF_OP(VUnZip); - DEF_OP(VTrn); - DEF_OP(VBSL); - DEF_OP(VCMPEQ); - DEF_OP(VCMPEQZ); - DEF_OP(VCMPGT); - DEF_OP(VCMPGTZ); - DEF_OP(VCMPLTZ); - DEF_OP(VFCMPEQ); - DEF_OP(VFCMPNEQ); - DEF_OP(VFCMPLT); - DEF_OP(VFCMPGT); - DEF_OP(VFCMPLE); - DEF_OP(VFCMPORD); - DEF_OP(VFCMPUNO); - DEF_OP(VUShl); - DEF_OP(VUShr); - DEF_OP(VSShr); - DEF_OP(VUShlS); - DEF_OP(VUShrS); - DEF_OP(VSShrS); - DEF_OP(VUShrSWide); - DEF_OP(VSShrSWide); - DEF_OP(VUShlSWide); - DEF_OP(VInsElement); - DEF_OP(VDupElement); - DEF_OP(VExtr); - DEF_OP(VUShrI); - DEF_OP(VSShrI); - DEF_OP(VShlI); - DEF_OP(VUShrNI); - DEF_OP(VUShrNI2); - DEF_OP(VSXTL); - DEF_OP(VSXTL2); - DEF_OP(VUXTL); - DEF_OP(VUXTL2); - DEF_OP(VSQXTN); - DEF_OP(VSQXTN2); - DEF_OP(VSQXTNPair); - DEF_OP(VSQXTUN); - DEF_OP(VSQXTUN2); - DEF_OP(VSQXTUNPair); - DEF_OP(VUMul); - DEF_OP(VUMull); - DEF_OP(VSMul); - DEF_OP(VSMull); - DEF_OP(VUMull2); - DEF_OP(VSMull2); - DEF_OP(VUMulH); - DEF_OP(VSMulH); - DEF_OP(VUABDL); - DEF_OP(VUABDL2); - DEF_OP(VTBL1); - DEF_OP(VTBL2); - DEF_OP(VRev32); - DEF_OP(VRev64); - DEF_OP(VPCMPESTRX); - DEF_OP(VPCMPISTRX); - DEF_OP(VFCADD); - - ///< Encryption ops - DEF_OP(AESImc); - DEF_OP(AESEnc); - DEF_OP(AESEncLast); - DEF_OP(AESDec); - DEF_OP(AESDecLast); - DEF_OP(AESKeyGenAssist); - DEF_OP(CRC32); - DEF_OP(PCLMUL); - - ///< F80 ops - DEF_OP(F80LOADFCW); - DEF_OP(F80ADD); - DEF_OP(F80SUB); - DEF_OP(F80MUL); - DEF_OP(F80DIV); - DEF_OP(F80FYL2X); - DEF_OP(F80ATAN); - DEF_OP(F80FPREM1); - DEF_OP(F80FPREM); - DEF_OP(F80SCALE); - DEF_OP(F80CVT); - DEF_OP(F80CVTINT); - DEF_OP(F80CVTTO); - DEF_OP(F80CVTTOINT); - DEF_OP(F80ROUND); - DEF_OP(F80F2XM1); - DEF_OP(F80TAN); - DEF_OP(F80SQRT); - DEF_OP(F80SIN); - DEF_OP(F80COS); - DEF_OP(F80XTRACT_EXP); - DEF_OP(F80XTRACT_SIG); - DEF_OP(F80CMP); - DEF_OP(F80BCDLOAD); - DEF_OP(F80BCDSTORE); - - //< F64 ops - DEF_OP(F64SIN); - DEF_OP(F64COS); - DEF_OP(F64TAN); - DEF_OP(F64F2XM1); - DEF_OP(F64ATAN); - DEF_OP(F64FPREM); - DEF_OP(F64FPREM1); - DEF_OP(F64FYL2X); - DEF_OP(F64SCALE); -#undef DEF_OP - template - [[nodiscard]] static bool IsConditionTrue(uint8_t Cond, uint64_t Src1, uint64_t Src2) { - bool CompResult = false; - if constexpr (sizeof(unsigned_type) == 16) { - LOGMAN_THROW_A_FMT(Cond != FEXCore::IR::COND_FLU && - Cond != FEXCore::IR::COND_FGE && - Cond != FEXCore::IR::COND_FLEU && - Cond != FEXCore::IR::COND_FGT && - Cond != FEXCore::IR::COND_FU && - Cond != FEXCore::IR::COND_FNU, "Unsupported comparison for 128-bit floats"); - } - - switch (Cond) { - case FEXCore::IR::COND_EQ: - CompResult = static_cast(Src1) == static_cast(Src2); - break; - case FEXCore::IR::COND_NEQ: - CompResult = static_cast(Src1) != static_cast(Src2); - break; - case FEXCore::IR::COND_SGE: - CompResult = static_cast(Src1) >= static_cast(Src2); - break; - case FEXCore::IR::COND_SLT: - CompResult = static_cast(Src1) < static_cast(Src2); - break; - case FEXCore::IR::COND_SGT: - CompResult = static_cast(Src1) > static_cast(Src2); - break; - case FEXCore::IR::COND_SLE: - CompResult = static_cast(Src1) <= static_cast(Src2); - break; - case FEXCore::IR::COND_UGE: - CompResult = static_cast(Src1) >= static_cast(Src2); - break; - case FEXCore::IR::COND_ULT: - CompResult = static_cast(Src1) < static_cast(Src2); - break; - case FEXCore::IR::COND_UGT: - CompResult = static_cast(Src1) > static_cast(Src2); - break; - case FEXCore::IR::COND_ULE: - CompResult = static_cast(Src1) <= static_cast(Src2); - break; - - case FEXCore::IR::COND_FLU: - CompResult = reinterpret_cast(Src1) < reinterpret_cast(Src2) || (std::isnan(reinterpret_cast(Src1)) || std::isnan(reinterpret_cast(Src2))); - break; - case FEXCore::IR::COND_FGE: - CompResult = reinterpret_cast(Src1) >= reinterpret_cast(Src2) && !(std::isnan(reinterpret_cast(Src1)) || std::isnan(reinterpret_cast(Src2))); - break; - case FEXCore::IR::COND_FLEU: - CompResult = reinterpret_cast(Src1) <= reinterpret_cast(Src2) || (std::isnan(reinterpret_cast(Src1)) || std::isnan(reinterpret_cast(Src2))); - break; - case FEXCore::IR::COND_FGT: - CompResult = reinterpret_cast(Src1) > reinterpret_cast(Src2) && !(std::isnan(reinterpret_cast(Src1)) || std::isnan(reinterpret_cast(Src2))); - break; - case FEXCore::IR::COND_FU: - CompResult = (std::isnan(reinterpret_cast(Src1)) || std::isnan(reinterpret_cast(Src2))); - break; - case FEXCore::IR::COND_FNU: - CompResult = !(std::isnan(reinterpret_cast(Src1)) || std::isnan(reinterpret_cast(Src2))); - break; - case FEXCore::IR::COND_MI: - case FEXCore::IR::COND_PL: - case FEXCore::IR::COND_VS: - case FEXCore::IR::COND_VC: - default: - LOGMAN_MSG_A_FMT("Unsupported compare type"); - break; - } - - return CompResult; - } - - static uint8_t GetOpSize(FEXCore::IR::IRListView const *CurrentIR, IR::OrderedNodeWrapper Node) { - auto IROp = CurrentIR->GetOp(Node); - return IROp->Size; - } - - // The maximum size a vector can be within FEX's interpreter. - // NOTE: If we ever support AVX-512, this should be changed - // to 64 bytes in size. - static constexpr size_t MaxInterpeterVectorSize = Core::CPUState::XMM_AVX_REG_SIZE; - - // Alias for specifying temporary data that is operated on - // before storing into a destination. - using TempVectorDataArray = std::array; - }; } // namespace FEXCore::CPU diff --git a/FEXCore/include/FEXCore/Core/CoreState.h b/FEXCore/include/FEXCore/Core/CoreState.h index fb2a11b73..42d8f1ebd 100644 --- a/FEXCore/include/FEXCore/Core/CoreState.h +++ b/FEXCore/include/FEXCore/Core/CoreState.h @@ -284,12 +284,6 @@ namespace FEXCore::Core { struct { // None so far } X86; - - struct { - uint64_t FragmentExecuter; - using IntCallbackReturn = void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP); - IntCallbackReturn CallbackReturn; - } Interpreter; }; }; diff --git a/FEXCore/include/FEXCore/Core/X86Enums.h b/FEXCore/include/FEXCore/Core/X86Enums.h index b32dbc355..5af85ac1e 100644 --- a/FEXCore/include/FEXCore/Core/X86Enums.h +++ b/FEXCore/include/FEXCore/Core/X86Enums.h @@ -75,7 +75,7 @@ enum X86RegLocation : uint32_t { RFLAG_VIP_LOC = 20, RFLAG_ID_LOC = 21, - // So we can implement arm64-like flag manipulaton on the interpreter/x86 jit.. + // So we can implement arm64-like flag manipulaton on the x86 jit.. // SF/ZF/CF/OF packed into a 32-bit word, matching arm64's NZCV structure (not semantics). RFLAG_NZCV_LOC = 24, RFLAG_NZCV_1_LOC = 25,