From 199cfd76d8a1c7c7ba7d72e8b633eabfe797ace4 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Wed, 9 Oct 2019 19:16:27 -0700 Subject: [PATCH] Refactor IR and other changes that are hard to split I had to change how blocks are represented to make it easier to parse This required a fairly substantial refactor that makes it so blocks are represented differently and we can walk them sequentially. This will make future analysis easier to deal with. Had to rewrite the passes and core's parsing of the IR afterwards. Moved RA in to a optimization pass to be shared between the JIT backends This works because x86-64 and AArch64 RA can be identical. Still doesn't support PHI nodes or spilling correctly, this is the first step in the process of getting there. --- Scripts/json_ir_generator.py | 50 +- Source/CMakeLists.txt | 4 +- Source/Interface/Context/Context.h | 18 +- Source/Interface/Core/Core.cpp | 279 +- .../Core/Interpreter/InterpreterCore.cpp | 31 +- Source/Interface/Core/JIT/Arm64/JIT.cpp | 541 +- Source/Interface/Core/JIT/x86_64/JIT.cpp | 5324 +++++++++-------- Source/Interface/Core/LLVMJIT/LLVMCore.cpp | 109 +- Source/Interface/Core/OpcodeDispatcher.cpp | 310 +- Source/Interface/Core/OpcodeDispatcher.h | 79 +- Source/Interface/Core/RegisterAllocation.cpp | 230 - Source/Interface/Core/RegisterAllocation.h | 30 - Source/Interface/HLE/Syscalls.cpp | 86 +- Source/Interface/IR/IR.cpp | 117 +- Source/Interface/IR/IR.json | 46 +- Source/Interface/IR/PassManager.cpp | 7 +- Source/Interface/IR/PassManager.h | 3 + Source/Interface/IR/Passes/ConstProp.cpp | 65 +- .../IR/Passes/DeadContextStoreElimination.cpp | 118 +- Source/Interface/IR/Passes/IRCompaction.cpp | 245 +- Source/Interface/IR/Passes/IRValidation.cpp | 26 +- .../RedundantFlagCalculationElimination.cpp | 79 +- .../IR/Passes/RegisterAllocationPass.cpp | 114 +- .../IR/Passes/RegisterAllocationPass.h | 10 +- .../IR/Passes/SyscallOptimization.cpp | 53 +- include/FEXCore/Core/CodeLoader.h | 3 + include/FEXCore/IR/IR.h | 129 +- include/FEXCore/IR/IntrusiveIRList.h | 17 +- 28 files changed, 4353 insertions(+), 3770 deletions(-) delete mode 100644 Source/Interface/Core/RegisterAllocation.cpp delete mode 100644 Source/Interface/Core/RegisterAllocation.h diff --git a/Scripts/json_ir_generator.py b/Scripts/json_ir_generator.py index 23cd2ee95..a85437107 100644 --- a/Scripts/json_ir_generator.py +++ b/Scripts/json_ir_generator.py @@ -36,7 +36,7 @@ def print_ir_structs(ops, defines): output_file.write("\ttemplate\n") output_file.write("\tT* CW() { return reinterpret_cast(Data); }\n") - output_file.write("\tNodeWrapper Args[0];\n") + output_file.write("\tOrderedNodeWrapper Args[0];\n") output_file.write("};\n\n"); @@ -48,6 +48,7 @@ def print_ir_structs(ops, defines): for op_key, op_vals in ops.items(): SSAArgs = 0 HasArgs = False + HasSSANames = False if ("SSAArgs" in op_vals): SSAArgs = int(op_vals["SSAArgs"]) @@ -55,16 +56,24 @@ def print_ir_structs(ops, defines): if ("Args" in op_vals and len(op_vals["Args"]) != 0): HasArgs = True + if ("SSANames" in op_vals and len(op_vals["SSANames"]) != 0): + HasSSANames = True + if (HasArgs or SSAArgs != 0): output_file.write("struct __attribute__((packed)) IROp_%s {\n" % op_key) output_file.write("\tIROp_Header Header;\n\n") # SSA arguments have a hard requirement to appear after the header if (SSAArgs != 0): - output_file.write("private:\n") - for i in range(0, SSAArgs): - output_file.write("\tuint64_t : (sizeof(NodeWrapper) * 8);\n"); - output_file.write("public:\n") + if (HasSSANames): + for i in range(0, SSAArgs): + output_file.write("\tOrderedNodeWrapper %s;\n" % (op_vals["SSANames"][i])); + + else: + output_file.write("private:\n") + for i in range(0, SSAArgs): + output_file.write("\tuint64_t : (sizeof(OrderedNodeWrapper) * 8);\n"); + output_file.write("public:\n") if (HasArgs): output_file.write("\t// User defined data\n") @@ -182,13 +191,32 @@ def print_ir_allocator_helpers(ops, defines): output_file.write("#ifdef IROP_ALLOCATE_HELPERS\n") output_file.write("\ttemplate \n") - output_file.write("\tusing IRPair = FEXCore::IR::Wrapper;\n\n") + output_file.write("\tstruct Wrapper final {\n") + output_file.write("\t\tT *first;\n") + output_file.write("\t\tOrderedNode *Node; ///< Actual offset of this IR in ths list\n") + output_file.write("\t\t\n") + output_file.write("\t\toperator Wrapper() const { return Wrapper {reinterpret_cast(first), Node}; }\n") + output_file.write("\t\toperator OrderedNode *() { return Node; }\n") + output_file.write("\t\toperator OpNodeWrapper () { return Node->Header.Value; }\n") + output_file.write("\t};\n") + + output_file.write("\ttemplate \n") + output_file.write("\tusing IRPair = Wrapper;\n\n") output_file.write("\tIRPair AllocateRawOp(size_t HeaderSize) {\n") output_file.write("\t\tauto Op = reinterpret_cast(Data.Allocate(HeaderSize));\n") output_file.write("\t\tmemset(Op, 0, HeaderSize);\n") output_file.write("\t\tOp->Op = IROps::OP_DUMMY;\n") - output_file.write("\t\treturn FEXCore::IR::Wrapper{Op, CreateNode(Op)};\n") + output_file.write("\t\treturn IRPair{Op, CreateNode(Op)};\n") + output_file.write("\t}\n\n") + + output_file.write("\ttemplate\n") + output_file.write("\tT *AllocateOrphanOp() {\n") + output_file.write("\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n") + output_file.write("\t\tauto Op = reinterpret_cast(Data.Allocate(Size));\n") + output_file.write("\t\tmemset(Op, 0, Size);\n") + output_file.write("\t\tOp->Header.Op = T2;\n") + output_file.write("\t\treturn Op;\n") output_file.write("\t}\n\n") output_file.write("\ttemplate\n") @@ -197,23 +225,23 @@ def print_ir_allocator_helpers(ops, defines): output_file.write("\t\tauto Op = reinterpret_cast(Data.Allocate(Size));\n") output_file.write("\t\tmemset(Op, 0, Size);\n") output_file.write("\t\tOp->Header.Op = T2;\n") - output_file.write("\t\treturn FEXCore::IR::Wrapper{Op, CreateNode(&Op->Header)};\n") + output_file.write("\t\treturn IRPair{Op, CreateNode(&Op->Header)};\n") output_file.write("\t}\n\n") output_file.write("\tuint8_t GetOpSize(OrderedNode *Op) const {\n") - output_file.write("\t\tauto HeaderOp = reinterpret_cast(Op->Header.Value.GetPtr(Data.Begin()));\n") + output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(Data.Begin());\n") output_file.write("\t\tLogMan::Throw::A(HeaderOp->HasDest, \"Op %s has no dest\\n\", GetName(HeaderOp->Op));\n") output_file.write("\t\treturn HeaderOp->Size;\n") output_file.write("\t}\n\n") output_file.write("\tuint8_t GetOpElements(OrderedNode *Op) const {\n") - output_file.write("\t\tauto HeaderOp = reinterpret_cast(Op->Header.Value.GetPtr(Data.Begin()));\n") + output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(Data.Begin());\n") output_file.write("\t\tLogMan::Throw::A(HeaderOp->HasDest, \"Op %s has no dest\\n\", GetName(HeaderOp->Op));\n") output_file.write("\t\treturn HeaderOp->Elements;\n") output_file.write("\t}\n\n") output_file.write("\tbool OpHasDest(OrderedNode *Op) const {\n") - output_file.write("\t\tauto HeaderOp = reinterpret_cast(Op->Header.Value.GetPtr(Data.Begin()));\n") + output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(Data.Begin());\n") output_file.write("\t\treturn HeaderOp->HasDest;\n") output_file.write("\t}\n\n") diff --git a/Source/CMakeLists.txt b/Source/CMakeLists.txt index ff5a263ed..b9058db6d 100644 --- a/Source/CMakeLists.txt +++ b/Source/CMakeLists.txt @@ -54,7 +54,6 @@ set (SRCS Interface/Core/CPUID.cpp Interface/Core/Frontend.cpp Interface/Core/OpcodeDispatcher.cpp - Interface/Core/RegisterAllocation.cpp Interface/Core/X86Tables.cpp Interface/Core/X86DebugInfo.cpp Interface/Core/Interpreter/InterpreterCore.cpp @@ -71,6 +70,7 @@ set (SRCS Interface/IR/Passes/IRCompaction.cpp Interface/IR/Passes/IRValidation.cpp Interface/IR/Passes/RedundantFlagCalculationElimination.cpp + Interface/IR/Passes/RegisterAllocationPass.cpp Interface/IR/Passes/SyscallOptimization.cpp ) @@ -108,7 +108,7 @@ add_custom_target(IR_INC add_library(${PROJECT_NAME} STATIC ${SRCS}) add_dependencies(${PROJECT_NAME} IR_INC) -target_link_libraries(${PROJECT_NAME} LLVM pthread rt ${JIT_LIBS}) +target_link_libraries(${PROJECT_NAME} LLVM pthread rt ${JIT_LIBS} dl) target_include_directories(${PROJECT_NAME} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}") diff --git a/Source/Interface/Context/Context.h b/Source/Interface/Context/Context.h index f8c14b3bd..85e6a9205 100644 --- a/Source/Interface/Context/Context.h +++ b/Source/Interface/Context/Context.h @@ -14,6 +14,13 @@ namespace FEXCore { class SyscallHandler; +namespace CPU { + class JITCore; +} +} + +namespace FEXCore::IR { + class RegisterAllocationPass; } namespace FEXCore::Context { @@ -24,6 +31,8 @@ namespace FEXCore::Context { struct Context { friend class FEXCore::SyscallHandler; + friend class FEXCore::CPU::JITCore; + struct { bool Multiblock {false}; bool BreakOnFrontendFailure {true}; @@ -78,6 +87,11 @@ namespace FEXCore::Context { FEXCore::Core::ThreadState *GetThreadState(); void LoadEntryList(); + uintptr_t CompileBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP); + uintptr_t CompileFallbackBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP); + protected: + IR::RegisterAllocationPass *GetRegisterAllocatorPass(); + private: void WaitForIdle(); FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID, uint64_t ChildTID); @@ -89,7 +103,6 @@ namespace FEXCore::Context { void ExecutionThread(FEXCore::Core::InternalThreadState *Thread); void RunThread(FEXCore::Core::InternalThreadState *Thread); - uintptr_t CompileBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP); uintptr_t AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr); FEXCore::CodeLoader *LocalLoader{}; @@ -99,5 +112,8 @@ namespace FEXCore::Context { void AddThreadRIPsToEntryList(FEXCore::Core::InternalThreadState *Thread); void SaveEntryList(); std::set EntryList; + std::vector InitLocations; + uint64_t StartingRIP; + IR::RegisterAllocationPass *RAPass {}; }; } diff --git a/Source/Interface/Core/Core.cpp b/Source/Interface/Core/Core.cpp index 373aa616d..ea9243bfc 100644 --- a/Source/Interface/Core/Core.cpp +++ b/Source/Interface/Core/Core.cpp @@ -8,6 +8,7 @@ #include "Interface/Core/Interpreter/InterpreterCore.h" #include "Interface/Core/JIT/JITCore.h" #include "Interface/Core/LLVMJIT/LLVMCore.h" +#include "Interface/IR/Passes/RegisterAllocationPass.h" #include #include @@ -20,7 +21,7 @@ constexpr uint64_t STACK_OFFSET = 0xc000'0000; constexpr uint64_t FS_OFFSET = 0xb000'0000; -constexpr uint64_t FS_SIZE = 0x1000; +constexpr uint64_t FS_SIZE = 0x1000'0000; namespace FEXCore::CPU { bool CreateCPUCore(FEXCore::Context::Context *CTX) { @@ -212,7 +213,7 @@ namespace FEXCore::Context { } memset(NewThreadState.flags, 0, 32); NewThreadState.gs = 0; - NewThreadState.fs = FS_OFFSET + FS_SIZE / 2; + NewThreadState.fs = FS_OFFSET; NewThreadState.flags[1] = 1; FEXCore::Core::InternalThreadState *Thread = CreateThread(&NewThreadState, 0, 0); @@ -220,17 +221,11 @@ namespace FEXCore::Context { // We are the parent thread ParentThread = Thread; - auto MemLayout = Loader->GetLayout(); - - uint64_t BasePtr = AlignDown(std::get<0>(MemLayout), PAGE_SIZE); - uint64_t BaseSize = AlignUp(std::get<2>(MemLayout), PAGE_SIZE); - - Thread->BlockCache->HintUsedRange(BasePtr, BaseSize); - - uintptr_t BaseRegion = reinterpret_cast(MapRegion(Thread, BasePtr, BaseSize, true)); + uintptr_t MemoryBase = MemoryMapper.GetBaseOffset(0); auto MemoryMapperFunction = [&](uint64_t Base, uint64_t Size) -> void* { - return MapRegion(Thread, Base, Size); + Thread->BlockCache->HintUsedRange(Base, Base); + return MapRegion(Thread, Base, Size, true); }; Loader->MapMemoryRegion(MemoryMapperFunction); @@ -244,15 +239,19 @@ namespace FEXCore::Context { // Now let the code loader setup memory auto MemoryWriterFunction = [&](void const *Data, uint64_t Addr, uint64_t Size) -> void { // Writes the machine code to be emulated in to memory - memcpy(reinterpret_cast(BaseRegion + Addr), Data, Size); + memcpy(reinterpret_cast(MemoryBase + Addr), Data, Size); }; Loader->LoadMemory(MemoryWriterFunction); + Loader->GetInitLocations(&InitLocations); - // Set the RIP to what the code loader wants - Thread->State.State.rip = Loader->DefaultRIP(); + auto TLSSlotWriter = [&](void const *Data, uint64_t Size) -> void { + memcpy(reinterpret_cast(MemoryBase + FS_OFFSET), Data, Size); + }; - LogMan::Msg::D("Memory Base: 0x%016lx", MemoryMapper.GetBaseOffset(0)); + // Offset next thread's FS_OFFSET by slot size + uint64_t SlotSize = Loader->InitializeThreadSlot(TLSSlotWriter); + StartingRIP = Loader->DefaultRIP(); InitializeThread(Thread); @@ -275,7 +274,7 @@ namespace FEXCore::Context { if (AllPaused) break; - PauseWait.WaitFor(std::chrono::seconds(1)); + PauseWait.WaitFor(std::chrono::milliseconds(10)); } while (true); } @@ -325,10 +324,11 @@ namespace FEXCore::Context { Thread->FallbackBackend->Initialize(); // Compile all of our cached entries - LogMan::Msg::D("Precompiling: %ld blocks", EntryList.size()); + LogMan::Msg::D("Precompiling: %ld blocks...", EntryList.size()); for (auto Entry : EntryList) { CompileRIP(Thread, Entry); } + LogMan::Msg::D("Done", EntryList.size()); // This will create the execution thread but it won't actually start executing Thread->ExecutionThread = std::thread(&Context::ExecutionThread, this, Thread); @@ -379,6 +379,15 @@ namespace FEXCore::Context { return Thread; } + IR::RegisterAllocationPass *Context::GetRegisterAllocatorPass() { + if (!RAPass) { + RAPass = new IR::RegisterAllocationPass(); + PassManager.InsertPass(RAPass); + } + + return RAPass; + } + uintptr_t Context::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) { auto BlockMapPtr = Thread->BlockCache->AddBlockMapping(Address, Ptr); if (BlockMapPtr == 0) { @@ -406,7 +415,6 @@ namespace FEXCore::Context { bool HadDispatchError {false}; [[maybe_unused]] bool HadRIPSetter {false}; - Thread->OpDispatcher->BeginBlock(); if (!FrontendDecoder.DecodeInstructionsInBlock(&GuestCode[TotalInstructionsLength], GuestRIP + TotalInstructionsLength)) { if (Config.BreakOnFrontendFailure) { LogMan::Msg::E("Had Frontend decoder error"); @@ -415,6 +423,7 @@ namespace FEXCore::Context { return 0; } + Thread->OpDispatcher->BeginFunction(GuestRIP); auto DecodedOps = FrontendDecoder.GetDecodedInsts(); for (size_t i = 0; i < DecodedOps.second; ++i) { FEXCore::X86Tables::X86InstInfo const* TableInfo {nullptr}; @@ -515,15 +524,17 @@ namespace FEXCore::Context { if (!Thread->OpDispatcher->Information.HadUnconditionalExit) { - Thread->OpDispatcher->EndBlock(TotalInstructionsLength); + Thread->OpDispatcher->CreateNewEndBlock(TotalInstructionsLength); + Thread->OpDispatcher->CreateNewBeginBlock(); Thread->OpDispatcher->ExitFunction(); + Thread->OpDispatcher->CreateNewEndBlock(0); } + Thread->OpDispatcher->Finalize(); // Run the passmanager over the IR from the dispatcher PassManager.Run(Thread->OpDispatcher.get()); - if (Thread->OpDispatcher->ShouldDump) -// if (GuestRIP == 0x48b680) + // if (Thread->OpDispatcher->ShouldDump) { std::stringstream out; auto NewIR = Thread->OpDispatcher->ViewIR(); @@ -531,17 +542,15 @@ namespace FEXCore::Context { printf("IR 0x%lx:\n%s\n@@@@@\n", GuestRIP, out.str().c_str()); } - // Do RA on the IR right now? - // Create a copy of the IR and place it in this thread's IR cache - auto IR = Thread->IRLists.try_emplace(GuestRIP, Thread->OpDispatcher->CreateIRCopy()); + auto AddedIR = Thread->IRLists.try_emplace(GuestRIP, Thread->OpDispatcher->CreateIRCopy()); Thread->OpDispatcher->ResetWorkingList(); - auto Debugit = Thread->DebugData.try_emplace(GuestRIP, FEXCore::Core::DebugData{}); + auto Debugit = Thread->DebugData.try_emplace(GuestRIP); Debugit.first->second.GuestCodeSize = TotalInstructionsLength; Debugit.first->second.GuestInstructionCount = TotalInstructions; - IRList = IR.first->second.get(); + IRList = AddedIR.first->second.get(); DebugData = &Debugit.first->second; Thread->Stats.BlocksCompiled.fetch_add(1); } @@ -560,6 +569,20 @@ namespace FEXCore::Context { return 0; } + using BlockFn = void (*)(FEXCore::Core::InternalThreadState *Thread); + uintptr_t Context::CompileFallbackBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) { + // We have ONE more chance to try and fallback to the fallback CPU backend + // This will most likely fail since regular code use won't be using a fallback core. + // It's mainly for testing new instruction encodings + void *CodePtr = Thread->FallbackBackend->CompileCode(nullptr, nullptr); + if (CodePtr) { + uintptr_t Ptr = reinterpret_cast(AddBlockMapping(Thread, GuestRIP, CodePtr)); + return Ptr; + } + + return 0; + } + void Context::ExecutionThread(FEXCore::Core::InternalThreadState *Thread) { Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING; @@ -579,119 +602,143 @@ namespace FEXCore::Context { Thread->State.RunningEvents.Running = true; Thread->State.RunningEvents.ShouldPause = false; constexpr uint32_t CoreDebugLevel = 0; - while (!ShouldStop.load() && !Thread->State.RunningEvents.ShouldStop.load()) { - uint64_t GuestRIP = Thread->State.State.rip; - if (CoreDebugLevel >= 1) { - char const *Name = LocalLoader->FindSymbolNameInRange(GuestRIP); - LogMan::Msg::D(">>>>RIP: 0x%lx: '%s'", GuestRIP, Name ? Name : ""); - } + bool Initializing = false; - using BlockFn = void (*)(FEXCore::Core::InternalThreadState *Thread); + uint64_t InitializationStep = 0; + if (Initializing) { + Thread->State.State.rip = ~0ULL; + } + else { + Thread->State.State.rip = StartingRIP; + } - if (!Thread->CPUBackend->NeedsOpDispatch()) { - BlockFn Ptr = reinterpret_cast(Thread->CPUBackend->CompileCode(nullptr, nullptr)); - Ptr(Thread); - } - else { - // Do have have this block compiled? - auto it = Thread->BlockCache->FindBlock(GuestRIP); - if (it == 0) { - // If not compile it - it = CompileBlock(Thread, GuestRIP); + if (Thread->CPUBackend->HasCustomDispatch()) { + Thread->CPUBackend->ExecuteCustomDispatch(&Thread->State); + } + else { + while (!ShouldStop.load() && !Thread->State.RunningEvents.ShouldStop.load()) { + if (Initializing) { + if (Thread->State.State.rip == ~0ULL) { + if (InitializationStep < InitLocations.size()) { + Thread->State.State.gregs[X86State::REG_RSP] -= 8; + *MemoryMapper.GetPointer(Thread->State.State.gregs[X86State::REG_RSP]) = ~0ULL; + LogMan::Msg::D("Going down init path: 0x%lx", InitLocations[InitializationStep]); + Thread->State.State.rip = InitLocations[InitializationStep++]; + } + else { + Initializing = false; + Thread->State.State.rip = StartingRIP; + } + } + } + uint64_t GuestRIP = Thread->State.State.rip; + + if (CoreDebugLevel >= 1) { + char const *Name = LocalLoader->FindSymbolNameInRange(GuestRIP); + LogMan::Msg::D(">>>>RIP: 0x%lx: '%s'", GuestRIP, Name ? Name : ""); } - // Did we successfully compile this block? - if (it != 0) { - // Block is compiled, run it - BlockFn Ptr = reinterpret_cast(it); + if (!Thread->CPUBackend->NeedsOpDispatch()) { + BlockFn Ptr = reinterpret_cast(Thread->CPUBackend->CompileCode(nullptr, nullptr)); Ptr(Thread); } else { - // We have ONE more chance to try and fallback to the fallback CPU backend - // This will most likely fail since regular code use won't be using a fallback core. - // It's mainly for testing new instruction encodings - void *CodePtr = Thread->FallbackBackend->CompileCode(nullptr, nullptr); - if (CodePtr) { - BlockFn Ptr = reinterpret_cast(AddBlockMapping(Thread, GuestRIP, CodePtr)); - Ptr(Thread); + // Do have have this block compiled? + auto it = Thread->BlockCache->FindBlock(GuestRIP); + if (it == 0) { + // If not compile it + it = CompileBlock(Thread, GuestRIP); + } + + // Did we successfully compile this block? + if (it != 0) { + // Block is compiled, run it + BlockFn Ptr = reinterpret_cast(it); + Ptr(Thread); } else { - // Let the frontend know that something has happened that is unhandled - Thread->State.RunningEvents.ShouldPause = true; - Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_UNKNOWNERROR; + // We have ONE more chance to try and fallback to the fallback CPU backend + // This will most likely fail since regular code use won't be using a fallback core. + // It's mainly for testing new instruction encodings + uintptr_t CodePtr = CompileFallbackBlock(Thread, GuestRIP); + if (CodePtr) { + BlockFn Ptr = reinterpret_cast(CodePtr); + Ptr(Thread); + } + else { + // Let the frontend know that something has happened that is unhandled + Thread->State.RunningEvents.ShouldPause = true; + Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_UNKNOWNERROR; + } } } - } -// if (GuestRIP == 0x48c8dd) { -// fflush(stdout); -// __builtin_trap(); -// } - if (CoreDebugLevel >= 2) { - int i = 0; - LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); - i += 4; - LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); - i += 4; - LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); - i += 4; - LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); - uint64_t PackedFlags{}; - for (unsigned i = 0; i < 32; ++i) { - PackedFlags |= static_cast(Thread->State.State.flags[i]) << i; + if (CoreDebugLevel >= 2) { + int i = 0; + LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); + i += 4; + LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); + i += 4; + LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); + i += 4; + LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); + uint64_t PackedFlags{}; + for (i = 0; i < 32; ++i) { + PackedFlags |= static_cast(Thread->State.State.flags[i]) << i; + } + LogMan::Msg::D("\tFlags: %016lx", PackedFlags); } - LogMan::Msg::D("\tFlags: %016lx", PackedFlags); - } - if (CoreDebugLevel >= 3) { - int i = 0; - LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); - LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); + if (CoreDebugLevel >= 3) { + int i = 0; + LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); + LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); - i += 4; - LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); - LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); - i += 4; - LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); - LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); - i += 4; - LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); - LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); - uint64_t PackedFlags{}; - for (unsigned i = 0; i < 32; ++i) { - PackedFlags |= static_cast(Thread->State.State.flags[i]) << i; + i += 4; + LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); + LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); + i += 4; + LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); + LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); + i += 4; + LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); + LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); + uint64_t PackedFlags{}; + for (i = 0; i < 32; ++i) { + PackedFlags |= static_cast(Thread->State.State.flags[i]) << i; + } + LogMan::Msg::D("\tFlags: %016lx", PackedFlags); } - LogMan::Msg::D("\tFlags: %016lx", PackedFlags); - } - if (Thread->State.RunningEvents.ShouldStop.load()) { - // If it is the parent thread that died then just leave - // XXX: This doesn't make sense when the parent thread doesn't outlive its children - if (Thread->State.ThreadManager.GetTID() == 1) { - ShouldStop = true; - Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_SHUTDOWN; + if (Thread->State.RunningEvents.ShouldStop.load()) { + // If it is the parent thread that died then just leave + // XXX: This doesn't make sense when the parent thread doesn't outlive its children + if (Thread->State.ThreadManager.GetTID() == 1) { + ShouldStop = true; + Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_SHUTDOWN; + } + break; } - break; - } - if (RunningMode == FEXCore::Context::CoreRunningMode::MODE_SINGLESTEP || Thread->State.RunningEvents.ShouldPause) { - Thread->State.RunningEvents.Running = false; - Thread->State.RunningEvents.WaitingToStart = false; + if (RunningMode == FEXCore::Context::CoreRunningMode::MODE_SINGLESTEP || Thread->State.RunningEvents.ShouldPause) { + Thread->State.RunningEvents.Running = false; + Thread->State.RunningEvents.WaitingToStart = false; - // If something previously hasn't set the exit state then set it now - if (Thread->ExitReason == FEXCore::Context::ExitReason::EXIT_NONE) - Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_DEBUG; + // If something previously hasn't set the exit state then set it now + if (Thread->ExitReason == FEXCore::Context::ExitReason::EXIT_NONE) + Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_DEBUG; - PauseWait.NotifyAll(); - Thread->StartRunning.Wait(); + PauseWait.NotifyAll(); + Thread->StartRunning.Wait(); - // If we set it to debug then set it back to none after this - // We want to retain the state if the frontend decides to leave - if (Thread->ExitReason == FEXCore::Context::ExitReason::EXIT_DEBUG) - Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_NONE; + // If we set it to debug then set it back to none after this + // We want to retain the state if the frontend decides to leave + if (Thread->ExitReason == FEXCore::Context::ExitReason::EXIT_DEBUG) + Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_NONE; - Thread->State.RunningEvents.Running = true; + Thread->State.RunningEvents.Running = true; + } } } @@ -731,7 +778,7 @@ namespace FEXCore::Context { return MemoryMapper.GetMemoryBase(); } - void Context::CopyMemoryMapping([[maybe_unused]] FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread) { + void Context::CopyMemoryMapping([[maybe_unused]] FEXCore::Core::InternalThreadState*, FEXCore::Core::InternalThreadState *ChildThread) { auto Regions = MemoryMapper.MappedRegions; for (auto const& Region : Regions) { ChildThread->CPUBackend->MapRegion(Region.Ptr, Region.Offset, Region.Size); diff --git a/Source/Interface/Core/Interpreter/InterpreterCore.cpp b/Source/Interface/Core/Interpreter/InterpreterCore.cpp index a4eab635e..527ecf500 100644 --- a/Source/Interface/Core/Interpreter/InterpreterCore.cpp +++ b/Source/Interface/Core/Interpreter/InterpreterCore.cpp @@ -39,10 +39,10 @@ private: void *AllocateTmpSpace(size_t Size); template - Res GetDest(IR::NodeWrapper Op); + Res GetDest(IR::OrderedNodeWrapper Op); template - Res GetSrc(IR::NodeWrapper Src); + Res GetSrc(IR::OrderedNodeWrapper Src); std::vector TmpSpace; DestMapType DestMap; @@ -86,13 +86,13 @@ void *InterpreterCore::AllocateTmpSpace(size_t Size) { } template -Res InterpreterCore::GetDest(IR::NodeWrapper Op) { +Res InterpreterCore::GetDest(IR::OrderedNodeWrapper Op) { auto DstPtr = DestMap[Op.NodeOffset]; return reinterpret_cast(DstPtr); } template -Res InterpreterCore::GetSrc(IR::NodeWrapper Src) { +Res InterpreterCore::GetSrc(IR::OrderedNodeWrapper Src) { #if DESTMAP_AS_MAP LogMan::Throw::A(DestMap.find(Src.NodeOffset) != DestMap.end(), "Op had source but it wasn't in the destination map"); #endif @@ -138,8 +138,8 @@ void InterpreterCore::ExecuteCode(FEXCore::Core::InternalThreadState *Thread) { using namespace FEXCore::IR; using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); uint8_t OpSize = IROp->Size; @@ -287,15 +287,16 @@ void InterpreterCore::ExecuteCode(FEXCore::Core::InternalThreadState *Thread) { case IR::OP_STOREMEM: { #define STORE_DATA(x, y) \ case x: { \ - y *Data = Thread->CTX->MemoryMapper.GetBaseOffset(*GetSrc(Op->Header.Args[0])); \ + uint64_t SrcPtr = *GetSrc(Op->Header.Args[0]); \ + y *Data = Thread->CTX->MemoryMapper.GetPointer(SrcPtr); \ LogMan::Throw::A(Data != nullptr, "Couldn't Map pointer to 0x%lx for size %d store\n", *GetSrc(Op->Header.Args[0]), x);\ - *Data = *GetSrc(Op->Header.Args[1]); \ + memcpy(Data, GetSrc(Op->Header.Args[1]), sizeof(y)); \ } \ break auto Op = IROp->C(); - //LogMan::Msg::D("Storing guestmem: 0x%lx (%d)", *GetSrc(Op->Header.Args[0]), Op->Size); - //LogMan::Msg::D("\tStoring: 0x%016lx", (uint64_t)*GetSrc(Op->Header.Args[1])); + // LogMan::Msg::D("Storing guestmem: 0x%lx (%d)", *GetSrc(Op->Header.Args[0]), Op->Size); + // LogMan::Msg::D("\tStoring: 0x%016lx", (uint64_t)*GetSrc(Op->Header.Args[1])); switch (Op->Size) { STORE_DATA(1, uint8_t); @@ -1449,6 +1450,16 @@ void InterpreterCore::ExecuteCode(FEXCore::Core::InternalThreadState *Thread) { default: LogMan::Msg::A("Unknown LREM Size: %d", Size); break; } break; + } + case IR::OP_VEXTR: { + auto Op = IROp->C(); + __uint128_t Src1 = *GetSrc<__uint128_t*>(Op->Header.Args[0]); + __uint128_t Src2 = *GetSrc<__uint128_t*>(Op->Header.Args[1]); + + uint8_t Offset = Op->Index * 8; + __uint128_t Dst = (Src1 << (sizeof(__uint128_t) - Offset)) | (Src2 >> Offset); + memcpy(GDP, &Dst, 16); + break; } default: diff --git a/Source/Interface/Core/JIT/Arm64/JIT.cpp b/Source/Interface/Core/JIT/Arm64/JIT.cpp index b691281a0..0e03165d3 100644 --- a/Source/Interface/Core/JIT/Arm64/JIT.cpp +++ b/Source/Interface/Core/JIT/Arm64/JIT.cpp @@ -1,10 +1,12 @@ #include "Interface/Context/Context.h" -#include "Interface/Core/RegisterAllocation.h" + +#include "Interface/Core/BlockCache.h" #include "Interface/Core/InternalThreadState.h" -#include "Interface/Core/JIT/x86_64/JIT.h" #include "Interface/HLE/Syscalls.h" +#include "Interface/IR/Passes/RegisterAllocationPass.h" + #if _M_X86_64 #define VIXL_INCLUDE_SIMULATOR_AARCH64 #include "aarch64/simulator-aarch64.h" @@ -22,17 +24,17 @@ namespace FEXCore::CPU { using namespace vixl; using namespace vixl::aarch64; -#define STATE x0 -#define MEM_BASE x1 -#define TMP1 x2 -#define TMP2 x3 + +#define MEM_BASE x28 +#define STATE x27 +#define TMP1 x1 +#define TMP2 x2 #define VTMP1 v1 #define VTMP2 v2 #define VTMP3 v3 -static uint64_t SyscallThunk(FEXCore::Core::InternalThreadState *Thread, FEXCore::SyscallHandler *Handler, FEXCore::HLE::SyscallArguments *Args) -{ +static uint64_t SyscallThunk(FEXCore::SyscallHandler *Handler, FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args) { return Handler->HandleSyscall(Thread, Args); } @@ -41,6 +43,16 @@ static void CPUIDThunk(FEXCore::CPUIDEmu *CPUID, uint64_t Function, FEXCore::CPU memcpy(Results, &Res, sizeof(FEXCore::CPUIDEmu::FunctionResults)); } +static uint64_t CompileBlockThunk(FEXCore::Context::Context* CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) { + uint64_t Result = CTX->CompileBlock(Thread, RIP); + return Result; +} + +static uint64_t CompileFallbackBlockThunk(FEXCore::Context::Context* CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) { + uint64_t Result = CTX->CompileFallbackBlock(Thread, RIP); + return Result; +} + // XXX: Switch from MacroAssembler to Assembler once we drop the simulator class JITCore final : public CPUBackend, public vixl::aarch64::MacroAssembler { public: @@ -57,12 +69,22 @@ public: void SimulationExecution(FEXCore::Core::InternalThreadState *Thread); #endif + bool HasCustomDispatch() const override { return CustomDispatchGenerated; } + +#if _M_X86_64 + void ExecuteCustomDispatch(FEXCore::Core::ThreadState *Thread) override; +#else + void ExecuteCustomDispatch(FEXCore::Core::ThreadState *Thread) override { + DispatchPtr(reinterpret_cast(Thread->InternalState)); + } +#endif + private: FEXCore::Context::Context *CTX; FEXCore::Core::InternalThreadState *State; FEXCore::IR::IRListView const *CurrentIR; - std::unordered_map JumpTargets; + std::map JumpTargets; /** * @name Register Allocation @@ -73,21 +95,19 @@ private: constexpr static uint32_t RegisterClasses = 2; constexpr static uint32_t GPRBase = 0; - constexpr static uint32_t GPRClass = 0; + constexpr static uint32_t GPRClass = IR::RegisterAllocationPass::GPRClass; constexpr static uint32_t FPRBase = NumGPRs; - constexpr static uint32_t FPRClass = 1; + constexpr static uint32_t FPRClass = IR::RegisterAllocationPass::FPRClass; - RA::RegisterSet *RASet; + IR::RegisterAllocationPass::RegisterSet *RASet; /** @} */ - void FindNodeClasses(); - bool CalculateLiveRange(uint32_t Nodes); constexpr static uint8_t RA_32 = 0; constexpr static uint8_t RA_64 = 1; constexpr static uint8_t RA_FPR = 2; bool HasRA = false; - RA::RegisterGraph *Graph; + IR::RegisterAllocationPass::RegisterGraph *Graph; uint32_t GetPhys(uint32_t Node); template @@ -118,9 +138,27 @@ private: std::unordered_map> HostToGuest; #endif void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant); + + void CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread); + bool CustomDispatchGenerated {false}; + using CustomDispatch = void(*)(FEXCore::Core::InternalThreadState *Thread); + CustomDispatch DispatchPtr{}; + IR::RegisterAllocationPass *RAPass; + +#if _M_X86_64 + uint64_t CustomDispatchEnd; +#endif }; #if _M_X86_64 +void JITCore::ExecuteCustomDispatch(FEXCore::Core::ThreadState *Thread) { + PrintDisassembler PrintDisasm(stdout); + PrintDisasm.DisassembleBuffer(vixl::aarch64::Instruction::Cast(DispatchPtr), vixl::aarch64::Instruction::Cast(CustomDispatchEnd)); + + Sim.WriteXRegister(0, reinterpret_cast(Thread)); + Sim.RunFrom(vixl::aarch64::Instruction::Cast(DispatchPtr)); +} + static void SimulatorExecution(FEXCore::Core::InternalThreadState *Thread) { JITCore *Core = reinterpret_cast(Thread->CPUBackend.get()); Core->SimulationExecution(Thread); @@ -129,8 +167,8 @@ static void SimulatorExecution(FEXCore::Core::InternalThreadState *Thread) { void JITCore::SimulationExecution(FEXCore::Core::InternalThreadState *Thread) { using namespace vixl::aarch64; auto SimulatorAddress = HostToGuest[Thread->State.State.rip]; - //PrintDisassembler PrintDisasm(stdout); - //PrintDisasm.DisassembleBuffer(vixl::aarch64::Instruction::Cast(SimulatorAddress.first), vixl::aarch64::Instruction::Cast(SimulatorAddress.second)); + // PrintDisassembler PrintDisasm(stdout); + // PrintDisasm.DisassembleBuffer(vixl::aarch64::Instruction::Cast(SimulatorAddress.first), vixl::aarch64::Instruction::Cast(SimulatorAddress.second)); Sim.WriteXRegister(0, reinterpret_cast(Thread)); Sim.RunFrom(vixl::aarch64::Instruction::Cast(SimulatorAddress.first)); @@ -149,11 +187,14 @@ JITCore::JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadSt // XXX: Set this to a real minimum feature set in the future SetCPUFeatures(vixl::CPUFeatures::All()); - RASet = RA::AllocateRegisterSet(RegisterCount, RegisterClasses); - RA::AddRegisters(RASet, GPRClass, GPRBase, NumGPRs); - RA::AddRegisters(RASet, FPRClass, FPRBase, NumFPRs); + RAPass = CTX->GetRegisterAllocatorPass(); + RAPass->SetSupportsSpills(false); - Graph = RA::AllocateRegisterGraph(RASet, 9000); + RASet = RAPass->AllocateRegisterSet(RegisterCount, RegisterClasses); + RAPass->AddRegisters(RASet, GPRClass, GPRBase, NumGPRs); + RAPass->AddRegisters(RASet, FPRClass, FPRBase, NumFPRs); + + Graph = RAPass->AllocateRegisterGraph(RASet, 9000); LiveRanges.resize(9000); // Just set the entire range as executable @@ -165,11 +206,13 @@ JITCore::JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadSt #if _M_X86_64 Sim.SetCPUFeatures(vixl::CPUFeatures::All()); #endif + SetAllowAssembler(true); + CreateCustomDispatch(Thread); } JITCore::~JITCore() { - FreeRegisterGraph(Graph); - FreeRegisterSet(RASet); + RAPass->FreeRegisterGraph(); + RAPass->FreeRegisterSet(RASet); } void JITCore::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant) { @@ -202,7 +245,7 @@ const std::array RAFPR = { v29, v30, v31}; uint32_t JITCore::GetPhys(uint32_t Node) { - uint32_t Reg = RA::GetNodeRegister(Graph, Node); + uint32_t Reg = RAPass->GetNodeRegister(Node); if (Reg < FPRBase) return Reg; @@ -242,114 +285,6 @@ aarch64::VRegister JITCore::GetDst(uint32_t Node) { return RAFPR[Reg]; } -void JITCore::FindNodeClasses() { - uintptr_t ListBegin = CurrentIR->GetListData(); - uintptr_t DataBegin = CurrentIR->GetData(); - - IR::NodeWrapperIterator Begin = CurrentIR->begin(); - IR::NodeWrapperIterator End = CurrentIR->end(); - - while (Begin != End) { - using namespace FEXCore::IR; - - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); - - if (IROp->HasDest) { - // XXX: This needs to be better - switch (IROp->Op) { - case OP_LOADCONTEXT: { - auto Op = IROp->C(); - if (Op->Size == 16) - RA::SetNodeClass(Graph, WrapperOp->ID(), FPRClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - case IR::OP_LOADMEM: { - auto Op = IROp->C(); - if (Op->Size == 16) - RA::SetNodeClass(Graph, WrapperOp->ID(), FPRClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - - case OP_ZEXT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - - if (Op->SrcSize == 64) { - RA::SetNodeClass(Graph, WrapperOp->ID(), FPRClass); - } - else { - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - } - break; - } - case OP_CPUID: RA::SetNodeClass(Graph, WrapperOp->ID(), FPRClass); break; - default: - if (IROp->Op >= IR::OP_VOR) - RA::SetNodeClass(Graph, WrapperOp->ID(), FPRClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - } - ++Begin; - } -} - -bool JITCore::CalculateLiveRange(uint32_t Nodes) { - if (Nodes > LiveRanges.size()) { - LiveRanges.resize(Nodes); - } - memset(&LiveRanges.at(0), 0xFF, Nodes * sizeof(LiveRange)); - - uintptr_t ListBegin = CurrentIR->GetListData(); - uintptr_t DataBegin = CurrentIR->GetData(); - - IR::NodeWrapperIterator Begin = CurrentIR->begin(); - IR::NodeWrapperIterator End = CurrentIR->end(); - - while (Begin != End) { - using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); - - uint32_t Node = WrapperOp->ID(); - - // If the destination hasn't yet been set then set it now - if (IROp->HasDest && LiveRanges[Node].Begin == ~0U) { - LiveRanges[Node].Begin = Node; - // Default to ending right where it starts - LiveRanges[Node].End = Node; - } - - for (uint8_t i = 0; i < IROp->NumArgs; ++i) { - uint32_t ArgNode = IROp->Args[i].ID(); - // Set the node end to be at least here - LiveRanges[ArgNode].End = Node; - } - - ++Begin; - } - - // Now that we have all the live ranges calculated we need to add them to our interference graph - for (uint32_t i = 0; i < Nodes; ++i) { - for (uint32_t j = i + 1; j < Nodes; ++j) { - if (!(LiveRanges[i].Begin >= LiveRanges[j].End || - LiveRanges[j].Begin >= LiveRanges[i].End)) { - RA::AddNodeInterference(Graph, i, j); - } - } - } - - return RA::AllocateRegisters(Graph); -} - void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData) { using namespace aarch64; JumpTargets.clear(); @@ -362,11 +297,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const IR::NodeWrapperIterator End = CurrentIR->end(); uintptr_t ListSize = CurrentIR->GetListSize(); - uint32_t SSACount = ListSize / sizeof(IR::OrderedNode); - - ResetRegisterGraph(Graph, SSACount); - FindNodeClasses(); - HasRA = CalculateLiveRange(SSACount); + HasRA = RAPass->HasFullRA(); LogMan::Throw::A(HasRA, "Arm64 JIT only works with RA"); @@ -393,14 +324,17 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const auto Buffer = GetBuffer(); auto Entry = Buffer->GetOffsetAddress(GetCursorOffset()); - void *Memory = CTX->MemoryMapper.GetMemoryBase(); - LoadConstant(MEM_BASE, (uint64_t)Memory); + if (!CustomDispatchGenerated) { + void *Memory = CTX->MemoryMapper.GetMemoryBase(); + LoadConstant(MEM_BASE, (uint64_t)Memory); + mov(STATE, x0); + } while (Begin != End) { using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); uint8_t OpSize = IROp->Size; uint32_t Node = WrapperOp->ID(); @@ -411,7 +345,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const auto Name = FEXCore::IR::GetName(IROp->Op); if (IROp->HasDest) { - uint32_t PhysReg = RA::GetNodeRegister(Graph, Node); + uint32_t PhysReg = RAPass->GetNodeRegister(Node); if (PhysReg >= FPRBase) Inst << "\tFPR" << GetPhys(Node) << " = " << Name << " "; else @@ -423,14 +357,12 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const for (uint8_t i = 0; i < IROp->NumArgs; ++i) { uint32_t ArgNode = IROp->Args[i].ID(); - uint32_t PhysReg = RA::GetNodeRegister(Graph, ArgNode); + uint32_t PhysReg = RAPass->GetNodeRegister(ArgNode); if (PhysReg >= FPRBase) Inst << "FPR" << GetPhys(ArgNode) << (i + 1 == IROp->NumArgs ? "" : ", "); else Inst << "Reg" << GetPhys(ArgNode) << (i + 1 == IROp->NumArgs ? "" : ", "); } - - LogMan::Msg::D("%s", Inst.str().c_str()); } } @@ -439,10 +371,10 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const auto IsTarget = JumpTargets.find(WrapperOp->ID()); if (IsTarget == JumpTargets.end()) { // XXX: This is a memory leak - JumpTargets.try_emplace(WrapperOp->ID(), new aarch64::Label); + JumpTargets.try_emplace(WrapperOp->ID()); } else { - bind(IsTarget->second); + bind(&IsTarget->second); } break; } @@ -467,7 +399,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const // X1: ThreadState // X2: Pointer to SyscallArguments - uint64_t SPOffset = AlignUp((2 + RA64.size() + 7 + 2) * 8, 16); + uint64_t SPOffset = AlignUp((RA64.size() + 7 + 1) * 8, 16); sub(sp, sp, SPOffset); for (uint32_t i = 0; i < 7; ++i) @@ -478,14 +410,26 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const str(RA, MemOperand(sp, 7 * 8 + i * 8)); i++; } - str(STATE, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 1 * 8)); - str(MEM_BASE, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 2 * 8)); - str(lr, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 3 * 8)); + str(lr, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 0 * 8)); - // x0 = threadstate already - LoadConstant(x1, reinterpret_cast(&CTX->SyscallHandler)); - mov (x2, sp); + LoadConstant(x0, reinterpret_cast(&CTX->SyscallHandler)); + mov(x1, STATE); + mov(x2, sp); + +#if _M_X86_64 CallRuntime(SyscallThunk); +#else + using ClassPtrType = uint64_t (FEXCore::SyscallHandler::*)(FEXCore::Core::InternalThreadState *, FEXCore::HLE::SyscallArguments *); + union PtrCast { + ClassPtrType ClassPtr; + uintptr_t Data; + }; + + PtrCast Ptr; + Ptr.ClassPtr = &FEXCore::SyscallHandler::HandleSyscall; + LoadConstant(x3, Ptr.Data); + blr(x3); +#endif // Result is now in x0 // Fix the stack and any values that were stepped on @@ -498,9 +442,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const // Move result to its destination register mov(GetDst(Node), x0); - ldr(STATE, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 1 * 8)); - ldr(MEM_BASE, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 2 * 8)); - ldr(lr, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 3 * 8)); + ldr(lr, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 0 * 8)); add(sp, sp, SPOffset); break; @@ -517,9 +459,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const i++; } - str(STATE, MemOperand(sp, RA64.size() * 8 + 0 * 8)); - str(MEM_BASE, MemOperand(sp, RA64.size() * 8 + 1 * 8)); - str(lr, MemOperand(sp, RA64.size() * 8 + 2 * 8)); + str(lr, MemOperand(sp, RA64.size() * 8 + 0 * 8)); // x0 = CPUID Handler // x1 = CPUID Function @@ -540,9 +480,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const auto Dst = GetDst(Node); ldr(Dst, MemOperand(sp, RA64.size() * 8 + 3 * 8)); - ldr(STATE, MemOperand(sp, RA64.size() * 8 + 0 * 8)); - ldr(MEM_BASE, MemOperand(sp, RA64.size() * 8 + 1 * 8)); - ldr(lr, MemOperand(sp, RA64.size() * 8 + 2 * 8)); + ldr(lr, MemOperand(sp, RA64.size() * 8 + 0 * 8)); add(sp, sp, SPOffset); @@ -551,7 +489,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const case IR::OP_EXTRACTELEMENT: { auto Op = IROp->C(); - uint32_t PhysReg = RA::GetNodeRegister(Graph, Op->Header.Args[0].ID()); + uint32_t PhysReg = RAPass->GetNodeRegister(Op->Header.Args[0].ID()); if (PhysReg >= FPRBase) { switch (OpSize) { case 4: @@ -574,16 +512,15 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const Label *TargetLabel; auto IsTarget = JumpTargets.find(Op->Header.Args[0].ID()); if (IsTarget == JumpTargets.end()) { - TargetLabel = JumpTargets.try_emplace(Op->Header.Args[0].ID(), new aarch64::Label).first->second; + TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[0].ID()).first->second; } else { - TargetLabel = IsTarget->second; + TargetLabel = &IsTarget->second; } b(TargetLabel); break; } - case IR::OP_CONDJUMP: { auto Op = IROp->C(); @@ -591,10 +528,10 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const auto IsTarget = JumpTargets.find(Op->Header.Args[1].ID()); if (IsTarget == JumpTargets.end()) { // XXX: This is a memory leak - TargetLabel = JumpTargets.try_emplace(Op->Header.Args[1].ID(), new aarch64::Label).first->second; + TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[1].ID()).first->second; } else { - TargetLabel = IsTarget->second; + TargetLabel = &IsTarget->second; } cbnz(GetSrc(Op->Header.Args[0].ID()), TargetLabel); @@ -709,7 +646,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const lsrv(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); else lsrv(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; + break; } case IR::OP_ASHR: { auto Op = IROp->C(); @@ -717,7 +654,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const asrv(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); else asrv(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; + break; } case IR::OP_LSHL: { auto Op = IROp->C(); @@ -725,7 +662,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const lslv(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); else lslv(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; + break; } case IR::OP_ROR: { auto Op = IROp->C(); @@ -745,7 +682,6 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } break; } - case IR::OP_ROL: { auto Op = IROp->C(); uint8_t Mask = OpSize * 8 - 1; @@ -768,7 +704,6 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } break; } - case IR::OP_SEXT: { auto Op = IROp->C(); LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); @@ -794,7 +729,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const case IR::OP_ZEXT: { auto Op = IROp->C(); LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - uint32_t PhysReg = RA::GetNodeRegister(Graph, Op->Header.Args[0].ID()); + uint32_t PhysReg = RAPass->GetNodeRegister(Op->Header.Args[0].ID()); if (PhysReg >= FPRBase) { // FPR -> GPR transfer with free truncation switch (Op->SrcSize) { @@ -886,19 +821,20 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } break; } - case IR::OP_BFE: { auto Op = IROp->C(); LogMan::Throw::A(OpSize <= 16, "OpSize is too large for BFE: %d", OpSize); + LogMan::Throw::A(Op->Width != 0, "Invalid BFE width of 0"); + auto Dst = GetDst(Node); if (OpSize == 16) { LogMan::Throw::A(!(Op->lsb < 64 && (Op->lsb + Op->Width > 64)), "Trying to BFE an XMM across the 64bit split: Beginning at %d, ending at %d", Op->lsb, Op->lsb + Op->Width); uint8_t Offset = Op->lsb; if (Offset < 64) { - mov(Dst, GetSrc(Op->Header.Args[0].ID()), 0); + mov(Dst, GetSrc(Op->Header.Args[0].ID()).D(), 0); } else { - mov(Dst, GetSrc(Op->Header.Args[0].ID()), 1); + mov(Dst, GetSrc(Op->Header.Args[0].ID()).D(), 1); Offset -= 64; } @@ -912,18 +848,20 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } else { lsr(Dst, GetSrc(Op->Header.Args[0].ID()), Op->lsb); - and_(Dst, Dst, ((1ULL << Op->Width) - 1)); + if (Op->Width != 64) { + and_(Dst, Dst, ((1ULL << Op->Width) - 1)); + } } break; } case IR::OP_POPCOUNT: { auto Op = IROp->C(); auto Dst = GetDst(Node); - fmov(VTMP1, GetSrc(Op->Header.Args[0].ID())); + fmov(VTMP1.V1D(), GetSrc(Op->Header.Args[0].ID())); cnt(VTMP1.V8B(), VTMP1.V8B()); addv(VTMP1.B(), VTMP1.V8B()); - umov(Dst, VTMP1.B(), 0); - break; + umov(Dst.W(), VTMP1.B(), 0); + break; } case IR::OP_FINDLSB: { auto Op = IROp->C(); @@ -980,7 +918,6 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const mov(GetDst(Node), TMP2); break; } - case IR::OP_SELECT: { auto Op = IROp->C(); @@ -1162,7 +1099,6 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } break; } - case IR::OP_LUREM: { auto Op = IROp->C(); // Each source is OpSize in size @@ -1219,7 +1155,6 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } break; } - case IR::OP_VINSELEMENT: { auto Op = IROp->C(); mov(VTMP1, GetSrc(Op->Header.Args[0].ID())); @@ -1397,17 +1332,28 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const eor(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B()); break; } + case IR::OP_VEXTR: { + auto Op = IROp->C(); + // AArch64 ext op has bit arrangement as [Vm:Vn] so arguments need to be swapped + ext(GetDst(Node).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), Op->Index); + break; + } case IR::OP_VUSHLS: { auto Op = IROp->C(); switch (Op->ElementSize) { + case 1: { + dup(VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID())); + ushl(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), VTMP1.V16B()); + break; + } case 2: { - dup(VTMP1.V8H(), GetSrc(Op->Header.Args[1].ID())); + dup(VTMP1.V8H(), GetSrc(Op->Header.Args[1].ID())); ushl(GetDst(Node).V8H(), GetSrc(Op->Header.Args[0].ID()).V8H(), VTMP1.V8H()); break; } case 4: { - dup(VTMP1.V4S(), GetSrc(Op->Header.Args[1].ID())); + dup(VTMP1.V4S(), GetSrc(Op->Header.Args[1].ID())); ushl(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S(), VTMP1.V4S()); break; } @@ -1420,11 +1366,58 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } break; } + case IR::OP_VUMIN: { + auto Op = IROp->C(); + switch (Op->ElementSize) { + case 1: { + umin(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B()); + break; + } + case 2: { + umin(GetDst(Node).V8H(), GetSrc(Op->Header.Args[0].ID()).V8H(), GetSrc(Op->Header.Args[1].ID()).V8H()); + break; + } + case 4: { + umin(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S(), GetSrc(Op->Header.Args[1].ID()).V4S()); + break; + } + case 8: { + umin(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D(), GetSrc(Op->Header.Args[1].ID()).V2D()); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + break; + } + case IR::OP_VSMIN: { + auto Op = IROp->C(); + switch (Op->ElementSize) { + case 1: { + smin(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B()); + break; + } + case 2: { + smin(GetDst(Node).V8H(), GetSrc(Op->Header.Args[0].ID()).V8H(), GetSrc(Op->Header.Args[1].ID()).V8H()); + break; + } + case 4: { + smin(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S(), GetSrc(Op->Header.Args[1].ID()).V4S()); + break; + } + case 8: { + smin(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D(), GetSrc(Op->Header.Args[1].ID()).V2D()); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + break; + } case IR::OP_CYCLECOUNTER: { - if (0) +#ifdef DEBUG_CYCLES movz(GetDst(Node), 0); - else +#else mrs(GetDst(Node), CNTVCT_EL0); +#endif break; } default: @@ -1437,14 +1430,174 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const FinalizeCode(); #if _M_X86_64 - auto CodeEnd = Buffer->GetOffsetAddress(GetCursorOffset()); - HostToGuest[State->State.State.rip] = std::make_pair(Entry, CodeEnd); - return (void*)SimulatorExecution; -#else - return reinterpret_cast(Entry); + if (!CustomDispatchGenerated) { + auto CodeEnd = Buffer->GetOffsetAddress(GetCursorOffset()); + HostToGuest[State->State.State.rip] = std::make_pair(Entry, CodeEnd); + return (void*)SimulatorExecution; + } #endif + + return reinterpret_cast(Entry); } +void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) { + auto Buffer = GetBuffer(); + DispatchPtr = Buffer->GetOffsetAddress(GetCursorOffset()); + EmissionCheckScope(this, 0); + + // while (!Thread->State.RunningEvents.ShouldStop.load()) { + // Ptr = FindBlock(RIP) + // if (!Ptr) + // Ptr = CTX->CompileBlock(RIP); + // + // if (Ptr) + // Ptr(); + // else + // { + // Ptr = FallbackCore->CompileBlock() + // if (Ptr) + // Ptr() + // else { + // ShouldStop = true; + // } + // } + // } + + // Push all the register we need to save + PushCalleeSavedRegisters(); + + // Push our memory base to the correct register + void *Memory = CTX->MemoryMapper.GetMemoryBase(); + LoadConstant(MEM_BASE, (uint64_t)Memory); + // Move our thread pointer to the correct register + // This is passed in to parameter 0 (x0) + mov(STATE, x0); + + aarch64::Label LoopTop; + bind(&LoopTop); + + // Load in our RIP + ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::ThreadState, State.rip))); + LoadConstant(x0, Thread->BlockCache->GetPagePointer()); + + // Steal the page offset + and_(x1, x2, 0x0FFF); + // Offset the address and add to our page pointer + add(x3, x0, Operand(x2, LSR, 12)); + + // Load the pointer from the offset + ldr(x3, MemOperand(x3)); + aarch64::Label NoBlock; + + // If page pointer is zero then we have no block + cbz(x3, &NoBlock); + + // Now load from that pointer offset by the page offset to get our real block + ldr(x3, MemOperand(x3, x1)); + cbz(x3, &NoBlock); + + // If we've made it here then we have a real compiled block + { + blr(x3); + } + + aarch64::Label ExitCheck; + bind(&ExitCheck); + + constexpr uint64_t ShouldStopOffset = offsetof(FEXCore::Core::ThreadState, RunningEvents.ShouldStop); + // If we don't need to stop then keep going + add(x1, STATE, ShouldStopOffset); + ldarb(x0, MemOperand(x1)); + cbz(x0, &LoopTop); + + PopCalleeSavedRegisters(); + + // Return from the function + // LR is set to the correct return location now + ret(); + + aarch64::Label FallbackCore; + // Need to create the block + { + bind(&NoBlock); + + LoadConstant(x0, reinterpret_cast(CTX)); + mov(x1, STATE); + +#if _M_X86_64 + CallRuntime(CompileBlockThunk); +#else + using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::InternalThreadState *, uint64_t); + union PtrCast { + ClassPtrType ClassPtr; + uintptr_t Data; + }; + + PtrCast Ptr; + Ptr.ClassPtr = &FEXCore::Context::Context::CompileBlock; + LoadConstant(x3, Ptr.Data); + + // X2 contains our guest RIP + blr(x3); // { ThreadState, RIP} +#endif + // X0 now contains either nullptr or block pointer + cbz(x0, &FallbackCore); + blr(x0); + + b(&ExitCheck); + } + + aarch64::Label ExitError; + // We need to fallback to our fallback core + { + bind(&FallbackCore); + +#if _M_X86_64 + // XXX: Fallback core doesn't work on x86-64 + // We can't tell the difference between simulator entry points and not + b(&ExitError); +#else + LoadConstant(x0, reinterpret_cast(CTX)); + mov(x1, STATE); + + using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::InternalThreadState *, uint64_t); + union PtrCast { + ClassPtrType ClassPtr; + uintptr_t Data; + }; + + PtrCast Ptr; + Ptr.ClassPtr = &FEXCore::Context::Context::CompileFallbackBlock; + LoadConstant(x3, Ptr.Data); + + // X2 contains our guest RIP + blr(x3); // {ThreadState, RIP} +#endif + // X0 now contains either nullptr or block pointer + cbz(x0, &ExitError); + blr(x0); + + b(&ExitCheck); + } + + // Exit error + { + bind(&ExitError); + LoadConstant(x0, 1); + add(x1, STATE, ShouldStopOffset); + stlrb(x0, MemOperand(x1)); + b(&ExitCheck); + } + +#if _M_X86_64 + CustomDispatchEnd = Buffer->GetOffsetAddress(GetCursorOffset()); +#endif + + FinalizeCode(); + // XXX: Crashes currently. + // Disabling will be useful for debugging ThreadState + // CustomDispatchGenerated = true; +} FEXCore::CPU::CPUBackend *CreateJITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) { return new JITCore(ctx, Thread); diff --git a/Source/Interface/Core/JIT/x86_64/JIT.cpp b/Source/Interface/Core/JIT/x86_64/JIT.cpp index b0468cce7..ea6c8deba 100644 --- a/Source/Interface/Core/JIT/x86_64/JIT.cpp +++ b/Source/Interface/Core/JIT/x86_64/JIT.cpp @@ -1,6 +1,6 @@ #include "Interface/Context/Context.h" -#include "Interface/Core/RegisterAllocation.h" #include "Interface/Core/InternalThreadState.h" +#include "Interface/IR/Passes/RegisterAllocationPass.h" #include "Interface/Core/JIT/x86_64/JIT.h" #include @@ -9,6 +9,7 @@ using namespace Xbyak; #include #include #include +// #define DEBUG_RA 1 namespace FEXCore::CPU { // Temp registers @@ -24,6 +25,14 @@ namespace FEXCore::CPU { // r11 assigned to temp state #define TEMP_STACK r11 #define STATE rdi +using namespace Xbyak::util; +const std::array RA64 = { rsi, r8, r9, r10, r11, rbx, rbp, r12, r13, r14, r15 }; +const std::array RA32 = { esi, r8d, r9d, r10d, r11d, ebx, ebp, r12d, r13d, r14d, r15d }; +const std::array RA16 = { si, r8w, r9w, r10w, r11w, bx, bp, r12w, r13w, r14w, r15w }; +const std::array RA8 = { sil, r8b, r9b, r10b, r11b, bl, bpl, r12b, r13b, r14b, r15b }; +const std::array RAXMM = { xmm0, xmm1, xmm2, xmm3, xmm4, xmm5, xmm6, xmm7, xmm8, xmm9, xmm10 }; +const std::array RAXMM_x = { xmm0, xmm1, xmm2, xmm3, xmm4, xmm5, xmm6, xmm7, xmm8, xmm9, xmm10 }; + class JITCore final : public CPUBackend, public Xbyak::CodeGenerator { public: @@ -36,10 +45,16 @@ public: bool NeedsOpDispatch() override { return true; } + bool HasCustomDispatch() const override { return CustomDispatchGenerated; } + + void ExecuteCustomDispatch(FEXCore::Core::ThreadState *Thread) override { + DispatchPtr(reinterpret_cast(Thread->InternalState)); + } + private: FEXCore::Context::Context *CTX; FEXCore::IR::IRListView const *CurrentIR; - std::unordered_map JumpTargets; + std::unordered_map JumpTargets; std::vector Stack; bool MemoryDebug = false; @@ -47,21 +62,19 @@ private: /** * @name Register Allocation * @{ */ - constexpr static uint32_t NumGPRs = 11; - constexpr static uint32_t NumXMMs = 11; + constexpr static uint32_t NumGPRs = RA64.size(); // 4 is the minimum required for GPR ops + constexpr static uint32_t NumXMMs = RAXMM.size(); constexpr static uint32_t RegisterCount = NumGPRs + NumXMMs; constexpr static uint32_t RegisterClasses = 2; constexpr static uint32_t GPRBase = 0; - constexpr static uint32_t GPRClass = 0; + constexpr static uint32_t GPRClass = IR::RegisterAllocationPass::GPRClass; constexpr static uint32_t XMMBase = NumGPRs; - constexpr static uint32_t XMMClass = 1; + constexpr static uint32_t XMMClass = IR::RegisterAllocationPass::FPRClass; - RA::RegisterSet *RASet; + IR::RegisterAllocationPass::RegisterSet *RASet; /** @} */ - void FindNodeClasses(); - bool CalculateLiveRange(uint32_t Nodes); constexpr static uint8_t RA_8 = 0; constexpr static uint8_t RA_16 = 1; constexpr static uint8_t RA_32 = 2; @@ -69,7 +82,7 @@ private: constexpr static uint8_t RA_XMM = 4; bool HasRA = false; - RA::RegisterGraph *Graph; + IR::RegisterAllocationPass::RegisterGraph *Graph; uint32_t GetPhys(uint32_t Node); template @@ -81,12 +94,11 @@ private: Xbyak::Xmm GetSrc(uint32_t Node); Xbyak::Xmm GetDst(uint32_t Node); - struct LiveRange { - uint32_t Begin; - uint32_t End; - }; - - std::vector LiveRanges; + void CreateCustomDispatch(); + bool CustomDispatchGenerated {false}; + using CustomDispatch = void(*)(FEXCore::Core::InternalThreadState *Thread); + CustomDispatch DispatchPtr{}; + IR::RegisterAllocationPass *RAPass; }; JITCore::JITCore(FEXCore::Context::Context *ctx) @@ -94,18 +106,21 @@ JITCore::JITCore(FEXCore::Context::Context *ctx) , CTX {ctx} { Stack.resize(9000 * 16 * 64); - RASet = RA::AllocateRegisterSet(RegisterCount, RegisterClasses); - RA::AddRegisters(RASet, GPRClass, GPRBase, NumGPRs); - RA::AddRegisters(RASet, XMMClass, XMMBase, NumXMMs); + RAPass = CTX->GetRegisterAllocatorPass(); + RAPass->SetSupportsSpills(false); - Graph = RA::AllocateRegisterGraph(RASet, 9000); - LiveRanges.resize(9000); + RASet = RAPass->AllocateRegisterSet(RegisterCount, RegisterClasses); + RAPass->AddRegisters(RASet, GPRClass, GPRBase, NumGPRs); + RAPass->AddRegisters(RASet, XMMClass, XMMBase, NumXMMs); + + Graph = RAPass->AllocateRegisterGraph(RASet, 9000); + CreateCustomDispatch(); } JITCore::~JITCore() { printf("Used %ld bytes for compiling\n", getCurr() - getCode()); - FreeRegisterGraph(Graph); - FreeRegisterSet(RASet); + RAPass->FreeRegisterSet(RASet); + RAPass->FreeRegisterGraph(); } static void LoadMem(uint64_t Addr, uint64_t Data, uint8_t Size) { @@ -118,124 +133,8 @@ static void StoreMem(uint64_t Addr, uint64_t Data, uint8_t Size) { LogMan::Msg::D("\tStoring: 0x%016lx", Data); } -void JITCore::FindNodeClasses() { - uintptr_t ListBegin = CurrentIR->GetListData(); - uintptr_t DataBegin = CurrentIR->GetData(); - - IR::NodeWrapperIterator Begin = CurrentIR->begin(); - IR::NodeWrapperIterator End = CurrentIR->end(); - - while (Begin != End) { - using namespace FEXCore::IR; - - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); - - if (IROp->HasDest) { - // XXX: This needs to be better - switch (IROp->Op) { - case OP_LOADCONTEXT: { - auto Op = IROp->C(); - if (Op->Size == 16) - RA::SetNodeClass(Graph, WrapperOp->ID(), XMMClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - case OP_STORECONTEXT: { - auto Op = IROp->C(); - if (Op->Size == 16) - RA::SetNodeClass(Graph, WrapperOp->ID(), XMMClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - case IR::OP_LOADMEM: { - auto Op = IROp->C(); - if (Op->Size == 16) - RA::SetNodeClass(Graph, WrapperOp->ID(), XMMClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - - case OP_ZEXT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - - if (Op->SrcSize == 64) { - RA::SetNodeClass(Graph, WrapperOp->ID(), XMMClass); - } - else { - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - } - break; - } - case OP_CPUID: RA::SetNodeClass(Graph, WrapperOp->ID(), XMMClass); break; - default: - if (IROp->Op >= IR::OP_VOR) - RA::SetNodeClass(Graph, WrapperOp->ID(), XMMClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - } - ++Begin; - } -} - -bool JITCore::CalculateLiveRange(uint32_t Nodes) { - if (Nodes > LiveRanges.size()) { - LiveRanges.resize(Nodes); - } - memset(&LiveRanges.at(0), 0xFF, Nodes * sizeof(LiveRange)); - - uintptr_t ListBegin = CurrentIR->GetListData(); - uintptr_t DataBegin = CurrentIR->GetData(); - - IR::NodeWrapperIterator Begin = CurrentIR->begin(); - IR::NodeWrapperIterator End = CurrentIR->end(); - - while (Begin != End) { - using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); - - uint32_t Node = WrapperOp->ID(); - - // If the destination hasn't yet been set then set it now - if (IROp->HasDest && LiveRanges[Node].Begin == ~0U) { - LiveRanges[Node].Begin = Node; - // Default to ending right where it starts - LiveRanges[Node].End = Node; - } - - for (uint8_t i = 0; i < IROp->NumArgs; ++i) { - uint32_t ArgNode = IROp->Args[i].ID(); - // Set the node end to be at least here - LiveRanges[ArgNode].End = Node; - } - - ++Begin; - } - - // Now that we have all the live ranges calculated we need to add them to our interference graph - for (uint32_t i = 0; i < Nodes; ++i) { - for (uint32_t j = i + 1; j < Nodes; ++j) { - if (!(LiveRanges[i].Begin >= LiveRanges[j].End || - LiveRanges[j].Begin >= LiveRanges[i].End)) { - RA::AddNodeInterference(Graph, i, j); - } - } - } - - return RA::AllocateRegisters(Graph); -} - uint32_t JITCore::GetPhys(uint32_t Node) { - uint32_t Reg = RA::GetNodeRegister(Graph, Node); + uint32_t Reg = RAPass->GetNodeRegister(Node); if (Reg < XMMBase) return Reg; @@ -247,14 +146,6 @@ uint32_t JITCore::GetPhys(uint32_t Node) { return ~0U; } -using namespace Xbyak::util; -const std::array RA64 = { rsi, r8, r9, r10, r11, rbx, rbp, r12, r13, r14, r15 }; -const std::array RA32 = { esi, r8d, r9d, r10d, r11d, ebx, ebp, r12d, r13d, r14d, r15d }; -const std::array RA16 = { si, r8w, r9w, r10w, r11w, bx, bp, r12w, r13w, r14w, r15w }; -const std::array RA8 = { sil, r8b, r9b, r10b, r11b, bl, bpl, r12b, r13b, r14b, r15b }; -const std::array RAXMM = { xmm0, xmm1, xmm2, xmm3, xmm4, xmm5, xmm6, xmm7, xmm8, xmm9, xmm10 }; -const std::array RAXMM_x = { xmm0, xmm1, xmm2, xmm3, xmm4, xmm5, xmm6, xmm7, xmm8, xmm9, xmm10 }; - template Xbyak::Reg JITCore::GetSrc(uint32_t Node) { // rax, rcx, rdx, rsi, r8, r9, @@ -306,12 +197,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const uintptr_t ListBegin = CurrentIR->GetListData(); uintptr_t DataBegin = CurrentIR->GetData(); - IR::NodeWrapperIterator Begin = CurrentIR->begin(); - IR::NodeWrapperIterator End = CurrentIR->end(); - - uintptr_t ListSize = CurrentIR->GetListSize(); - - uint32_t SSACount = ListSize / sizeof(IR::OrderedNode); + uint32_t SSACount = CurrentIR->GetSSACount(); uint64_t ListStackSize = SSACount * 16; if (ListStackSize > Stack.size()) { Stack.resize(ListStackSize); @@ -319,10 +205,9 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const void *Entry = getCurr(); - ResetRegisterGraph(Graph, SSACount); - FindNodeClasses(); - HasRA = CalculateLiveRange(SSACount); + HasRA = RAPass->HasFullRA(); + uint32_t SpillSlots = RAPass->SpillSlots(); if (HasRA) { push(rbx); push(rbp); @@ -334,2464 +219,2691 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const else { mov(TEMP_STACK, reinterpret_cast(&Stack.at(0))); } - while (Begin != End) { - using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); - uint8_t OpSize = IROp->Size; - uint32_t Node = WrapperOp->ID(); - - if (HasRA) { -#ifdef DEBUG_RA - std::stringstream Inst; - auto Name = FEXCore::IR::GetName(IROp->Op); - - if (IROp->HasDest) { - uint32_t PhysReg = RA::GetNodeRegister(Graph, Node); - if (PhysReg >= XMMBase) - Inst << "\tXMM" << GetPhys(Node) << " = " << Name << " "; - else - Inst << "\tReg" << GetPhys(Node) << " = " << Name << " "; - } - else { - Inst << "\t" << Name << " "; - } - - for (uint8_t i = 0; i < IROp->NumArgs; ++i) { - uint32_t ArgNode = IROp->Args[i].ID(); - uint32_t PhysReg = RA::GetNodeRegister(Graph, ArgNode); - if (PhysReg >= XMMBase) - Inst << "XMM" << GetPhys(ArgNode) << (i + 1 == IROp->NumArgs ? "" : ", "); - else - Inst << "Reg" << GetPhys(ArgNode) << (i + 1 == IROp->NumArgs ? "" : ", "); - } - - LogMan::Msg::D("%s", Inst.str().c_str()); -#endif - } - - switch (IROp->Op) { - case IR::OP_BEGINBLOCK: { - auto IsTarget = JumpTargets.find(WrapperOp->ID()); - if (IsTarget == JumpTargets.end()) { - JumpTargets[WrapperOp->ID()] = L(); - } - else { - L(IsTarget->second); - } - break; - } - case IR::OP_ENDBLOCK: { - auto Op = IROp->C(); - if (Op->RIPIncrement) { - add(qword [STATE + offsetof(FEXCore::Core::CPUState, rip)], Op->RIPIncrement); - } - break; - } - case IR::OP_EXITFUNCTION: - case IR::OP_ENDFUNCTION: { - if (HasRA) { - pop(r15); - pop(r14); - pop(r13); - pop(r12); - pop(rbp); - pop(rbx); - } - ret(); - break; - } - case IR::OP_BREAK: { - auto Op = IROp->C(); - switch (Op->Reason) { - case 4: // HLT - ud2(); - break; - default: LogMan::Msg::A("Unknown Break reason: %d", Op->Reason); - } - } - break; - case IR::OP_JUMP: { - auto Op = IROp->C(); - - Label *TargetLabel; - auto IsTarget = JumpTargets.find(Op->Header.Args[0].ID()); - if (IsTarget == JumpTargets.end()) { - TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[0].ID(), Label{}).first->second; - } - else { - TargetLabel = &IsTarget->second; - } - - jmp(*TargetLabel); - break; - } - default: break; - } - - if (HasRA) { - switch (IROp->Op) { - case IR::OP_CONDJUMP: { - auto Op = IROp->C(); - - Label *TargetLabel; - auto IsTarget = JumpTargets.find(Op->Header.Args[1].ID()); - if (IsTarget == JumpTargets.end()) { - TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[1].ID(), Label{}).first->second; - } - else { - TargetLabel = &IsTarget->second; - } - - cmp(GetSrc(Op->Header.Args[0].ID()), 0); - jne(*TargetLabel); - break; - } - case IR::OP_LOADCONTEXT: { - auto Op = IROp->C(); - switch (Op->Size) { - case 1: { - mov(GetDst(Node), byte [STATE + Op->Offset]); - } - break; - case 2: { - mov(GetDst(Node), word [STATE + Op->Offset]); - } - break; - case 4: { - mov(GetDst(Node), dword [STATE + Op->Offset]); - } - break; - case 8: { - mov(GetDst(Node), qword [STATE + Op->Offset]); - } - break; - case 16: { - if (Op->Offset % 16 == 0) - movaps(GetDst(Node), xword [STATE + Op->Offset]); - else - movups(GetDst(Node), xword [STATE + Op->Offset]); - } - break; - default: LogMan::Msg::A("Unhandled LoadContext size: %d", Op->Size); - } - break; - } - case IR::OP_STORECONTEXT: { - auto Op = IROp->C(); - - switch (Op->Size) { - case 1: { - mov(byte [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); - } - break; - - case 2: { - mov(word [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); - } - break; - case 4: { - mov(dword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); - } - break; - case 8: { - mov(qword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); - } - break; - case 16: { - if (Op->Offset % 16 == 0) - movaps(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); - else - movups(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); - } - break; - default: LogMan::Msg::A("Unhandled StoreContext size: %d", Op->Size); - } - break; - } - case IR::OP_ADD: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - mov(rax, GetSrc(Op->Header.Args[1].ID())); - add(rax, GetSrc(Op->Header.Args[0].ID())); - mov(Dst, rax); - break; - } - case IR::OP_SUB: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - mov(rax, GetSrc(Op->Header.Args[0].ID())); - sub(rax, GetSrc(Op->Header.Args[1].ID())); - mov(Dst, rax); - break; - } - case IR::OP_XOR: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - mov(rax, GetSrc(Op->Header.Args[1].ID())); - xor(rax, GetSrc(Op->Header.Args[0].ID())); - mov(Dst, rax); - break; - } - case IR::OP_AND: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - mov(rax, GetSrc(Op->Header.Args[1].ID())); - and(rax, GetSrc(Op->Header.Args[0].ID())); - mov(Dst, rax); - break; - } - case IR::OP_OR: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - mov(rax, GetSrc(Op->Header.Args[1].ID())); - or (rax, GetSrc(Op->Header.Args[0].ID())); - mov(Dst, rax); - break; - } - case IR::OP_MOV: { - auto Op = IROp->C(); - mov (GetDst(Node), GetSrc(Op->Header.Args[0].ID())); - break; - } - case IR::OP_CONSTANT: { - auto Op = IROp->C(); - mov(GetDst(Node), Op->Constant); - break; - } - case IR::OP_POPCOUNT: { - auto Op = IROp->C(); - auto Dst64 = GetDst(Node); - - switch (OpSize) { - case 1: - movzx(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); - popcnt(Dst64, Dst64); - break; - case 2: { - movzx(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); - popcnt(Dst64, Dst64); - break; - } - case 4: - popcnt(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); - break; - case 8: - popcnt(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); - break; - } - break; - } - case IR::OP_ZEXT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - - uint32_t PhysReg = RA::GetNodeRegister(Graph, Op->Header.Args[0].ID()); - if (PhysReg >= XMMBase) { - // XMM -> GPR transfer with free truncation - switch (Op->SrcSize) { - case 8: - pextrb(al, GetSrc(Op->Header.Args[0].ID()), 0); - break; - case 16: - pextrw(ax, GetSrc(Op->Header.Args[0].ID()), 0); - break; - case 32: - pextrd(eax, GetSrc(Op->Header.Args[0].ID()), 0); - break; - case 64: - pextrw(rax, GetSrc(Op->Header.Args[0].ID()), 0); - break; - default: LogMan::Msg::A("Unhandled Zext size: %d", Op->SrcSize); break; - } - auto Dst = GetDst(Node); - mov(Dst, rax); - } - else { - if (Op->SrcSize == 64) { - vmovq(xmm15, Reg64(GetSrc(Op->Header.Args[0].ID()).getIdx())); - movapd(GetDst(Node), xmm15); - } - else { - auto Dst = GetDst(Node); - mov(rax, uint64_t((1ULL << Op->SrcSize) - 1)); - and(rax, GetSrc(Op->Header.Args[0].ID())); - mov(Dst, rax); - } - } - break; - } - case IR::OP_SEXT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - auto Dst = GetDst(Node); - - switch (Op->SrcSize / 8) { - case 1: - movsx(Dst, GetSrc(Op->Header.Args[0].ID())); - break; - case 2: - movsx(Dst, GetSrc(Op->Header.Args[0].ID())); - break; - case 4: - movsxd(Reg64(Dst.getIdx()), GetSrc(Op->Header.Args[0].ID())); - break; - case 8: - mov(Dst, GetSrc(Op->Header.Args[0].ID())); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", Op->SrcSize / 8); - } - break; - } - case IR::OP_BFE: { - auto Op = IROp->C(); - LogMan::Throw::A(OpSize <= 16, "OpSize is too large for BFE: %d", OpSize); - if (OpSize == 16) { - LogMan::Throw::A(!(Op->lsb < 64 && (Op->lsb + Op->Width > 64)), "Trying to BFE an XMM across the 64bit split: Beginning at %d, ending at %d", Op->lsb, Op->lsb + Op->Width); - movups(xmm15, GetSrc(Op->Header.Args[0].ID())); - uint8_t Offset = Op->lsb; - if (Offset < 64) { - pextrq(rax, xmm15, 0); - } - else { - pextrq(rax, xmm15, 1); - Offset -= 64; - } - - if (Offset) { - shr(rax, Offset); - } - - if (Op->Width != 64) { - mov(rcx, uint64_t((1ULL << Op->Width) - 1)); - and(rax, rcx); - } - - mov (GetDst(Node), rax); - } - else { - auto Dst = GetDst(Node); - mov(rax, GetSrc(Op->Header.Args[0].ID())); - - if (Op->lsb != 0) - shr(rax, Op->lsb); - - if (Op->Width != 64) { - mov(rcx, uint64_t((1ULL << Op->Width) - 1)); - and(rax, rcx); - } - mov(Dst, rax); - } - break; - } - case IR::OP_LSHR: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - auto Dst = GetDst(Node); - mov (rcx, GetSrc(Op->Header.Args[1].ID())); - and(rcx, Mask); - - shrx(Reg32e(Dst.getIdx(), 64), GetSrc(Op->Header.Args[0].ID()), rcx); - break; - } - case IR::OP_LSHL: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - auto Dst = GetDst(Node); - mov (rcx, GetSrc(Op->Header.Args[1].ID())); - and(rcx, Mask); - - shlx(Reg32e(Dst.getIdx(), 64), GetSrc(Op->Header.Args[0].ID()), rcx); - break; - } - case IR::OP_ASHR: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov (rcx, GetSrc(Op->Header.Args[1].ID())); - and(rcx, Mask); - switch (OpSize) { - case 1: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - sar(al, cl); - movsx(GetDst(Node), al); - break; - case 2: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - sar(ax, cl); - movsx(GetDst(Node), ax); - break; - case 4: - sarx(Reg32e(GetDst(Node).getIdx(), 32), GetSrc(Op->Header.Args[0].ID()), ecx); - break; - case 8: - sarx(Reg32e(GetDst(Node).getIdx(), 64), GetSrc(Op->Header.Args[0].ID()), rcx); - break; - default: LogMan::Msg::A("Unknown ASHR Size: %d\n", OpSize); break; - }; - break; - } - case IR::OP_ROL: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov (rcx, GetSrc(Op->Header.Args[1].ID())); - and(rcx, Mask); - switch (OpSize) { - case 1: { - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - rol(al, cl); - break; - } - case 2: { - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - rol(ax, cl); - break; - } - case 4: { - mov(eax, GetSrc(Op->Header.Args[0].ID())); - rol(eax, cl); - break; - } - case 8: { - mov(rax, GetSrc(Op->Header.Args[0].ID())); - rol(rax, cl); - break; - } - } - mov(GetDst(Node), rax); - break; - } - case IR::OP_ROR: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov (rcx, GetSrc(Op->Header.Args[1].ID())); - and(rcx, Mask); - switch (OpSize) { - case 1: { - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - ror(al, cl); - break; - } - case 2: { - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - ror(ax, cl); - break; - } - case 4: { - mov(eax, GetSrc(Op->Header.Args[0].ID())); - ror(eax, cl); - break; - } - case 8: { - mov(rax, GetSrc(Op->Header.Args[0].ID())); - ror(rax, cl); - break; - } - } - mov(GetDst(Node), rax); - break; - } - case IR::OP_MUL: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - - switch (OpSize) { - case 1: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - movsx(rcx, GetSrc(Op->Header.Args[1].ID())); - imul(cl); - movsx(Dst, al); - break; - case 2: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - movsx(rcx, GetSrc(Op->Header.Args[1].ID())); - imul(cx); - movsx(Dst, ax); - break; - case 4: - movsxd(rax, GetSrc(Op->Header.Args[0].ID())); - imul(eax, GetSrc(Op->Header.Args[1].ID())); - movsx(Dst, eax); - break; - case 8: - mov(rax, GetSrc(Op->Header.Args[0].ID())); - imul(rax, GetSrc(Op->Header.Args[1].ID())); - mov(Dst, rax); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_MULH: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - movsx(rcx, GetSrc(Op->Header.Args[1].ID())); - imul(cl); - movsx(rax, ax); - mov(GetDst(Node), rax); - break; - case 2: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - movsx(rcx, GetSrc(Op->Header.Args[1].ID())); - imul(cx); - movsx(rax, dx); - mov(GetDst(Node), rax); - break; - case 4: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - imul(GetSrc(Op->Header.Args[1].ID())); - movsxd(rax, edx); - mov(GetDst(Node), rdx); - break; - case 8: - mov(rax, GetSrc(Op->Header.Args[0].ID())); - imul(GetSrc(Op->Header.Args[1].ID())); - mov(GetDst(Node), rdx); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_UMUL: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - movzx(rcx, GetSrc(Op->Header.Args[1].ID())); - mul(cl); - movzx(rax, al); - mov(GetDst(Node), rax); - break; - case 2: - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - movzx(rcx, GetSrc(Op->Header.Args[1].ID())); - mul(cx); - movzx(rax, ax); - mov(GetDst(Node), rax); - break; - case 4: - mov(rax, GetSrc(Op->Header.Args[0].ID())); - mul(GetSrc(Op->Header.Args[1].ID())); - mov(GetDst(Node), rax); - break; - case 8: - mov(rax, GetSrc(Op->Header.Args[0].ID())); - mul(GetSrc(Op->Header.Args[1].ID())); - mov(GetDst(Node), rax); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_UMULH: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - movzx(rcx, GetSrc(Op->Header.Args[1].ID())); - mul(cl); - movzx(rax, ax); - mov(GetDst(Node), rax); - break; - case 2: - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - movzx(rcx, GetSrc(Op->Header.Args[1].ID())); - mul(cx); - movzx(rax, dx); - mov(GetDst(Node), rax); - break; - case 4: - mov(rax, GetSrc(Op->Header.Args[0].ID())); - mul(GetSrc(Op->Header.Args[1].ID())); - mov(GetDst(Node), rdx); - break; - case 8: - mov(rax, GetSrc(Op->Header.Args[0].ID())); - mul(GetSrc(Op->Header.Args[1].ID())); - mov(GetDst(Node), rdx); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_LDIV: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov(eax, GetSrc(Op->Header.Args[0].ID())); - mov(edx, GetSrc(Op->Header.Args[1].ID())); - mov(ecx, GetSrc(Op->Header.Args[2].ID())); - idiv(ecx); - mov(GetDst(Node), rax); - break; - } - case 8: { - mov(rax, GetSrc(Op->Header.Args[0].ID())); - mov(rdx, GetSrc(Op->Header.Args[1].ID())); - mov(rcx, GetSrc(Op->Header.Args[2].ID())); - idiv(rcx); - mov(GetDst(Node), rax); - break; - } - default: LogMan::Msg::A("Unknown LDIV Size: %d", Size); break; - } - break; - } - case IR::OP_LREM: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov(eax, GetSrc(Op->Header.Args[0].ID())); - mov(edx, GetSrc(Op->Header.Args[1].ID())); - mov(ecx, GetSrc(Op->Header.Args[2].ID())); - idiv(ecx); - mov(GetDst(Node), rdx); - break; - } - - case 8: { - mov(rax, GetSrc(Op->Header.Args[0].ID())); - mov(rdx, GetSrc(Op->Header.Args[1].ID())); - mov(rcx, GetSrc(Op->Header.Args[2].ID())); - idiv(rcx); - mov(GetDst(Node), rdx); - break; - } - default: LogMan::Msg::A("Unknown LREM Size: %d", Size); break; - } - break; - } - case IR::OP_LUDIV: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov (eax, GetSrc(Op->Header.Args[0].ID())); - mov (edx, GetSrc(Op->Header.Args[1].ID())); - mov (ecx, GetSrc(Op->Header.Args[2].ID())); - div(ecx); - mov(GetDst(Node), rax); - break; - } - case 8: { - mov (rax, GetSrc(Op->Header.Args[0].ID())); - mov (rdx, GetSrc(Op->Header.Args[1].ID())); - mov (rcx, GetSrc(Op->Header.Args[2].ID())); - div(rcx); - mov(GetDst(Node), rax); - break; - } - default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; - } - break; - } - case IR::OP_LUREM: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov (eax, GetSrc(Op->Header.Args[0].ID())); - mov (edx, GetSrc(Op->Header.Args[1].ID())); - mov (ecx, GetSrc(Op->Header.Args[2].ID())); - div(ecx); - mov(GetDst(Node), rdx); - break; - } - - case 8: { - mov (rax, GetSrc(Op->Header.Args[0].ID())); - mov (rdx, GetSrc(Op->Header.Args[1].ID())); - mov (rcx, GetSrc(Op->Header.Args[2].ID())); - div(rcx); - mov(GetDst(Node), rdx); - break; - } - default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; - } - break; - } - case IR::OP_LOADFLAG: { - auto Op = IROp->C(); - - auto Dst = GetDst(Node); - movzx(Dst, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]); - and(Dst, 1); - break; - } - case IR::OP_STOREFLAG: { - auto Op = IROp->C(); - - mov (rax, GetSrc(Op->Header.Args[0].ID())); - and(rax, 1); - mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al); - break; - } - case IR::OP_SELECT: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - - mov(rax, GetSrc(Op->Header.Args[0].ID())); - cmp(rax, GetSrc(Op->Header.Args[1].ID())); - - switch (Op->Cond) { - case FEXCore::IR::COND_EQ: - mov(rax, GetSrc(Op->Header.Args[3].ID())); - cmove(rax, GetSrc(Op->Header.Args[2].ID())); - break; - case FEXCore::IR::COND_NEQ: - mov(rax, GetSrc(Op->Header.Args[3].ID())); - cmovne(rax, GetSrc(Op->Header.Args[2].ID())); - break; - case FEXCore::IR::COND_GE: - mov(rax, GetSrc(Op->Header.Args[3].ID())); - cmovge(rax, GetSrc(Op->Header.Args[2].ID())); - break; - case FEXCore::IR::COND_LT: - mov(rax, GetSrc(Op->Header.Args[2].ID())); - cmovae(rax, GetSrc(Op->Header.Args[3].ID())); - break; - case FEXCore::IR::COND_GT: - mov(rax, GetSrc(Op->Header.Args[3].ID())); - cmovg(rax, GetSrc(Op->Header.Args[2].ID())); - break; - case FEXCore::IR::COND_LE: - mov(rax, GetSrc(Op->Header.Args[3].ID())); - cmovle(rax, GetSrc(Op->Header.Args[2].ID())); - break; - case FEXCore::IR::COND_CS: - case FEXCore::IR::COND_CC: - case FEXCore::IR::COND_MI: - case FEXCore::IR::COND_PL: - case FEXCore::IR::COND_VS: - case FEXCore::IR::COND_VC: - case FEXCore::IR::COND_HI: - case FEXCore::IR::COND_LS: - default: - LogMan::Msg::A("Unsupported compare type"); - break; - } - mov (Dst, rax); - break; - } - case IR::OP_LOADMEM: { - auto Op = IROp->C(); - uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); - - auto Dst = GetDst(Node); - mov(rax, Memory); - add(rax, GetSrc(Op->Header.Args[0].ID())); - switch (Op->Size) { - case 1: { - movzx (Dst, byte [rax]); - } - break; - case 2: { - movzx (Dst, word [rax]); - } - break; - case 4: { - mov(Dst, dword [rax]); - } - break; - case 8: { - mov(Dst, qword [rax]); - } - break; - case 16: { - movups(GetDst(Node), xword [rax]); - if (MemoryDebug) { - movq(rcx, GetDst(Node)); - } - } - break; - default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size); - } - break; - } - case IR::OP_STOREMEM: { - auto Op = IROp->C(); - uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); - - mov(rax, Memory); - add(rax, GetSrc(Op->Header.Args[0].ID())); - switch (Op->Size) { - case 1: - mov(byte [rax], GetSrc(Op->Header.Args[1].ID())); - break; - case 2: - mov(word [rax], GetSrc(Op->Header.Args[1].ID())); - break; - case 4: - mov(dword [rax], GetSrc(Op->Header.Args[1].ID())); - break; - case 8: - mov(qword [rax], GetSrc(Op->Header.Args[1].ID())); - break; - case 16: - movups(xword [rax], GetSrc(Op->Header.Args[1].ID())); - break; - default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size); - } - break; - } - case IR::OP_SYSCALL: { - auto Op = IROp->C(); - // XXX: This is very terrible, but I don't care for right now - - push(rdi); - const std::array RA64 = { rsi, r8, r9, r10, r11, rbx, rbp, r12, r13, r14, r15 }; - for (auto &Reg : RA64) - push(Reg); - - // Syscall ABI for x86-64 - // this: rdi - // Thread: rsi - // ArgPointer: rdx (Stack) - // - // Result: RAX - - // These are pushed in reverse order because stacks - for (uint32_t i = 7; i > 0; --i) - push(GetSrc(Op->Header.Args[i - 1].ID())); - - mov(rsi, rdi); // Move thread in to rsi - mov(rdi, reinterpret_cast(&CTX->SyscallHandler)); - mov(rdx, rsp); - - using PtrType = uint64_t (FEXCore::SyscallHandler::*)(FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args); - union { - PtrType ptr; - uint64_t Raw; - } PtrCast; - PtrCast.ptr = &FEXCore::SyscallHandler::HandleSyscall; - mov(rax, PtrCast.Raw); - call(rax); - - // Reload arguments just in case they are sill live after the fact - for (uint32_t i = 0; i < 7; ++i) - pop(GetSrc(Op->Header.Args[i].ID())); - - for (uint32_t i = RA64.size(); i > 0; --i) - pop(RA64[i - 1]); - - pop(rdi); - - mov (GetDst(Node), rax); - break; - } - case IR::OP_CPUID: { - auto Op = IROp->C(); - using ClassPtrType = FEXCore::CPUIDEmu::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t Function); - union { - ClassPtrType ClassPtr; - uint64_t Raw; - } Ptr; - Ptr.ClassPtr = &CPUIDEmu::RunFunction; - - const std::array RA64 = { rsi, r8, r9, r10, r11, rbx, rbp, r12, r13, r14, r15 }; - for (auto &Reg : RA64) - push(Reg); - - // CPUID ABI - // this: rdi - // Function: rsi - // - // Result: RAX, RDX. 4xi32 - push(rdi); - mov (rsi, GetSrc(Op->Header.Args[0].ID())); - mov (rdi, reinterpret_cast(&CTX->CPUID)); - - sub(rsp, 8); // Align - - mov(rax, Ptr.Raw); - call(rax); - - add(rsp, 8); // Align - - pop(rdi); - - for (uint32_t i = RA64.size(); i > 0; --i) - pop(RA64[i - 1]); - - auto Dst = GetDst(Node); - pinsrq(Dst, rax, 0); - pinsrd(Dst, rdx, 1); - break; - } - case IR::OP_EXTRACTELEMENT: { - auto Op = IROp->C(); - - uint32_t PhysReg = RA::GetNodeRegister(Graph, Op->Header.Args[0].ID()); - if (PhysReg >= XMMBase) { - switch (OpSize) { - case 1: - pextrb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); - break; - case 2: - pextrw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); - break; - case 4: - pextrd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); - break; - case 8: - pextrq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); - break; - default: LogMan::Msg::A("Unhandled ExtractElementSize: %d", OpSize); - } - } - else { - LogMan::Msg::A("Can't handle extract from GPR yet"); - } - break; - } - case IR::OP_VINSELEMENT: { - auto Op = IROp->C(); - movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); - - // Dst_d[Op->DestIdx] = Src2_d[Op->SrcIdx]; - - // pextrq reg64/mem64, xmm, imm - // pinsrq xmm, reg64/mem64, imm8 - switch (Op->ElementSize) { - case 1: { - pextrb(al, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); - pinsrb(xmm15, al, Op->DestIdx); - break; - } - case 2: { - pextrw(ax, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); - pinsrw(xmm15, ax, Op->DestIdx); - break; - } - case 4: { - pextrd(eax, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); - pinsrd(xmm15, eax, Op->DestIdx); - break; - } - case 8: { - pextrq(rax, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); - pinsrq(xmm15, rax, Op->DestIdx); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - - movapd(GetDst(Node), xmm15); - break; - } - case IR::OP_VADD: { - auto Op = IROp->C(); - switch (Op->ElementSize) { - case 1: { - vpaddb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case 2: { - vpaddw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case 4: { - vpaddd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case 8: { - vpaddq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - break; - } - case IR::OP_VSUB: { - auto Op = IROp->C(); - switch (Op->ElementSize) { - case 1: { - vpsubb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case 2: { - vpsubw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case 4: { - vpsubd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case 8: { - vpsubq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - break; - } - case IR::OP_VXOR: { - auto Op = IROp->C(); - vpxor(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case IR::OP_VOR: { - auto Op = IROp->C(); - vpor(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case IR::OP_VCMPEQ: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); - - switch (Op->ElementSize) { - case 1: - vpcmpeqb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - case 2: - vpcmpeqw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - case 4: - vpcmpeqd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - case 8: - vpcmpeqq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); - } - break; - } - case IR::OP_VCMPGT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); - - switch (Op->ElementSize) { - case 1: - vpcmpgtb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - case 2: - vpcmpgtw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - case 4: - vpcmpgtd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - case 8: - vpcmpgtq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); - } - break; - } - case IR::OP_VZIP: { - auto Op = IROp->C(); - movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); - - switch (Op->ElementSize) { - case 1: { - punpcklbw(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - case 2: { - punpcklwd(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - case 4: { - punpckldq(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - case 8: { - punpcklqdq(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movapd(GetDst(Node), xmm15); - break; - } - case IR::OP_VZIP2: { - auto Op = IROp->C(); - movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); - - switch (Op->ElementSize) { - case 1: { - punpckhbw(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - case 2: { - punpckhwd(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - case 4: { - punpckhdq(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - case 8: { - punpckhqdq(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movapd(GetDst(Node), xmm15); - break; - } - case IR::OP_VUSHLS: { - auto Op = IROp->C(); - movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); - vmovq(xmm14, Reg64(GetSrc(Op->Header.Args[1].ID()).getIdx())); - - switch (Op->ElementSize) { - case 2: { - psllw(xmm15, xmm14); - break; - } - case 4: { - pslld(xmm15, xmm14); - break; - } - case 8: { - psllq(xmm15, xmm14); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movapd(GetDst(Node), xmm15); - - break; - } - case IR::OP_CAS: { - auto Op = IROp->C(); - // Args[0]: Desired - // Args[1]: Expected - // Args[2]: Pointer - // DataSrc = *Src1 - // if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc - // This will write to memory! Careful! - // Third operand must be a calculated guest memory address - //OrderedNode *CASResult = _CAS(Src3, Src2, Src1); - uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); - - mov(rcx, Memory); - add(rcx, GetSrc(Op->Header.Args[2].ID())); - mov(rdx, GetSrc(Op->Header.Args[1].ID())); - mov(rax, GetSrc(Op->Header.Args[0].ID())); - - // RCX now contains pointer - // RAX contains our expected value - // RDX contains our desired - - lock(); - - switch (OpSize) { - case 1: { - cmpxchg(byte [rcx], dl); - movzx(rax, al); - break; - } - case 2: { - cmpxchg(word [rcx], dx); - movzx(rax, ax); - break; - } - case 4: { - cmpxchg(dword [rcx], edx); - break; - } - case 8: { - cmpxchg(qword [rcx], rdx); - break; - } - default: LogMan::Msg::A("Unsupported: %d", OpSize); - } - - // RAX now contains the result - mov (GetDst(Node), rax); - break; - } - case IR::OP_CYCLECOUNTER: { -#ifdef DEBUG_CYCLES - mov (GetDst(Node), 0); -#else - rdtsc(); - shl(rdx, 32); - or(rax, rdx); - mov (GetDst(Node), rax); -#endif - break; - } - case IR::OP_FINDLSB: { - auto Op = IROp->C(); - tzcnt(rcx, GetSrc(Op->Header.Args[0].ID())); - xor(rax, rax); - cmp(GetSrc(Op->Header.Args[0].ID()), 1); - sbb(rax, rax); - or(rax, rcx); - mov (GetDst(Node), rax); - break; - } - case IR::OP_FINDMSB: { - auto Op = IROp->C(); - mov(rax, OpSize * 8); - lzcnt(rcx, GetSrc(Op->Header.Args[0].ID())); - sub(rax, rcx); - mov (GetDst(Node), rax); - break; - } - default: break; - } - } - else { - switch (IROp->Op) { - case IR::OP_LOADCONTEXT: { - auto Op = IROp->C(); -#define LOAD_CTX(x, y) \ - case x: { \ - movzx(rax, y [STATE + Op->Offset]); \ - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); \ - } \ - break - switch (Op->Size) { - LOAD_CTX(1, byte); - LOAD_CTX(2, word); - case 4: { - mov(eax, dword [STATE + Op->Offset]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - } - break; - case 8: { - mov(rax, qword [STATE + Op->Offset]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - } - break; - case 16: { - if (Op->Offset % 16 == 0) { - movaps(xmm0, xword [STATE + Op->Offset]); - movaps(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - } - else { - movups(xmm0, xword [STATE + Op->Offset]); - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - } - } - break; - default: LogMan::Msg::A("Unhandled LoadContext size: %d", Op->Size); - } -#undef LOAD_CTX - break; - } - case IR::OP_STORECONTEXT: { - auto Op = IROp->C(); - - switch (Op->Size) { - case 1: { - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov(byte [STATE + Op->Offset], al); - } - break; - - case 2: { - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov(word [STATE + Op->Offset], ax); - } - break; - case 4: { - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov(dword [STATE + Op->Offset], eax); - } - break; - case 8: { - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov(qword [STATE + Op->Offset], rax); - } - break; - case 16: { - if (Op->Offset % 16 == 0) { - movaps(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movaps(xword [STATE + Op->Offset], xmm0); - } - else { - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xword [STATE + Op->Offset], xmm0); - } - } - break; - default: LogMan::Msg::A("Unhandled StoreContext size: %d", Op->Size); - } - - break; - } - - case IR::OP_SYSCALL: { - auto Op = IROp->C(); - - push(rdi); - push(r11); - - // Syscall ABI for x86-64 - // this: rdi - // Thread: rsi - // ArgPointer: rdx (Stack) - // - // Result: RAX - - mov(rsi, rdi); // Move thread in to rsi - mov(rdi, reinterpret_cast(&CTX->SyscallHandler)); - - // These are pushed in reverse order because stacks - push(qword [TEMP_STACK + (Op->Header.Args[6].ID() * 16)]); - push(qword [TEMP_STACK + (Op->Header.Args[5].ID() * 16)]); - push(qword [TEMP_STACK + (Op->Header.Args[4].ID() * 16)]); - push(qword [TEMP_STACK + (Op->Header.Args[3].ID() * 16)]); - push(qword [TEMP_STACK + (Op->Header.Args[2].ID() * 16)]); - push(qword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - push(qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov (rdx, rsp); - - using PtrType = uint64_t (FEXCore::SyscallHandler::*)(FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args); - union { - PtrType ptr; - uint64_t Raw; - } PtrCast; - PtrCast.ptr = &FEXCore::SyscallHandler::HandleSyscall; - mov(rax, PtrCast.Raw); - call(rax); - add(rsp, 7 * 8); - - pop(r11); - pop(rdi); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_CPUID: { - auto Op = IROp->C(); - using ClassPtrType = FEXCore::CPUIDEmu::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t Function); - union { - ClassPtrType ClassPtr; - uint64_t Raw; - } Ptr; - Ptr.ClassPtr = &CPUIDEmu::RunFunction; - - - // CPUID ABI - // this: rdi - // Function: rsi - // - // Result: RAX, RDX. 4xi32 - push(rdi); - push(r11); - mov (rsi, qword [TEMP_STACK + (Op->Header.Args[0].ID() *16)]); - mov (rdi, reinterpret_cast(&CTX->CPUID)); - - push(rax); // align - - mov(rax, Ptr.Raw); - call(rax); - - pop(r11); // align - - pop(r11); - pop(rdi); - - mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 0], eax); - shr(rax, 32); - mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 4], eax); - - mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 8], edx); - shr(rdx, 32); - mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 12], edx); - break; - } - case IR::OP_EXTRACTELEMENT: { - auto Op = IROp->C(); - - uint32_t Offset = Op->Header.Args[0].ID() * 16 + OpSize * Op->Idx; - switch (OpSize) { - case 1: - movzx(rax, byte [TEMP_STACK + Offset]); - break; - case 2: - movzx(rax, word [TEMP_STACK + Offset]); - break; - case 4: - mov(eax, dword [TEMP_STACK + Offset]); - break; - case 8: - mov(rax, qword [TEMP_STACK + Offset]); - break; - default: LogMan::Msg::A("Unhandled ExtractElementSize: %d", OpSize); - } - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_LOADFLAG: { - auto Op = IROp->C(); - - movzx(rax, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]); - and(rax, 1); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_STOREFLAG: { - auto Op = IROp->C(); - - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - and(rax, 1); - mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al); - break; - } - - case IR::OP_CONDJUMP: { - auto Op = IROp->C(); - - Label *TargetLabel; - auto IsTarget = JumpTargets.find(Op->Header.Args[1].ID()); - if (IsTarget == JumpTargets.end()) { - TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[1].ID(), Label{}).first->second; - } - else { - TargetLabel = &IsTarget->second; - } - - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - cmp(rax, 0); - jne(*TargetLabel); - break; - } - - case IR::OP_LOADMEM: { - auto Op = IROp->C(); - uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); - - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov(rcx, Memory); - add(rax, rcx); - switch (Op->Size) { - case 1: { - movzx (rcx, byte [rax]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); - } - break; - case 2: { - movzx (rcx, word [rax]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); - } - break; - case 4: { - mov(ecx, dword [rax]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); - } - break; - case 8: { - mov(rcx, qword [rax]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); - } - break; - case 16: { - movups(xmm0, xword [rax]); - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - if (MemoryDebug) { - movq(rcx, xmm0); - } - } - break; - default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size); - } - - if (MemoryDebug) { - push(rdi); - push(r11); - sub(rsp, 8); - - // Load the address in to Arg1 - mov(rdi, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - // Move the loaded value to Arg2 - mov(rsi, rcx); - mov (rdx, Op->Size); - - mov(rax, reinterpret_cast(LoadMem)); - call(rax); - - add(rsp, 8); - - pop(r11); - pop(rdi); - } - - break; - } - case IR::OP_STOREMEM: { - auto Op = IROp->C(); - uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); - - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov(rcx, Memory); - add(rax, rcx); - switch (Op->Size) { - case 1: { - mov(cl, byte [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - mov(byte [rax], cl); - if (MemoryDebug) { - movzx(rcx, cl); - } - } - break; - case 2: { - mov(cx, word [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - mov(word [rax], cx); - - if (MemoryDebug) { - movzx(rcx, cx); - } - } - break; - case 4: { - mov(ecx, dword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - mov(dword [rax], ecx); - } - break; - case 8: { - mov(rcx, qword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - mov(qword [rax], rcx); - } - break; - case 16: { - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - movups(xword [rax], xmm0); - if (MemoryDebug) { - movq(rcx, xmm0); - } - } - break; - default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size); - } - - if (MemoryDebug) { - push(rdi); - push(r11); - sub(rsp, 8); - - // Load the address in to Arg1 - mov(rdi, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - // Load the value from RAX in to Arg2 - mov(rsi, rcx); - - mov (rdx, Op->Size); - - mov(rax, reinterpret_cast(StoreMem)); - call(rax); - - add(rsp, 8); - - pop(r11); - pop(rdi); - } - break; - } - case IR::OP_MOV: { - auto Op = IROp->C(); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_CONSTANT: { - auto Op = IROp->C(); - if (Op->Constant >> 31) { - mov(rax, Op->Constant); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - } - else { - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], Op->Constant); - } - break; - } - case IR::OP_POPCOUNT: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movzx(al, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - popcnt(eax, eax); - break; - case 2: - popcnt(ax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movzx(rax, ax); - break; - case 4: - popcnt(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - break; - case 8: - popcnt(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - break; - } - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_ADD: { - auto Op = IROp->C(); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - add(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_SUB: { - auto Op = IROp->C(); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - sub(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_XOR: { - auto Op = IROp->C(); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - xor(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_AND: { - auto Op = IROp->C(); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - and(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_OR: { - auto Op = IROp->C(); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - or(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_MUL: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movsx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - imul(cl); - movsx(rax, al); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 2: - movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movsx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - imul(cx); - movsx(rax, ax); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 4: - movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - imul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - movsxd(rax, eax); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 8: - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - imul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_UMUL: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movzx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mul(cl); - movzx(rax, al); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 2: - movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movzx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mul(cx); - movzx(rax, ax); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 4: - mov(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 8: - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - - case IR::OP_MULH: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movsx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - imul(cl); - movsx(rax, ax); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 2: - movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movsx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - imul(cx); - movsx(rax, dx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 4: - movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - imul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - movsxd(rax, edx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 8: - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - imul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_UMULH: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movzx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mul(cl); - movzx(rax, ax); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 2: - movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movzx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mul(cx); - movzx(rax, dx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 4: - mov(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - case 8: - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_LDIV: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - idiv(ecx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - - case 8: { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - idiv(rcx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - default: LogMan::Msg::A("Unknown LDIV Size: %d", Size); break; - } - break; - } - case IR::OP_LREM: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - idiv(ecx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - } - - case 8: { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - idiv(rcx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - } - default: LogMan::Msg::A("Unknown LREM Size: %d", Size); break; - } - break; - } - - - case IR::OP_LUDIV: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - div(ecx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - - case 8: { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - div(rcx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; - } - break; - } - case IR::OP_LUREM: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - div(ecx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - } - - case 8: { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - div(rcx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - } - default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; - } - break; - } - - case IR::OP_ZEXT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - - if (Op->SrcSize == 64) { - movd(xmm0, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movups(qword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - } - else { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(rcx, uint64_t((1ULL << Op->SrcSize) - 1)); - and(rax, rcx); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - } - break; - } - - case IR::OP_SEXT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - switch (Op->SrcSize / 8) { - case 1: - movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 2: - movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 4: - movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 8: - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", Op->SrcSize / 8); - } - break; - } - case IR::OP_BFI: { - auto Op = IROp->C(); - LogMan::Throw::A(OpSize <= 8, "OpSize is too large for BFI: %d", OpSize); - - uint64_t SourceMask = (1ULL << Op->Width) - 1; - - uint64_t DestMask = ~(SourceMask << Op->lsb); - - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - - if (Op->Width != 64) { - mov(rdx, SourceMask); - and(rcx, rdx); - } - - mov(rdx, DestMask); - and(rax, rdx); - shl(rdx, Op->lsb); - or(rax, rdx); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - - break; - } - - case IR::OP_BFE: { - auto Op = IROp->C(); - LogMan::Throw::A(OpSize <= 16, "OpSize is too large for BFE: %d", OpSize); - // %ssa64 i128 = Bfe %ssa48 i128, 0x1, 0x7 - if (OpSize == 16) { - LogMan::Throw::A(!(Op->lsb < 64 && (Op->lsb + Op->Width > 64)), "Trying to BFE an XMM across the 64bit split: Beginning at %d, ending at %d", Op->lsb, Op->lsb + Op->Width); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - uint8_t Offset = Op->lsb; - if (Offset < 64) { - pextrq(rax, xmm0, 0); - } - else { - pextrq(rax, xmm0, 1); - Offset -= 64; - } - - if (Offset) { - shr(rax, Offset); - } - - if (Op->Width != 64) { - mov(rcx, uint64_t((1ULL << Op->Width) - 1)); - and(rax, rcx); - } - - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - } - else { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - if (Op->lsb != 0) - shr(rax, Op->lsb); - - if (Op->Width != 64) { - mov(rcx, uint64_t((1ULL << Op->Width) - 1)); - and(rax, rcx); - } - - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - } - break; - } - case IR::OP_FINDLSB: { - auto Op = IROp->C(); - tzcnt(rcx, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - xor(rax, rax); - cmp(qword [TEMP_STACK + Op->Header.Args[0].ID() * 16], 1); - sbb(rax, rax); - or(rax, rcx); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - - break; - } - case IR::OP_FINDMSB: { - auto Op = IROp->C(); - mov(rax, OpSize * 8); - lzcnt(rcx, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - sub(rax, rcx); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - - case IR::OP_LSHR: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - and(rcx, Mask); - shrx(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16], rcx); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_LSHL: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - and(rcx, Mask); - shlx(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16], rcx); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_ASHR: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - and(rcx, Mask); - switch (OpSize) { - case 1: - movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - sar(al, cl); - break; - case 2: - movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - sar(ax, cl); - break; - case 4: - movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - sar(eax, cl); - break; - case 8: - mov(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - sar(rax, cl); - break; - default: LogMan::Msg::A("Unknown ASHR Size: %d\n", OpSize); break; - }; - - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - - case IR::OP_ROL: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - and(rcx, Mask); - switch (OpSize) { - case 1: { - movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - rol(al, cl); - break; - } - case 2: { - movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - rol(ax, cl); - break; - } - case 4: { - mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - rol(eax, cl); - break; - } - case 8: { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - rol(rax, cl); - break; - } - } - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_ROR: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - and(rcx, Mask); - switch (OpSize) { - case 1: { - movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - ror(al, cl); - break; - } - case 2: { - movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - ror(ax, cl); - break; - } - case 4: { - mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - ror(eax, cl); - break; - } - case 8: { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - ror(rax, cl); - break; - } - } - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - - case IR::OP_SELECT: { - auto Op = IROp->C(); - - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - cmp(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - - switch (Op->Cond) { - case FEXCore::IR::COND_EQ: - mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); - cmove(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - break; - case FEXCore::IR::COND_NEQ: - mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); - cmovne(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - break; - case FEXCore::IR::COND_GE: - mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); - cmovge(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - break; - case FEXCore::IR::COND_LT: - mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - cmovae(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); - break; - case FEXCore::IR::COND_GT: - mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); - cmovg(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - break; - case FEXCore::IR::COND_LE: - mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); - cmovle(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - break; - case FEXCore::IR::COND_CS: - case FEXCore::IR::COND_CC: - case FEXCore::IR::COND_MI: - case FEXCore::IR::COND_PL: - case FEXCore::IR::COND_VS: - case FEXCore::IR::COND_VC: - case FEXCore::IR::COND_HI: - case FEXCore::IR::COND_LS: - default: - LogMan::Msg::A("Unsupported compare type"); - break; - } - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); - break; - } - - case IR::OP_CAS: { - auto Op = IROp->C(); - // Args[0]: Expected - // Args[1]: Desired - // Args[2]: Pointer - // DataSrc = *Src1 - // if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc - // This will write to memory! Careful! - // Third operand must be a calculated guest memory address - //OrderedNode *CASResult = _CAS(Src3, Src2, Src1); - uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); - - mov(rcx, Memory); - add(rcx, qword [TEMP_STACK + (Op->Header.Args[2].ID() * 16)]); - - mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - - // RCX now contains pointer - // RAX contains our expected value - // RDX contains our desired - - lock(); - - switch (OpSize) { - case 1: { - cmpxchg(byte [rcx], dl); - movzx(rax, al); - break; - } - case 2: { - cmpxchg(word [rcx], dx); - movzx(rax, ax); - break; - } - case 4: { - cmpxchg(dword [rcx], edx); - break; - } - case 8: { - cmpxchg(qword [rcx], rdx); - break; - } - default: LogMan::Msg::A("Unsupported: %d", OpSize); - } - - // RAX now contains the result - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_VCMPEQ: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - - LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); - - switch (Op->ElementSize) { - case 1: { - pcmpeqb(xmm0, xmm1); - break; - } - case 2: { - pcmpeqw(xmm0, xmm1); - break; - } - case 4: { - pcmpeqd(xmm0, xmm1); - break; - } - case 8: { - pcmpeqq(xmm0, xmm1); - break; - } - - default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); - } - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - break; - } - case IR::OP_VCMPGT: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - - LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); - - switch (Op->ElementSize) { - case 1: { - pcmpgtb(xmm0, xmm1); - break; - } - case 2: { - pcmpgtw(xmm0, xmm1); - break; - } - case 4: { - pcmpgtd(xmm0, xmm1); - break; - } - case 8: { - pcmpgtq(xmm0, xmm1); - break; - } - - default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); - } - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - break; - } - - case IR::OP_VXOR: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - pxor(xmm0, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - break; - } - case IR::OP_VOR: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - por(xmm0, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - break; - } - - case IR::OP_VINSELEMENT: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - - // Dst_d[Op->DestIdx] = Src2_d[Op->SrcIdx]; - - // pextrq reg64/mem64, xmm, imm - // pinsrq xmm, reg64/mem64, imm8 - switch (Op->ElementSize) { - case 1: { - pextrb(al, xmm1, Op->SrcIdx); - pinsrb(xmm0, al, Op->DestIdx); - break; - } - case 2: { - pextrw(ax, xmm1, Op->SrcIdx); - pinsrw(xmm0, ax, Op->DestIdx); - break; - } - case 4: { - pextrd(eax, xmm1, Op->SrcIdx); - pinsrd(xmm0, eax, Op->DestIdx); - break; - } - case 8: { - pextrq(rax, xmm1, Op->SrcIdx); - pinsrq(xmm0, rax, Op->DestIdx); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - break; - } - case IR::OP_VADD: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - switch (Op->ElementSize) { - case 1: { - paddb(xmm0, xmm1); - break; - } - case 2: { - paddw(xmm0, xmm1); - break; - } - case 4: { - paddd(xmm0, xmm1); - break; - } - case 8: { - paddq(xmm0, xmm1); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - - break; - } - - case IR::OP_VUSHLS: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - - switch (Op->ElementSize) { - case 2: { - psllw(xmm0, xmm1); - break; - } - case 4: { - pslld(xmm0, xmm1); - break; - } - case 8: { - psllq(xmm0, xmm1); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - - break; - } - case IR::OP_VZIP: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - switch (Op->ElementSize) { - case 1: { - punpcklbw(xmm0, xmm1); - break; - } - case 2: { - punpcklwd(xmm0, xmm1); - break; - } - case 4: { - punpckldq(xmm0, xmm1); - break; - } - case 8: { - punpcklqdq(xmm0, xmm1); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - - break; - } - case IR::OP_VZIP2: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - switch (Op->ElementSize) { - case 1: { - punpckhbw(xmm0, xmm1); - break; - } - case 2: { - punpckhwd(xmm0, xmm1); - break; - } - case 4: { - punpckhdq(xmm0, xmm1); - break; - } - case 8: { - punpckhqdq(xmm0, xmm1); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - - break; - } - - case IR::OP_CYCLECOUNTER: { -#ifdef DEBUG_CYCLES - mov (rax, 0); -#else - rdtsc(); - shl(rdx, 32); - or(rax, rdx); -#endif - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - default: break; - } + if (SpillSlots) { + sub(rsp, SpillSlots * 16); } - ++Begin; + auto HeaderIterator = CurrentIR->begin(); + IR::OrderedNodeWrapper *HeaderNodeWrapper = HeaderIterator(); + IR::OrderedNode *HeaderNode = HeaderNodeWrapper->GetNode(ListBegin); + auto HeaderOp = HeaderNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == IR::OP_IRHEADER, "First op wasn't IRHeader"); + + IR::OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); + while (1) { + using namespace FEXCore::IR; + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block"); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR->at(BlockIROp->Begin); + auto CodeLast = CurrentIR->at(BlockIROp->Last); + + while (1) { + OrderedNodeWrapper *WrapperOp = CodeBegin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); + FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + uint8_t OpSize = IROp->Size; + uint32_t Node = WrapperOp->ID(); + + if (HasRA) { + #ifdef DEBUG_RA + if (IROp->Op != IR::OP_BEGINBLOCK && + IROp->Op != IR::OP_CONDJUMP && + IROp->Op != IR::OP_JUMP) { + std::stringstream Inst; + auto Name = FEXCore::IR::GetName(IROp->Op); + + if (IROp->HasDest) { + uint32_t PhysReg = RAPass->GetNodeRegister(Node); + if (PhysReg >= XMMBase) + Inst << "\tXMM" << GetPhys(Node) << " = " << Name << " "; + else + Inst << "\tReg" << GetPhys(Node) << " = " << Name << " "; + } + else { + Inst << "\t" << Name << " "; + } + + for (uint8_t i = 0; i < IROp->NumArgs; ++i) { + uint32_t ArgNode = IROp->Args[i].ID(); + uint32_t PhysReg = RAPass->GetNodeRegister(ArgNode); + if (PhysReg >= XMMBase) + Inst << "XMM" << GetPhys(ArgNode) << (i + 1 == IROp->NumArgs ? "" : ", "); + else + Inst << "Reg" << GetPhys(ArgNode) << (i + 1 == IROp->NumArgs ? "" : ", "); + } + + LogMan::Msg::D("%s", Inst.str().c_str()); + } + #endif + } + + switch (IROp->Op) { + case IR::OP_BEGINBLOCK: { + auto IsTarget = JumpTargets.find(WrapperOp->ID()); + if (IsTarget == JumpTargets.end()) { + JumpTargets.try_emplace(WrapperOp->ID()); + } + else { + L(IsTarget->second); + } + break; + } + case IR::OP_ENDBLOCK: { + auto Op = IROp->C(); + if (Op->RIPIncrement) { + add(qword [STATE + offsetof(FEXCore::Core::CPUState, rip)], Op->RIPIncrement); + } + break; + } + case IR::OP_EXITFUNCTION: + case IR::OP_ENDFUNCTION: { + if (SpillSlots) { + add(rsp, SpillSlots * 16); + } + + if (HasRA) { + pop(r15); + pop(r14); + pop(r13); + pop(r12); + pop(rbp); + pop(rbx); + } + + ret(); + break; + } + case IR::OP_BREAK: { + auto Op = IROp->C(); + switch (Op->Reason) { + case 4: // HLT + ud2(); + break; + default: LogMan::Msg::A("Unknown Break reason: %d", Op->Reason); + } + break; + } + case IR::OP_JUMP: { + auto Op = IROp->C(); + + Label *TargetLabel; + auto IsTarget = JumpTargets.find(Op->Header.Args[0].ID()); + if (IsTarget == JumpTargets.end()) { + TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[0].ID()).first->second; + } + else { + TargetLabel = &IsTarget->second; + } + + jmp(*TargetLabel, T_NEAR); + break; + } + default: break; + } + + if (HasRA) { + switch (IROp->Op) { + case IR::OP_CONDJUMP: { + auto Op = IROp->C(); + + Label *TargetLabel; + auto IsTarget = JumpTargets.find(Op->Header.Args[1].ID()); + if (IsTarget == JumpTargets.end()) { + TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[1].ID()).first->second; + } + else { + TargetLabel = &IsTarget->second; + } + + cmp(GetSrc(Op->Header.Args[0].ID()), 0); + jne(*TargetLabel, T_NEAR); + break; + } + case IR::OP_LOADCONTEXT: { + auto Op = IROp->C(); + switch (Op->Size) { + case 1: { + mov(GetDst(Node), byte [STATE + Op->Offset]); + } + break; + case 2: { + mov(GetDst(Node), word [STATE + Op->Offset]); + } + break; + case 4: { + mov(GetDst(Node), dword [STATE + Op->Offset]); + } + break; + case 8: { + mov(GetDst(Node), qword [STATE + Op->Offset]); + } + break; + case 16: { + if (Op->Offset % 16 == 0) + movaps(GetDst(Node), xword [STATE + Op->Offset]); + else + movups(GetDst(Node), xword [STATE + Op->Offset]); + } + break; + default: LogMan::Msg::A("Unhandled LoadContext size: %d", Op->Size); + } + break; + } + case IR::OP_STORECONTEXT: { + auto Op = IROp->C(); + + switch (Op->Size) { + case 1: { + mov(byte [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); + } + break; + + case 2: { + mov(word [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 4: { + mov(dword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 8: { + mov(qword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 16: { + if (Op->Offset % 16 == 0) + movaps(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); + else + movups(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); + } + break; + default: LogMan::Msg::A("Unhandled StoreContext size: %d", Op->Size); + } + break; + } + case IR::OP_FILLREGISTER: { + auto Op = IROp->C(); + uint32_t SlotOffset = Op->Slot * 16; + switch (OpSize) { + case 1: { + mov(GetDst(Node), byte [rsp + SlotOffset]); + } + break; + case 2: { + mov(GetDst(Node), word [rsp + SlotOffset]); + } + break; + case 4: { + mov(GetDst(Node), dword [rsp + SlotOffset]); + } + break; + case 8: { + mov(GetDst(Node), qword [rsp + SlotOffset]); + } + break; + case 16: { + movaps(GetDst(Node), xword [rsp + SlotOffset]); + } + break; + default: LogMan::Msg::A("Unhandled FillRegister size: %d", OpSize); + } + break; + } + case IR::OP_SPILLREGISTER: { + auto Op = IROp->C(); + uint32_t SlotOffset = Op->Slot * 16; + switch (OpSize) { + case 1: { + mov(byte [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 2: { + mov(word [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 4: { + mov(dword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 8: { + mov(qword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 16: { + movaps(xword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID())); + } + break; + default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize); + } + break; + } + case IR::OP_ADD: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + mov(rax, GetSrc(Op->Header.Args[1].ID())); + add(rax, GetSrc(Op->Header.Args[0].ID())); + mov(Dst, rax); + break; + } + case IR::OP_SUB: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + mov(rax, GetSrc(Op->Header.Args[0].ID())); + sub(rax, GetSrc(Op->Header.Args[1].ID())); + mov(Dst, rax); + break; + } + case IR::OP_XOR: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + mov(rax, GetSrc(Op->Header.Args[1].ID())); + xor(rax, GetSrc(Op->Header.Args[0].ID())); + mov(Dst, rax); + break; + } + case IR::OP_AND: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + mov(rax, GetSrc(Op->Header.Args[1].ID())); + and(rax, GetSrc(Op->Header.Args[0].ID())); + mov(Dst, rax); + break; + } + case IR::OP_OR: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + mov(rax, GetSrc(Op->Header.Args[1].ID())); + or (rax, GetSrc(Op->Header.Args[0].ID())); + mov(Dst, rax); + break; + } + case IR::OP_MOV: { + auto Op = IROp->C(); + mov (GetDst(Node), GetSrc(Op->Header.Args[0].ID())); + break; + } + case IR::OP_CONSTANT: { + auto Op = IROp->C(); + mov(GetDst(Node), Op->Constant); + break; + } + case IR::OP_POPCOUNT: { + auto Op = IROp->C(); + auto Dst64 = GetDst(Node); + + switch (OpSize) { + case 1: + movzx(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); + popcnt(Dst64, Dst64); + break; + case 2: { + movzx(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); + popcnt(Dst64, Dst64); + break; + } + case 4: + popcnt(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); + break; + case 8: + popcnt(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); + break; + } + break; + } + case IR::OP_ZEXT: { + auto Op = IROp->C(); + LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); + + uint32_t PhysReg = RAPass->GetNodeRegister(Op->Header.Args[0].ID()); + if (PhysReg >= XMMBase) { + // XMM -> GPR transfer with free truncation + switch (Op->SrcSize) { + case 8: + pextrb(al, GetSrc(Op->Header.Args[0].ID()), 0); + break; + case 16: + pextrw(ax, GetSrc(Op->Header.Args[0].ID()), 0); + break; + case 32: + pextrd(eax, GetSrc(Op->Header.Args[0].ID()), 0); + break; + case 64: + pextrw(rax, GetSrc(Op->Header.Args[0].ID()), 0); + break; + default: LogMan::Msg::A("Unhandled Zext size: %d", Op->SrcSize); break; + } + auto Dst = GetDst(Node); + mov(Dst, rax); + } + else { + if (Op->SrcSize == 64) { + vmovq(xmm15, Reg64(GetSrc(Op->Header.Args[0].ID()).getIdx())); + movapd(GetDst(Node), xmm15); + } + else { + auto Dst = GetDst(Node); + mov(rax, uint64_t((1ULL << Op->SrcSize) - 1)); + and(rax, GetSrc(Op->Header.Args[0].ID())); + mov(Dst, rax); + } + } + break; + } + case IR::OP_SEXT: { + auto Op = IROp->C(); + LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); + auto Dst = GetDst(Node); + + switch (Op->SrcSize / 8) { + case 1: + movsx(Dst, GetSrc(Op->Header.Args[0].ID())); + break; + case 2: + movsx(Dst, GetSrc(Op->Header.Args[0].ID())); + break; + case 4: + movsxd(Reg64(Dst.getIdx()), GetSrc(Op->Header.Args[0].ID())); + break; + case 8: + mov(Dst, GetSrc(Op->Header.Args[0].ID())); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", Op->SrcSize / 8); + } + break; + } + case IR::OP_BFE: { + auto Op = IROp->C(); + LogMan::Throw::A(OpSize <= 16, "OpSize is too large for BFE: %d", OpSize); + if (OpSize == 16) { + LogMan::Throw::A(!(Op->lsb < 64 && (Op->lsb + Op->Width > 64)), "Trying to BFE an XMM across the 64bit split: Beginning at %d, ending at %d", Op->lsb, Op->lsb + Op->Width); + movups(xmm15, GetSrc(Op->Header.Args[0].ID())); + uint8_t Offset = Op->lsb; + if (Offset < 64) { + pextrq(rax, xmm15, 0); + } + else { + pextrq(rax, xmm15, 1); + Offset -= 64; + } + + if (Offset) { + shr(rax, Offset); + } + + if (Op->Width != 64) { + mov(rcx, uint64_t((1ULL << Op->Width) - 1)); + and(rax, rcx); + } + + mov (GetDst(Node), rax); + } + else { + auto Dst = GetDst(Node); + mov(rax, GetSrc(Op->Header.Args[0].ID())); + + if (Op->lsb != 0) + shr(rax, Op->lsb); + + if (Op->Width != 64) { + mov(rcx, uint64_t((1ULL << Op->Width) - 1)); + and(rax, rcx); + } + mov(Dst, rax); + } + break; + } + case IR::OP_LSHR: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + auto Dst = GetDst(Node); + mov (rcx, GetSrc(Op->Header.Args[1].ID())); + and(rcx, Mask); + + shrx(Reg32e(Dst.getIdx(), 64), GetSrc(Op->Header.Args[0].ID()), rcx); + break; + } + case IR::OP_LSHL: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + auto Dst = GetDst(Node); + mov (rcx, GetSrc(Op->Header.Args[1].ID())); + and(rcx, Mask); + + shlx(Reg32e(Dst.getIdx(), 64), GetSrc(Op->Header.Args[0].ID()), rcx); + break; + } + case IR::OP_ASHR: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov (rcx, GetSrc(Op->Header.Args[1].ID())); + and(rcx, Mask); + switch (OpSize) { + case 1: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + sar(al, cl); + movsx(GetDst(Node), al); + break; + case 2: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + sar(ax, cl); + movsx(GetDst(Node), ax); + break; + case 4: + sarx(Reg32e(GetDst(Node).getIdx(), 32), GetSrc(Op->Header.Args[0].ID()), ecx); + break; + case 8: + sarx(Reg32e(GetDst(Node).getIdx(), 64), GetSrc(Op->Header.Args[0].ID()), rcx); + break; + default: LogMan::Msg::A("Unknown ASHR Size: %d\n", OpSize); break; + }; + break; + } + case IR::OP_ROL: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov (rcx, GetSrc(Op->Header.Args[1].ID())); + and(rcx, Mask); + switch (OpSize) { + case 1: { + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + rol(al, cl); + break; + } + case 2: { + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + rol(ax, cl); + break; + } + case 4: { + mov(eax, GetSrc(Op->Header.Args[0].ID())); + rol(eax, cl); + break; + } + case 8: { + mov(rax, GetSrc(Op->Header.Args[0].ID())); + rol(rax, cl); + break; + } + } + mov(GetDst(Node), rax); + break; + } + case IR::OP_ROR: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov (rcx, GetSrc(Op->Header.Args[1].ID())); + and(rcx, Mask); + switch (OpSize) { + case 1: { + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + ror(al, cl); + break; + } + case 2: { + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + ror(ax, cl); + break; + } + case 4: { + mov(eax, GetSrc(Op->Header.Args[0].ID())); + ror(eax, cl); + break; + } + case 8: { + mov(rax, GetSrc(Op->Header.Args[0].ID())); + ror(rax, cl); + break; + } + } + mov(GetDst(Node), rax); + break; + } + case IR::OP_MUL: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + + switch (OpSize) { + case 1: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + movsx(rcx, GetSrc(Op->Header.Args[1].ID())); + imul(cl); + movsx(Dst, al); + break; + case 2: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + movsx(rcx, GetSrc(Op->Header.Args[1].ID())); + imul(cx); + movsx(Dst, ax); + break; + case 4: + movsxd(rax, GetSrc(Op->Header.Args[0].ID())); + imul(eax, GetSrc(Op->Header.Args[1].ID())); + movsx(Dst, eax); + break; + case 8: + mov(rax, GetSrc(Op->Header.Args[0].ID())); + imul(rax, GetSrc(Op->Header.Args[1].ID())); + mov(Dst, rax); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_MULH: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + movsx(rcx, GetSrc(Op->Header.Args[1].ID())); + imul(cl); + movsx(rax, ax); + mov(GetDst(Node), rax); + break; + case 2: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + movsx(rcx, GetSrc(Op->Header.Args[1].ID())); + imul(cx); + movsx(rax, dx); + mov(GetDst(Node), rax); + break; + case 4: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + imul(GetSrc(Op->Header.Args[1].ID())); + movsxd(rax, edx); + mov(GetDst(Node), rdx); + break; + case 8: + mov(rax, GetSrc(Op->Header.Args[0].ID())); + imul(GetSrc(Op->Header.Args[1].ID())); + mov(GetDst(Node), rdx); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_UMUL: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + movzx(rcx, GetSrc(Op->Header.Args[1].ID())); + mul(cl); + movzx(rax, al); + mov(GetDst(Node), rax); + break; + case 2: + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + movzx(rcx, GetSrc(Op->Header.Args[1].ID())); + mul(cx); + movzx(rax, ax); + mov(GetDst(Node), rax); + break; + case 4: + mov(rax, GetSrc(Op->Header.Args[0].ID())); + mul(GetSrc(Op->Header.Args[1].ID())); + mov(GetDst(Node), rax); + break; + case 8: + mov(rax, GetSrc(Op->Header.Args[0].ID())); + mul(GetSrc(Op->Header.Args[1].ID())); + mov(GetDst(Node), rax); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_UMULH: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + movzx(rcx, GetSrc(Op->Header.Args[1].ID())); + mul(cl); + movzx(rax, ax); + mov(GetDst(Node), rax); + break; + case 2: + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + movzx(rcx, GetSrc(Op->Header.Args[1].ID())); + mul(cx); + movzx(rax, dx); + mov(GetDst(Node), rax); + break; + case 4: + mov(rax, GetSrc(Op->Header.Args[0].ID())); + mul(GetSrc(Op->Header.Args[1].ID())); + mov(GetDst(Node), rdx); + break; + case 8: + mov(rax, GetSrc(Op->Header.Args[0].ID())); + mul(GetSrc(Op->Header.Args[1].ID())); + mov(GetDst(Node), rdx); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_LDIV: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov(eax, GetSrc(Op->Header.Args[0].ID())); + mov(edx, GetSrc(Op->Header.Args[1].ID())); + mov(ecx, GetSrc(Op->Header.Args[2].ID())); + idiv(ecx); + mov(GetDst(Node), rax); + break; + } + case 8: { + mov(rax, GetSrc(Op->Header.Args[0].ID())); + mov(rdx, GetSrc(Op->Header.Args[1].ID())); + mov(rcx, GetSrc(Op->Header.Args[2].ID())); + idiv(rcx); + mov(GetDst(Node), rax); + break; + } + default: LogMan::Msg::A("Unknown LDIV Size: %d", Size); break; + } + break; + } + case IR::OP_LREM: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov(eax, GetSrc(Op->Header.Args[0].ID())); + mov(edx, GetSrc(Op->Header.Args[1].ID())); + mov(ecx, GetSrc(Op->Header.Args[2].ID())); + idiv(ecx); + mov(GetDst(Node), rdx); + break; + } + + case 8: { + mov(rax, GetSrc(Op->Header.Args[0].ID())); + mov(rdx, GetSrc(Op->Header.Args[1].ID())); + mov(rcx, GetSrc(Op->Header.Args[2].ID())); + idiv(rcx); + mov(GetDst(Node), rdx); + break; + } + default: LogMan::Msg::A("Unknown LREM Size: %d", Size); break; + } + break; + } + case IR::OP_LUDIV: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov (eax, GetSrc(Op->Header.Args[0].ID())); + mov (edx, GetSrc(Op->Header.Args[1].ID())); + mov (ecx, GetSrc(Op->Header.Args[2].ID())); + div(ecx); + mov(GetDst(Node), rax); + break; + } + case 8: { + mov (rax, GetSrc(Op->Header.Args[0].ID())); + mov (rdx, GetSrc(Op->Header.Args[1].ID())); + mov (rcx, GetSrc(Op->Header.Args[2].ID())); + div(rcx); + mov(GetDst(Node), rax); + break; + } + default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; + } + break; + } + case IR::OP_LUREM: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov (eax, GetSrc(Op->Header.Args[0].ID())); + mov (edx, GetSrc(Op->Header.Args[1].ID())); + mov (ecx, GetSrc(Op->Header.Args[2].ID())); + div(ecx); + mov(GetDst(Node), rdx); + break; + } + + case 8: { + mov (rax, GetSrc(Op->Header.Args[0].ID())); + mov (rdx, GetSrc(Op->Header.Args[1].ID())); + mov (rcx, GetSrc(Op->Header.Args[2].ID())); + div(rcx); + mov(GetDst(Node), rdx); + break; + } + default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; + } + break; + } + case IR::OP_LOADFLAG: { + auto Op = IROp->C(); + + auto Dst = GetDst(Node); + movzx(Dst, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]); + and(Dst, 1); + break; + } + case IR::OP_STOREFLAG: { + auto Op = IROp->C(); + + mov (rax, GetSrc(Op->Header.Args[0].ID())); + and(rax, 1); + mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al); + break; + } + case IR::OP_SELECT: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + + mov(rax, GetSrc(Op->Header.Args[0].ID())); + cmp(rax, GetSrc(Op->Header.Args[1].ID())); + + switch (Op->Cond) { + case FEXCore::IR::COND_EQ: + mov(rax, GetSrc(Op->Header.Args[3].ID())); + cmove(rax, GetSrc(Op->Header.Args[2].ID())); + break; + case FEXCore::IR::COND_NEQ: + mov(rax, GetSrc(Op->Header.Args[3].ID())); + cmovne(rax, GetSrc(Op->Header.Args[2].ID())); + break; + case FEXCore::IR::COND_GE: + mov(rax, GetSrc(Op->Header.Args[3].ID())); + cmovge(rax, GetSrc(Op->Header.Args[2].ID())); + break; + case FEXCore::IR::COND_LT: + mov(rax, GetSrc(Op->Header.Args[2].ID())); + cmovae(rax, GetSrc(Op->Header.Args[3].ID())); + break; + case FEXCore::IR::COND_GT: + mov(rax, GetSrc(Op->Header.Args[3].ID())); + cmovg(rax, GetSrc(Op->Header.Args[2].ID())); + break; + case FEXCore::IR::COND_LE: + mov(rax, GetSrc(Op->Header.Args[3].ID())); + cmovle(rax, GetSrc(Op->Header.Args[2].ID())); + break; + case FEXCore::IR::COND_CS: + case FEXCore::IR::COND_CC: + case FEXCore::IR::COND_MI: + case FEXCore::IR::COND_PL: + case FEXCore::IR::COND_VS: + case FEXCore::IR::COND_VC: + case FEXCore::IR::COND_HI: + case FEXCore::IR::COND_LS: + default: + LogMan::Msg::A("Unsupported compare type"); + break; + } + mov (Dst, rax); + break; + } + case IR::OP_LOADMEM: { + auto Op = IROp->C(); + uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); + + auto Dst = GetDst(Node); + mov(rax, Memory); + add(rax, GetSrc(Op->Header.Args[0].ID())); + switch (Op->Size) { + case 1: { + movzx (Dst, byte [rax]); + } + break; + case 2: { + movzx (Dst, word [rax]); + } + break; + case 4: { + mov(Dst, dword [rax]); + } + break; + case 8: { + mov(Dst, qword [rax]); + } + break; + case 16: { + movups(GetDst(Node), xword [rax]); + if (MemoryDebug) { + movq(rcx, GetDst(Node)); + } + } + break; + default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size); + } + break; + } + case IR::OP_STOREMEM: { + auto Op = IROp->C(); + uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); + + mov(rax, Memory); + add(rax, GetSrc(Op->Header.Args[0].ID())); + switch (Op->Size) { + case 1: + mov(byte [rax], GetSrc(Op->Header.Args[1].ID())); + break; + case 2: + mov(word [rax], GetSrc(Op->Header.Args[1].ID())); + break; + case 4: + mov(dword [rax], GetSrc(Op->Header.Args[1].ID())); + break; + case 8: + mov(qword [rax], GetSrc(Op->Header.Args[1].ID())); + break; + case 16: + movups(xword [rax], GetSrc(Op->Header.Args[1].ID())); + break; + default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size); + } + break; + } + case IR::OP_SYSCALL: { + auto Op = IROp->C(); + // XXX: This is very terrible, but I don't care for right now + + push(rdi); + + for (auto &Reg : RA64) + push(Reg); + + // Syscall ABI for x86-64 + // this: rdi + // Thread: rsi + // ArgPointer: rdx (Stack) + // + // Result: RAX + + // These are pushed in reverse order because stacks + for (uint32_t i = 7; i > 0; --i) + push(GetSrc(Op->Header.Args[i - 1].ID())); + + mov(rsi, rdi); // Move thread in to rsi + mov(rdi, reinterpret_cast(&CTX->SyscallHandler)); + mov(rdx, rsp); + + using PtrType = uint64_t (FEXCore::SyscallHandler::*)(FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args); + union { + PtrType ptr; + uint64_t Raw; + } PtrCast; + PtrCast.ptr = &FEXCore::SyscallHandler::HandleSyscall; + mov(rax, PtrCast.Raw); + call(rax); + + // Reload arguments just in case they are sill live after the fact + for (uint32_t i = 0; i < 7; ++i) + pop(GetSrc(Op->Header.Args[i].ID())); + + for (uint32_t i = RA64.size(); i > 0; --i) + pop(RA64[i - 1]); + + pop(rdi); + + mov (GetDst(Node), rax); + break; + } + case IR::OP_CPUID: { + auto Op = IROp->C(); + using ClassPtrType = FEXCore::CPUIDEmu::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t Function); + union { + ClassPtrType ClassPtr; + uint64_t Raw; + } Ptr; + Ptr.ClassPtr = &CPUIDEmu::RunFunction; + + for (auto &Reg : RA64) + push(Reg); + + // CPUID ABI + // this: rdi + // Function: rsi + // + // Result: RAX, RDX. 4xi32 + push(rdi); + mov (rsi, GetSrc(Op->Header.Args[0].ID())); + mov (rdi, reinterpret_cast(&CTX->CPUID)); + + sub(rsp, 8); // Align + + mov(rax, Ptr.Raw); + call(rax); + + add(rsp, 8); // Align + + pop(rdi); + + for (uint32_t i = RA64.size(); i > 0; --i) + pop(RA64[i - 1]); + + auto Dst = GetDst(Node); + pinsrq(Dst, rax, 0); + pinsrd(Dst, rdx, 1); + break; + } + case IR::OP_EXTRACTELEMENT: { + auto Op = IROp->C(); + + uint32_t PhysReg = RAPass->GetNodeRegister(Op->Header.Args[0].ID()); + if (PhysReg >= XMMBase) { + switch (OpSize) { + case 1: + pextrb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); + break; + case 2: + pextrw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); + break; + case 4: + pextrd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); + break; + case 8: + pextrq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); + break; + default: LogMan::Msg::A("Unhandled ExtractElementSize: %d", OpSize); + } + } + else { + LogMan::Msg::A("Can't handle extract from GPR yet"); + } + break; + } + case IR::OP_VINSELEMENT: { + auto Op = IROp->C(); + movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); + + // Dst_d[Op->DestIdx] = Src2_d[Op->SrcIdx]; + + // pextrq reg64/mem64, xmm, imm + // pinsrq xmm, reg64/mem64, imm8 + switch (Op->ElementSize) { + case 1: { + pextrb(al, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); + pinsrb(xmm15, al, Op->DestIdx); + break; + } + case 2: { + pextrw(ax, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); + pinsrw(xmm15, ax, Op->DestIdx); + break; + } + case 4: { + pextrd(eax, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); + pinsrd(xmm15, eax, Op->DestIdx); + break; + } + case 8: { + pextrq(rax, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); + pinsrq(xmm15, rax, Op->DestIdx); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + + movapd(GetDst(Node), xmm15); + break; + } + case IR::OP_VADD: { + auto Op = IROp->C(); + switch (Op->ElementSize) { + case 1: { + vpaddb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 2: { + vpaddw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 4: { + vpaddd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 8: { + vpaddq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + break; + } + case IR::OP_VSUB: { + auto Op = IROp->C(); + switch (Op->ElementSize) { + case 1: { + vpsubb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 2: { + vpsubw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 4: { + vpsubd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 8: { + vpsubq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + break; + } + case IR::OP_VXOR: { + auto Op = IROp->C(); + vpxor(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case IR::OP_VOR: { + auto Op = IROp->C(); + vpor(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case IR::OP_VCMPEQ: { + auto Op = IROp->C(); + LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); + + switch (Op->ElementSize) { + case 1: + vpcmpeqb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + case 2: + vpcmpeqw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + case 4: + vpcmpeqd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + case 8: + vpcmpeqq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); + } + break; + } + case IR::OP_VCMPGT: { + auto Op = IROp->C(); + LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); + + switch (Op->ElementSize) { + case 1: + vpcmpgtb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + case 2: + vpcmpgtw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + case 4: + vpcmpgtd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + case 8: + vpcmpgtq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); + } + break; + } + case IR::OP_VZIP: { + auto Op = IROp->C(); + movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); + + switch (Op->ElementSize) { + case 1: { + punpcklbw(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + case 2: { + punpcklwd(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + case 4: { + punpckldq(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + case 8: { + punpcklqdq(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movapd(GetDst(Node), xmm15); + break; + } + case IR::OP_VZIP2: { + auto Op = IROp->C(); + movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); + + switch (Op->ElementSize) { + case 1: { + punpckhbw(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + case 2: { + punpckhwd(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + case 4: { + punpckhdq(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + case 8: { + punpckhqdq(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movapd(GetDst(Node), xmm15); + break; + } + case IR::OP_VUSHLS: { + auto Op = IROp->C(); + movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); + vmovq(xmm14, Reg64(GetSrc(Op->Header.Args[1].ID()).getIdx())); + + switch (Op->ElementSize) { + case 2: { + psllw(xmm15, xmm14); + break; + } + case 4: { + pslld(xmm15, xmm14); + break; + } + case 8: { + psllq(xmm15, xmm14); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movapd(GetDst(Node), xmm15); + + break; + } + case IR::OP_VEXTR: { + auto Op = IROp->C(); + vpalignr(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()), Op->Index); + break; + } + case IR::OP_VUMIN: { + auto Op = IROp->C(); + switch (Op->ElementSize) { + case 1: { + vpminub(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 2: { + vpminuw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 4: { + vpminud(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + break; + } + case IR::OP_VSMIN: { + auto Op = IROp->C(); + switch (Op->ElementSize) { + case 1: { + vpminsb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 2: { + vpminsw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 4: { + vpminsd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + break; + } + case IR::OP_CAS: { + auto Op = IROp->C(); + // Args[0]: Desired + // Args[1]: Expected + // Args[2]: Pointer + // DataSrc = *Src1 + // if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc + // This will write to memory! Careful! + // Third operand must be a calculated guest memory address + //OrderedNode *CASResult = _CAS(Src3, Src2, Src1); + uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); + + mov(rcx, Memory); + add(rcx, GetSrc(Op->Header.Args[2].ID())); + mov(rdx, GetSrc(Op->Header.Args[1].ID())); + mov(rax, GetSrc(Op->Header.Args[0].ID())); + + // RCX now contains pointer + // RAX contains our expected value + // RDX contains our desired + + lock(); + + switch (OpSize) { + case 1: { + cmpxchg(byte [rcx], dl); + movzx(rax, al); + break; + } + case 2: { + cmpxchg(word [rcx], dx); + movzx(rax, ax); + break; + } + case 4: { + cmpxchg(dword [rcx], edx); + break; + } + case 8: { + cmpxchg(qword [rcx], rdx); + break; + } + default: LogMan::Msg::A("Unsupported: %d", OpSize); + } + + // RAX now contains the result + mov (GetDst(Node), rax); + break; + } + case IR::OP_CYCLECOUNTER: { + #ifdef DEBUG_CYCLES + mov (GetDst(Node), 0); + #else + rdtsc(); + shl(rdx, 32); + or(rax, rdx); + mov (GetDst(Node), rax); + #endif + break; + } + case IR::OP_FINDLSB: { + auto Op = IROp->C(); + tzcnt(rcx, GetSrc(Op->Header.Args[0].ID())); + xor(rax, rax); + cmp(GetSrc(Op->Header.Args[0].ID()), 1); + sbb(rax, rax); + or(rax, rcx); + mov (GetDst(Node), rax); + break; + } + case IR::OP_FINDMSB: { + auto Op = IROp->C(); + mov(rax, OpSize * 8); + lzcnt(rcx, GetSrc(Op->Header.Args[0].ID())); + sub(rax, rcx); + mov (GetDst(Node), rax); + break; + } + case IR::OP_CODEBLOCK: + case IR::OP_IRHEADER: + case IR::OP_BEGINBLOCK: + case IR::OP_ENDBLOCK: + case IR::OP_EXITFUNCTION: + case IR::OP_ENDFUNCTION: + case IR::OP_BREAK: + case IR::OP_JUMP: + break; + default: + LogMan::Msg::A("Unknown IR Op: %d(%s)", IROp->Op, FEXCore::IR::GetName(IROp->Op).data()); + break; + } + } + else { + switch (IROp->Op) { + case IR::OP_LOADCONTEXT: { + auto Op = IROp->C(); + #define LOAD_CTX(x, y) \ + case x: { \ + movzx(rax, y [STATE + Op->Offset]); \ + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); \ + } \ + break + switch (Op->Size) { + LOAD_CTX(1, byte); + LOAD_CTX(2, word); + case 4: { + mov(eax, dword [STATE + Op->Offset]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + case 8: { + mov(rax, qword [STATE + Op->Offset]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + case 16: { + if (Op->Offset % 16 == 0) { + movaps(xmm0, xword [STATE + Op->Offset]); + movaps(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + } + else { + movups(xmm0, xword [STATE + Op->Offset]); + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + } + } + break; + default: LogMan::Msg::A("Unhandled LoadContext size: %d", Op->Size); + } + #undef LOAD_CTX + break; + } + case IR::OP_STORECONTEXT: { + auto Op = IROp->C(); + + switch (Op->Size) { + case 1: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(byte [STATE + Op->Offset], al); + } + break; + + case 2: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(word [STATE + Op->Offset], ax); + } + break; + case 4: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(dword [STATE + Op->Offset], eax); + } + break; + case 8: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(qword [STATE + Op->Offset], rax); + } + break; + case 16: { + if (Op->Offset % 16 == 0) { + movaps(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movaps(xword [STATE + Op->Offset], xmm0); + } + else { + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xword [STATE + Op->Offset], xmm0); + } + } + break; + default: LogMan::Msg::A("Unhandled StoreContext size: %d", Op->Size); + } + + break; + } + case IR::OP_FILLREGISTER: { + auto Op = IROp->C(); + uint32_t SlotOffset = Op->Slot * 16; + switch (OpSize) { + case 1: { + movzx(rax, byte [rsp + SlotOffset]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + case 2: { + movzx(rax, word [rsp + SlotOffset]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + case 4: { + mov(rax, dword [rsp + SlotOffset]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + case 8: { + mov(rax, qword [rsp + SlotOffset]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + case 16: { + movaps(xmm0, xword [rsp + SlotOffset]); + movaps(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + } + break; + default: LogMan::Msg::A("Unhandled FillRegister size: %d", OpSize); + } + break; + } + case IR::OP_SPILLREGISTER: { + auto Op = IROp->C(); + uint32_t SlotOffset = Op->Slot * 16; + switch (OpSize) { + case 1: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(byte [rsp + SlotOffset], al); + } + break; + case 2: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(word [rsp + SlotOffset], ax); + } + break; + case 4: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(dword [rsp + SlotOffset], eax); + } + break; + case 8: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(qword [rsp + SlotOffset], rax); + } + break; + case 16: { + movaps(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movaps(xword [rsp + SlotOffset], xmm0); + } + break; + default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize); + } + break; + } + case IR::OP_SYSCALL: { + auto Op = IROp->C(); + + push(rdi); + push(r11); + + // Syscall ABI for x86-64 + // this: rdi + // Thread: rsi + // ArgPointer: rdx (Stack) + // + // Result: RAX + + mov(rsi, rdi); // Move thread in to rsi + mov(rdi, reinterpret_cast(&CTX->SyscallHandler)); + + // These are pushed in reverse order because stacks + push(qword [TEMP_STACK + (Op->Header.Args[6].ID() * 16)]); + push(qword [TEMP_STACK + (Op->Header.Args[5].ID() * 16)]); + push(qword [TEMP_STACK + (Op->Header.Args[4].ID() * 16)]); + push(qword [TEMP_STACK + (Op->Header.Args[3].ID() * 16)]); + push(qword [TEMP_STACK + (Op->Header.Args[2].ID() * 16)]); + push(qword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + push(qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov (rdx, rsp); + + using PtrType = uint64_t (FEXCore::SyscallHandler::*)(FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args); + union { + PtrType ptr; + uint64_t Raw; + } PtrCast; + PtrCast.ptr = &FEXCore::SyscallHandler::HandleSyscall; + mov(rax, PtrCast.Raw); + call(rax); + add(rsp, 7 * 8); + + pop(r11); + pop(rdi); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_CPUID: { + auto Op = IROp->C(); + using ClassPtrType = FEXCore::CPUIDEmu::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t Function); + union { + ClassPtrType ClassPtr; + uint64_t Raw; + } Ptr; + Ptr.ClassPtr = &CPUIDEmu::RunFunction; + + // CPUID ABI + // this: rdi + // Function: rsi + // + // Result: RAX, RDX. 4xi32 + push(rdi); + push(r11); + mov (rsi, qword [TEMP_STACK + (Op->Header.Args[0].ID() *16)]); + mov (rdi, reinterpret_cast(&CTX->CPUID)); + + push(rax); // align + + mov(rax, Ptr.Raw); + call(rax); + + pop(r11); // align + + pop(r11); + pop(rdi); + + mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 0], eax); + shr(rax, 32); + mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 4], eax); + + mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 8], edx); + shr(rdx, 32); + mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 12], edx); + break; + } + case IR::OP_EXTRACTELEMENT: { + auto Op = IROp->C(); + + uint32_t Offset = Op->Header.Args[0].ID() * 16 + OpSize * Op->Idx; + switch (OpSize) { + case 1: + movzx(rax, byte [TEMP_STACK + Offset]); + break; + case 2: + movzx(rax, word [TEMP_STACK + Offset]); + break; + case 4: + mov(eax, dword [TEMP_STACK + Offset]); + break; + case 8: + mov(rax, qword [TEMP_STACK + Offset]); + break; + default: LogMan::Msg::A("Unhandled ExtractElementSize: %d", OpSize); + } + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_LOADFLAG: { + auto Op = IROp->C(); + + movzx(rax, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]); + and(rax, 1); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_STOREFLAG: { + auto Op = IROp->C(); + + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + and(rax, 1); + mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al); + break; + } + case IR::OP_CONDJUMP: { + auto Op = IROp->C(); + + Label *TargetLabel; + auto IsTarget = JumpTargets.find(Op->Header.Args[1].ID()); + if (IsTarget == JumpTargets.end()) { + TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[1].ID()).first->second; + } + else { + TargetLabel = &IsTarget->second; + } + + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + cmp(rax, 0); + jne(*TargetLabel, T_NEAR); + break; + } + case IR::OP_LOADMEM: { + auto Op = IROp->C(); + uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); + + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(rcx, Memory); + add(rax, rcx); + switch (Op->Size) { + case 1: { + movzx (rcx, byte [rax]); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); + break; + } + case 2: { + movzx (rcx, word [rax]); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); + break; + } + case 4: { + mov(ecx, dword [rax]); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); + break; + } + case 8: { + mov(rcx, qword [rax]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); + break; + } + case 16: { + movups(xmm0, xword [rax]); + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + if (MemoryDebug) { + movq(rcx, xmm0); + } + break; + } + default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size); + } + + if (MemoryDebug) { + push(rdi); + push(r11); + sub(rsp, 8); + + // Load the address in to Arg1 + mov(rdi, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + // Move the loaded value to Arg2 + mov(rsi, rcx); + mov (rdx, Op->Size); + + mov(rax, reinterpret_cast(LoadMem)); + call(rax); + + add(rsp, 8); + + pop(r11); + pop(rdi); + } + + break; + } + case IR::OP_STOREMEM: { + auto Op = IROp->C(); + uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); + + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(rcx, Memory); + add(rax, rcx); + switch (Op->Size) { + case 1: { + mov(cl, byte [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + mov(byte [rax], cl); + if (MemoryDebug) { + movzx(rcx, cl); + } + break; + } + case 2: { + mov(cx, word [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + mov(word [rax], cx); + + if (MemoryDebug) { + movzx(rcx, cx); + } + break; + } + case 4: { + mov(ecx, dword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + mov(dword [rax], ecx); + break; + } + case 8: { + mov(rcx, qword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + mov(qword [rax], rcx); + break; + } + case 16: { + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + movups(xword [rax], xmm0); + if (MemoryDebug) { + movq(rcx, xmm0); + } + break; + } + default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size); + } + + if (MemoryDebug) { + push(rdi); + push(r11); + sub(rsp, 8); + + // Load the address in to Arg1 + mov(rdi, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + // Load the value from RAX in to Arg2 + mov(rsi, rcx); + + mov (rdx, Op->Size); + + mov(rax, reinterpret_cast(StoreMem)); + call(rax); + + add(rsp, 8); + + pop(r11); + pop(rdi); + } + break; + } + case IR::OP_MOV: { + auto Op = IROp->C(); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_CONSTANT: { + auto Op = IROp->C(); + if (Op->Constant >> 31) { + mov(rax, Op->Constant); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + else { + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], Op->Constant); + } + break; + } + case IR::OP_POPCOUNT: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movzx(al, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + popcnt(eax, eax); + break; + case 2: + popcnt(ax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movzx(rax, ax); + break; + case 4: + popcnt(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + break; + case 8: + popcnt(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + break; + } + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_ADD: { + auto Op = IROp->C(); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + add(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_SUB: { + auto Op = IROp->C(); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + sub(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_XOR: { + auto Op = IROp->C(); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + xor(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_AND: { + auto Op = IROp->C(); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + and(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_OR: { + auto Op = IROp->C(); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + or(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_MUL: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movsx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + imul(cl); + movsx(rax, al); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 2: + movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movsx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + imul(cx); + movsx(rax, ax); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 4: + movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + imul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + movsxd(rax, eax); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 8: + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + imul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_UMUL: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movzx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mul(cl); + movzx(rax, al); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 2: + movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movzx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mul(cx); + movzx(rax, ax); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 4: + mov(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 8: + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_MULH: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movsx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + imul(cl); + movsx(rax, ax); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 2: + movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movsx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + imul(cx); + movsx(rax, dx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 4: + movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + imul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + movsxd(rax, edx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 8: + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + imul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_UMULH: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movzx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mul(cl); + movzx(rax, ax); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 2: + movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movzx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mul(cx); + movzx(rax, dx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 4: + mov(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + case 8: + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_LDIV: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + idiv(ecx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case 8: { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + idiv(rcx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + default: LogMan::Msg::A("Unknown LDIV Size: %d", Size); break; + } + break; + } + case IR::OP_LREM: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + idiv(ecx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + } + case 8: { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + idiv(rcx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + } + default: LogMan::Msg::A("Unknown LREM Size: %d", Size); break; + } + break; + } + case IR::OP_LUDIV: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + div(ecx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case 8: { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + div(rcx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; + } + break; + } + case IR::OP_LUREM: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + div(ecx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + } + case 8: { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + div(rcx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + } + default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; + } + break; + } + case IR::OP_ZEXT: { + auto Op = IROp->C(); + LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); + + if (Op->SrcSize == 64) { + movd(xmm0, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movups(qword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + } + else { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(rcx, uint64_t((1ULL << Op->SrcSize) - 1)); + and(rax, rcx); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + } + case IR::OP_SEXT: { + auto Op = IROp->C(); + LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); + switch (Op->SrcSize / 8) { + case 1: + movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 2: + movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 4: + movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 8: + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", Op->SrcSize / 8); + } + break; + } + case IR::OP_BFI: { + auto Op = IROp->C(); + LogMan::Throw::A(OpSize <= 8, "OpSize is too large for BFI: %d", OpSize); + + uint64_t SourceMask = (1ULL << Op->Width) - 1; + + uint64_t DestMask = ~(SourceMask << Op->lsb); + + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + + if (Op->Width != 64) { + mov(rdx, SourceMask); + and(rcx, rdx); + } + + mov(rdx, DestMask); + and(rax, rdx); + shl(rdx, Op->lsb); + or(rax, rdx); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_BFE: { + auto Op = IROp->C(); + LogMan::Throw::A(OpSize <= 16, "OpSize is too large for BFE: %d", OpSize); + // %ssa64 i128 = Bfe %ssa48 i128, 0x1, 0x7 + if (OpSize == 16) { + LogMan::Throw::A(!(Op->lsb < 64 && (Op->lsb + Op->Width > 64)), "Trying to BFE an XMM across the 64bit split: Beginning at %d, ending at %d", Op->lsb, Op->lsb + Op->Width); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + uint8_t Offset = Op->lsb; + if (Offset < 64) { + pextrq(rax, xmm0, 0); + } + else { + pextrq(rax, xmm0, 1); + Offset -= 64; + } + + if (Offset) { + shr(rax, Offset); + } + + if (Op->Width != 64) { + mov(rcx, uint64_t((1ULL << Op->Width) - 1)); + and(rax, rcx); + } + + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + else { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + if (Op->lsb != 0) + shr(rax, Op->lsb); + + if (Op->Width != 64) { + mov(rcx, uint64_t((1ULL << Op->Width) - 1)); + and(rax, rcx); + } + + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + } + case IR::OP_FINDLSB: { + auto Op = IROp->C(); + tzcnt(rcx, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + xor(rax, rax); + cmp(qword [TEMP_STACK + Op->Header.Args[0].ID() * 16], 1); + sbb(rax, rax); + or(rax, rcx); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_FINDMSB: { + auto Op = IROp->C(); + mov(rax, OpSize * 8); + lzcnt(rcx, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + sub(rax, rcx); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_LSHR: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + and(rcx, Mask); + shrx(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16], rcx); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_LSHL: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + and(rcx, Mask); + shlx(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16], rcx); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_ASHR: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + and(rcx, Mask); + switch (OpSize) { + case 1: + movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + sar(al, cl); + break; + case 2: + movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + sar(ax, cl); + break; + case 4: + movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + sar(eax, cl); + break; + case 8: + mov(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + sar(rax, cl); + break; + default: LogMan::Msg::A("Unknown ASHR Size: %d\n", OpSize); break; + }; + + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_ROL: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + and(rcx, Mask); + switch (OpSize) { + case 1: { + movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + rol(al, cl); + break; + } + case 2: { + movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + rol(ax, cl); + break; + } + case 4: { + mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + rol(eax, cl); + break; + } + case 8: { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + rol(rax, cl); + break; + } + } + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_ROR: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + and(rcx, Mask); + switch (OpSize) { + case 1: { + movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + ror(al, cl); + break; + } + case 2: { + movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + ror(ax, cl); + break; + } + case 4: { + mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + ror(eax, cl); + break; + } + case 8: { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + ror(rax, cl); + break; + } + } + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_SELECT: { + auto Op = IROp->C(); + + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + cmp(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + + switch (Op->Cond) { + case FEXCore::IR::COND_EQ: + mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); + cmove(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + break; + case FEXCore::IR::COND_NEQ: + mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); + cmovne(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + break; + case FEXCore::IR::COND_GE: + mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); + cmovge(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + break; + case FEXCore::IR::COND_LT: + mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + cmovae(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); + break; + case FEXCore::IR::COND_GT: + mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); + cmovg(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + break; + case FEXCore::IR::COND_LE: + mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); + cmovle(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + break; + case FEXCore::IR::COND_CS: + case FEXCore::IR::COND_CC: + case FEXCore::IR::COND_MI: + case FEXCore::IR::COND_PL: + case FEXCore::IR::COND_VS: + case FEXCore::IR::COND_VC: + case FEXCore::IR::COND_HI: + case FEXCore::IR::COND_LS: + default: + LogMan::Msg::A("Unsupported compare type"); + break; + } + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); + break; + } + case IR::OP_CAS: { + auto Op = IROp->C(); + // Args[0]: Expected + // Args[1]: Desired + // Args[2]: Pointer + // DataSrc = *Src1 + // if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc + // This will write to memory! Careful! + // Third operand must be a calculated guest memory address + //OrderedNode *CASResult = _CAS(Src3, Src2, Src1); + uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); + + mov(rcx, Memory); + add(rcx, qword [TEMP_STACK + (Op->Header.Args[2].ID() * 16)]); + + mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + + // RCX now contains pointer + // RAX contains our expected value + // RDX contains our desired + + lock(); + + switch (OpSize) { + case 1: { + cmpxchg(byte [rcx], dl); + movzx(rax, al); + break; + } + case 2: { + cmpxchg(word [rcx], dx); + movzx(rax, ax); + break; + } + case 4: { + cmpxchg(dword [rcx], edx); + break; + } + case 8: { + cmpxchg(qword [rcx], rdx); + break; + } + default: LogMan::Msg::A("Unsupported: %d", OpSize); + } + + // RAX now contains the result + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_VCMPEQ: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + + LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); + + switch (Op->ElementSize) { + case 1: + pcmpeqb(xmm0, xmm1); + break; + case 2: + pcmpeqw(xmm0, xmm1); + break; + case 4: + pcmpeqd(xmm0, xmm1); + break; + case 8: + pcmpeqq(xmm0, xmm1); + break; + default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VCMPGT: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + + LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); + + switch (Op->ElementSize) { + case 1: + pcmpgtb(xmm0, xmm1); + case 2: + pcmpgtw(xmm0, xmm1); + case 4: + pcmpgtd(xmm0, xmm1); + case 8: + pcmpgtq(xmm0, xmm1); + default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VXOR: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + pxor(xmm0, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VOR: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + por(xmm0, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VINSELEMENT: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + + // Dst_d[Op->DestIdx] = Src2_d[Op->SrcIdx]; + + // pextrq reg64/mem64, xmm, imm + // pinsrq xmm, reg64/mem64, imm8 + switch (Op->ElementSize) { + case 1: + pextrb(al, xmm1, Op->SrcIdx); + pinsrb(xmm0, al, Op->DestIdx); + break; + case 2: + pextrw(ax, xmm1, Op->SrcIdx); + pinsrw(xmm0, ax, Op->DestIdx); + break; + case 4: + pextrd(eax, xmm1, Op->SrcIdx); + pinsrd(xmm0, eax, Op->DestIdx); + break; + case 8: + pextrq(rax, xmm1, Op->SrcIdx); + pinsrq(xmm0, rax, Op->DestIdx); + break; + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VADD: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + switch (Op->ElementSize) { + case 1: + paddb(xmm0, xmm1); + break; + case 2: + paddw(xmm0, xmm1); + break; + case 4: + paddd(xmm0, xmm1); + break; + case 8: + paddq(xmm0, xmm1); + break; + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VSUB: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + switch (Op->ElementSize) { + case 1: + psubb(xmm0, xmm1); + break; + case 2: + psubw(xmm0, xmm1); + break; + case 4: + psubd(xmm0, xmm1); + break; + case 8: + psubq(xmm0, xmm1); + break; + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VUSHLS: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + + switch (Op->ElementSize) { + case 2: + psllw(xmm0, xmm1); + break; + case 4: + pslld(xmm0, xmm1); + break; + case 8: + psllq(xmm0, xmm1); + break; + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VZIP: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + switch (Op->ElementSize) { + case 1: + punpcklbw(xmm0, xmm1); + break; + case 2: + punpcklwd(xmm0, xmm1); + break; + case 4: + punpckldq(xmm0, xmm1); + break; + case 8: + punpcklqdq(xmm0, xmm1); + break; + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VZIP2: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + switch (Op->ElementSize) { + case 1: + punpckhbw(xmm0, xmm1); + break; + case 2: + punpckhwd(xmm0, xmm1); + break; + case 4: + punpckhdq(xmm0, xmm1); + break; + case 8: + punpckhqdq(xmm0, xmm1); + break; + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VEXTR: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + palignr(xmm0, xmm1, Op->Index); + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_CYCLECOUNTER: { + #ifdef DEBUG_CYCLES + mov (rax, 0); + #else + rdtsc(); + shl(rdx, 32); + or(rax, rdx); + #endif + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_CODEBLOCK: + case IR::OP_IRHEADER: + case IR::OP_BEGINBLOCK: + case IR::OP_ENDBLOCK: + case IR::OP_EXITFUNCTION: + case IR::OP_ENDFUNCTION: + case IR::OP_BREAK: + case IR::OP_JUMP: + break; + default: + LogMan::Msg::A("Unknown IR Op: %d(%s)", IROp->Op, FEXCore::IR::GetName(IROp->Op).data()); + break; + } + } + + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; + } + + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + } } ready(); -// LogMan::Msg::D("Ptr: %p,+%ld", Entry, getCurr() - (uintptr_t)Entry); -// static int a = 0; -// if (a++ > 100) -// __builtin_trap(); + return Entry; } +void JITCore::CreateCustomDispatch() { +// Temp registers +// rax, rcx, rdx, rsi, r8, r9, +// r10, r11 +// +// Callee Saved +// rbx, rbp, r12, r13, r14, r15 +// +// 1St Argument: rdi +// XMM: +// All temp +// r11 assigned to temp state + void *Entry = getCurr(); + + // while (!Thread->State.RunningEvents.ShouldStop.load()) { + // Ptr = FindBlock(RIP) + // if (!Ptr) + // Ptr = CTX->CompileBlock(RIP); + // + // if (Ptr) + // Ptr(); + // else + // { + // Ptr = FallbackCore->CompileBlock() + // if (Ptr) + // Ptr() + // else { + // ShouldStop = true; + // } + // } + // } + // Bunch of exit state stuff + ready(); + // CustomDispatchGenerated = true; +} + FEXCore::CPU::CPUBackend *CreateJITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) { return new JITCore(ctx); } diff --git a/Source/Interface/Core/LLVMJIT/LLVMCore.cpp b/Source/Interface/Core/LLVMJIT/LLVMCore.cpp index fe42562e9..976531622 100644 --- a/Source/Interface/Core/LLVMJIT/LLVMCore.cpp +++ b/Source/Interface/Core/LLVMJIT/LLVMCore.cpp @@ -21,7 +21,7 @@ #include #include -#define DESTMAP_AS_MAP 0 +#define DESTMAP_AS_MAP 1 #if DESTMAP_AS_MAP using DestMapType = std::unordered_map; #else @@ -185,18 +185,18 @@ private: llvm::Value *CastVectorToType(llvm::Value *Arg, bool Integer, uint8_t RegisterSize, uint8_t ElementSize); llvm::Value *CastToOpaqueStructure(llvm::Value *Arg, llvm::Type *DstType); - void SetDest(IR::NodeWrapper Op, llvm::Value *Val); - llvm::Value *GetSrc(IR::NodeWrapper Src); + void SetDest(IR::OrderedNodeWrapper Op, llvm::Value *Val); + llvm::Value *GetSrc(IR::OrderedNodeWrapper Src); DestMapType DestMap; FEXCore::IR::IRListView const *CurrentIR; - std::unordered_map JumpTargets; - std::unordered_map ForwardJumpTargets; + std::unordered_map JumpTargets; + std::unordered_map ForwardJumpTargets; // Target Machines const std::string arch = "x86-64"; - const std::string cpu = "znver2"; + const std::string cpu = "skylake"; const llvm::Triple TargetTriple{"x86_64", "unknown", "linux", "gnu"}; const llvm::SmallVector Attrs; llvm::TargetMachine *LLVMTarget; @@ -754,16 +754,16 @@ llvm::Value *LLVMJITCore::CastToOpaqueStructure(llvm::Value *Arg, llvm::Type *Ds return JITState.IRBuilder->CreateZExtOrTrunc(Arg, DstType->getPointerElementType()); } -void LLVMJITCore::SetDest(IR::NodeWrapper Op, llvm::Value *Val) { - DestMap[Op.NodeOffset] = Val; +void LLVMJITCore::SetDest(IR::OrderedNodeWrapper Op, llvm::Value *Val) { + DestMap[Op.ID()] = Val; } -llvm::Value *LLVMJITCore::GetSrc(IR::NodeWrapper Src) { +llvm::Value *LLVMJITCore::GetSrc(IR::OrderedNodeWrapper Src) { #if DESTMAP_AS_MAP - LogMan::Throw::A(DestMap.find(Src.NodeOffset) != DestMap.end(), "Op had Src but wasn't added to the dest map"); + LogMan::Throw::A(DestMap.find(Src.ID()) != DestMap.end(), "Op had Src but wasn't added to the dest map"); #endif - auto DstPtr = DestMap[Src.NodeOffset]; + auto DstPtr = DestMap[Src.ID()]; LogMan::Throw::A(DstPtr != nullptr, "Destmap had slot but wasn't allocated memory"); return DstPtr; } @@ -771,11 +771,11 @@ llvm::Value *LLVMJITCore::GetSrc(IR::NodeWrapper Src) { void LLVMJITCore::HandleIR(FEXCore::IR::IRListView const *IR, IR::NodeWrapperIterator *Node) { using namespace llvm; - uintptr_t ListBegin = CurrentIR->GetListData(); - uintptr_t DataBegin = CurrentIR->GetData(); + uintptr_t ListBegin = CurrentIR->GetListData(); + uintptr_t DataBegin = CurrentIR->GetData(); - IR::NodeWrapper *WrapperOp = (*Node)(); - IR::OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + IR::OrderedNodeWrapper *WrapperOp = (*Node)(); + IR::OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); uint8_t OpSize = IROp->Size; @@ -1812,7 +1812,6 @@ void LLVMJITCore::HandleIR(FEXCore::IR::IRListView const *IR, IR::NodeWrap LogMan::Msg::A("Unknown IR Op: %d(%s)", IROp->Op, FEXCore::IR::GetName(IROp->Op).data()); break; } - ++*Node; } void* FEXCore::CPU::LLVMJITCore::CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData) { @@ -1855,35 +1854,66 @@ void* FEXCore::CPU::LLVMJITCore::CompileCode(FEXCore::IR::IRListView const FunctionModule); Func->setCallingConv(CallingConv::C); - { - auto Entry = BasicBlock::Create(*Con, "Entry", Func); - JITCurrentState.Blocks.emplace_back(Entry); - JITState.IRBuilder->SetInsertPoint(Entry); - JITCurrentState.CurrentBlock = Entry; + { + auto Entry = BasicBlock::Create(*Con, "Entry", Func); + JITCurrentState.Blocks.emplace_back(Entry); + JITState.IRBuilder->SetInsertPoint(Entry); + JITCurrentState.CurrentBlock = Entry; - CreateGlobalVariables(Engine, FunctionModule); + CreateGlobalVariables(Engine, FunctionModule); - auto Builder = JITState.IRBuilder; + auto Builder = JITState.IRBuilder; - // Let's create the exit block quick - JITCurrentState.ExitBlock = BasicBlock::Create(*Con, "ExitBlock", Func); - JITCurrentState.Blocks.emplace_back(JITCurrentState.ExitBlock); + // Let's create the exit block quick + JITCurrentState.ExitBlock = BasicBlock::Create(*Con, "ExitBlock", Func); + JITCurrentState.Blocks.emplace_back(JITCurrentState.ExitBlock); - JITState.IRBuilder->SetInsertPoint(JITCurrentState.ExitBlock); - Builder->CreateRetVoid(); + JITState.IRBuilder->SetInsertPoint(JITCurrentState.ExitBlock); + Builder->CreateRetVoid(); - JITState.IRBuilder->SetInsertPoint(Entry); - JITCurrentState.CurrentBlock = Entry; - JITCurrentState.Blocks.emplace_back(JITCurrentState.CurrentBlock); - JITCurrentState.CurrentBlockHasTerm = false; + JITState.IRBuilder->SetInsertPoint(Entry); + JITCurrentState.CurrentBlock = Entry; + JITCurrentState.Blocks.emplace_back(JITCurrentState.CurrentBlock); + JITCurrentState.CurrentBlockHasTerm = false; - IR::NodeWrapperIterator Begin = CurrentIR->begin(); - IR::NodeWrapperIterator End = CurrentIR->end(); + uintptr_t ListBegin = CurrentIR->GetListData(); + uintptr_t DataBegin = CurrentIR->GetData(); - while (Begin != End) { - HandleIR(CurrentIR, &Begin); - } - } + auto HeaderIterator = CurrentIR->begin(); + IR::OrderedNodeWrapper *HeaderNodeWrapper = HeaderIterator(); + IR::OrderedNode *HeaderNode = HeaderNodeWrapper->GetNode(ListBegin); + auto HeaderOp = HeaderNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == IR::OP_IRHEADER, "First op wasn't IRHeader"); + + IR::OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); + + while (1) { + using namespace FEXCore::IR; + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block"); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR->at(BlockIROp->Begin); + auto CodeLast = CurrentIR->at(BlockIROp->Last); + + while (1) { + HandleIR(CurrentIR, &CodeBegin); + + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; + + } + + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + } + } + } for (auto &Block : JITCurrentState.Blocks) { // If the block is empty then let is just jump to the exit block @@ -1904,8 +1934,7 @@ void* FEXCore::CPU::LLVMJITCore::CompileCode(FEXCore::IR::IRListView const raw_ostream &Out = outs(); - //if (CTX->Config.LLVM_PrinterPass) - if (ThreadState->State.State.rip == 0x4021b0) + // if (CTX->Config.LLVM_PrinterPass) { FPM.addPass(PrintModulePass(Out)); } diff --git a/Source/Interface/Core/OpcodeDispatcher.cpp b/Source/Interface/Core/OpcodeDispatcher.cpp index 78258300d..e7acac8aa 100644 --- a/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/Source/Interface/Core/OpcodeDispatcher.cpp @@ -94,6 +94,7 @@ void OpDispatchBuilder::RETOp(OpcodeArgs) { // Store the new RIP _StoreContext(8, offsetof(FEXCore::Core::CPUState, rip), NewRIP); _EndFunction(); + CreateNewEndBlock(0); Information.HadUnconditionalExit = true; } @@ -279,11 +280,7 @@ void OpDispatchBuilder::CALLOp(OpcodeArgs) { _StoreContext(8, offsetof(FEXCore::Core::CPUState, rip), NewRIP); _ExitFunction(); // If we get here then leave the function now - // Fracking RIPSetter check ending the block causes issues - // Split the block and leave early to work around the bug - _EndBlock(0); - // Make sure to start a new block after ending this one - _BeginBlock(); + CreateNewEndBlock(0); Information.HadUnconditionalExit = true; } @@ -306,13 +303,8 @@ void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) { _StoreContext(8, offsetof(FEXCore::Core::CPUState, rip), JMPPCOffset); _ExitFunction(); // If we get here then leave the function now - // Fracking RIPSetter check ending the block causes issues - // Split the block and leave early to work around the bug - _EndBlock(0); - // Make sure to start a new block after ending this one - _BeginBlock(); + CreateNewEndBlock(0); Information.HadUnconditionalExit = true; - } void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) { @@ -514,9 +506,6 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) { #endif // Fallback { - // XXX: Test - GetPackedRFLAG(false); - auto CondJump = _CondJump(SrcCond); auto RIPOffset = LoadSource(Op, Op->Src1, Op->Flags); @@ -528,10 +517,10 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) { _StoreContext(8, offsetof(FEXCore::Core::CPUState, rip), NewRIP); _ExitFunction(); - _EndBlock(0); + CreateNewEndBlock(0); // Make sure to start a new block after ending this one - auto JumpTarget = _BeginBlock(); + auto JumpTarget = CreateNewBeginBlock(); // This very explicitly avoids the isDest path for Ops. We want the actual destination here SetJumpTarget(CondJump, JumpTarget); } @@ -580,10 +569,7 @@ void OpDispatchBuilder::JUMPOp(OpcodeArgs) { _StoreContext(8, offsetof(FEXCore::Core::CPUState, rip), NewRIP); _ExitFunction(); - _EndBlock(0); - - // Make sure to start a new block after ending this one - _BeginBlock(); + CreateNewEndBlock(0); Information.HadUnconditionalExit = true; } } @@ -598,12 +584,8 @@ void OpDispatchBuilder::JUMPAbsoluteOp(OpcodeArgs) { _StoreContext(8, offsetof(FEXCore::Core::CPUState, rip), RIPOffset); _ExitFunction(); - _EndBlock(0); - - // Make sure to start a new block after ending this one - _BeginBlock(); + CreateNewEndBlock(0); Information.HadUnconditionalExit = true; - } void OpDispatchBuilder::SETccOp(OpcodeArgs) { @@ -1159,19 +1141,10 @@ void OpDispatchBuilder::SHLOp(OpcodeArgs) { GenerateFlags_Shift(Op, _Bfe(Size, 0, ALUOp), _Bfe(Size, 0, Dest), _Bfe(Size, 0, Src)); } +template void OpDispatchBuilder::SHROp(OpcodeArgs) { - bool SHR1Bit = false; -#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg)) - switch (Op->OP) { - case OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 5): - case OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 5): - SHR1Bit = true; - break; - } -#undef OPD - OrderedNode *Src; - OrderedNode *Dest = LoadSource(Op, Op->Dest, Op->Flags); + auto Dest = LoadSource(Op, Op->Dest, Op->Flags); if (SHR1Bit) { Src = _Constant(1); @@ -1496,9 +1469,7 @@ void OpDispatchBuilder::RDTSCOp(OpcodeArgs) { } void OpDispatchBuilder::INCOp(OpcodeArgs) { - if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX) { - LogMan::Msg::A("Can't handle REP on this\n"); - } + LogMan::Throw::A(!(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX), "Can't handle REP on this\n"); OrderedNode *Dest = LoadSource(Op, Op->Dest, Op->Flags); auto OneConst = _Constant(1); @@ -1511,9 +1482,7 @@ void OpDispatchBuilder::INCOp(OpcodeArgs) { } void OpDispatchBuilder::DECOp(OpcodeArgs) { - if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX) { - LogMan::Msg::A("Can't handle REP on this\n"); - } + LogMan::Throw::A(!(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX), "Can't handle REP on this\n"); OrderedNode *Dest = LoadSource(Op, Op->Dest, Op->Flags); auto OneConst = _Constant(1); @@ -1526,9 +1495,8 @@ void OpDispatchBuilder::DECOp(OpcodeArgs) { } void OpDispatchBuilder::STOSOp(OpcodeArgs) { - if (!(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX)) { - LogMan::Msg::A("Can't handle REP not existing on STOS\n"); - } + LogMan::Throw::A(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX, "Can't handle REP not existing on STOS\n"); + auto Size = GetSrcSize(Op); auto ZeroConst = _Constant(0); @@ -1545,10 +1513,10 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) { OrderedNode *Src = LoadSource(Op, Op->Src1, Op->Flags); auto JumpStart = _Jump(); - _EndBlock(0); + CreateNewEndBlock(0); // Make sure to start a new block after ending this one - auto LoopStart = _BeginBlock(); + auto LoopStart = CreateNewBeginBlock(); SetJumpTarget(JumpStart, LoopStart); OrderedNode *Counter = _LoadContext(8, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RCX])); @@ -1576,23 +1544,18 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) { // Jump back to the start, we have more work to do _Jump(LoopStart); - _EndBlock(0); + CreateNewEndBlock(0); // Make sure to start a new block after ending this one - auto LoopEnd = _BeginBlock(); + auto LoopEnd = CreateNewBeginBlock(); SetJumpTarget(CondJump, LoopEnd); } void OpDispatchBuilder::MOVSOp(OpcodeArgs) { - if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX) { - LogMan::Msg::A("Can't handle REP\n"); - } - + LogMan::Throw::A(!(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX), "Can't handle REP on this\n"); _Break(0, 0); } void OpDispatchBuilder::CMPSOp(OpcodeArgs) { - if (!(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX)) { - LogMan::Msg::A("Can't only handle REP\n"); - } + LogMan::Throw::A(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX, "Can't only handle REP\n"); auto Size = GetSrcSize(Op); @@ -1608,9 +1571,9 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) { SizeConst, NegSizeConst); auto JumpStart = _Jump(); - _EndBlock(0); + CreateNewEndBlock(0); // Make sure to start a new block after ending this one - auto LoopStart = _BeginBlock(); + auto LoopStart = CreateNewBeginBlock(); SetJumpTarget(JumpStart, LoopStart); OrderedNode *Counter = _LoadContext(8, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RCX])); @@ -1645,9 +1608,9 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) { // Jump back to the start, we have more work to do _Jump(LoopStart); - _EndBlock(0); + CreateNewEndBlock(0); // Make sure to start a new block after ending this one - auto LoopEnd = _BeginBlock(); + auto LoopEnd = CreateNewBeginBlock(); SetJumpTarget(CondJump, LoopEnd); } @@ -2178,12 +2141,43 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) { } } -void OpDispatchBuilder::BeginBlock() { - _BeginBlock(); +OpDispatchBuilder::IRPair OpDispatchBuilder::CreateNewBeginBlock() { + auto CodeNode = CreateCodeNode(); + auto BeginBlock = _BeginBlock(); + SetCodeNodeBegin(CodeNode, BeginBlock); + CurrentCodeBlock = CodeNode; + return BeginBlock; } -void OpDispatchBuilder::EndBlock(uint64_t RIPIncrement) { - _EndBlock(RIPIncrement); +OpDispatchBuilder::IRPair OpDispatchBuilder::CreateNewEndBlock(uint64_t RIPIncrement) { + auto EndBlock = _EndBlock(RIPIncrement); + SetCodeNodeLast(CurrentCodeBlock, EndBlock); + return EndBlock; +} + +void OpDispatchBuilder::BeginFunction(uint64_t RIP) { + _IRHeader(RIP, InvalidNode->Wrapped(ListData.Begin()), 0); + CreateNewBeginBlock(); +} + +void OpDispatchBuilder::Finalize() { + // Node 0 is invalid node + OrderedNode *RealNode = reinterpret_cast(GetNode(1)); + FEXCore::IR::IROp_Header *IROp = RealNode->Op(Data.Begin()); + LogMan::Throw::A(IROp->Op == OP_IRHEADER, "First op in function must be our header"); + FEXCore::IR::IROp_IRHeader *Op = IROp->CW(); + Op->BlockCount = CodeBlocks.size(); + + OrderedNode *PrevCodeBlock{}; + for (auto &CodeBlock : CodeBlocks) { + if (PrevCodeBlock) { + LinkCodeBlocks(PrevCodeBlock, CodeBlock); + } + PrevCodeBlock = CodeBlock; + } + + Op->Blocks = CodeBlocks[0]->Wrapped(ListData.Begin()); + CodeBlocks.clear(); } void OpDispatchBuilder::ExitFunction() { @@ -2437,7 +2431,7 @@ void OpDispatchBuilder::StoreResult(FEXCore::X86Tables::DecodedOp Op, OrderedNod void OpDispatchBuilder::TestFunction() { printf("Doing Test Function\n"); - _BeginBlock(); + CreateNewBeginBlock(); auto Load1 = _LoadContext(8, 0); auto Load2 = _LoadContext(8, 0); //auto Res = Load1 Load2; @@ -2461,12 +2455,14 @@ OpDispatchBuilder::OpDispatchBuilder() void OpDispatchBuilder::ResetWorkingList() { Data.Reset(); ListData.Reset(); + CodeBlocks.clear(); CurrentWriteCursor = nullptr; // This is necessary since we do "null" pointer checks InvalidNode = reinterpret_cast(ListData.Allocate(sizeof(OrderedNode))); DecodeFailure = false; Information.HadUnconditionalExit = false; ShouldDump = false; + CurrentCodeBlock = nullptr; } template @@ -2529,9 +2525,6 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(bool Lower8) { } void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - auto Size = GetSrcSize(Op) * 8; // AF { @@ -2542,9 +2535,9 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde // SF { - auto ThirtyOneConst = _Constant(Size - 1); + auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); - auto LshrOp = _Lshr(Res, ThirtyOneConst); + auto LshrOp = _Lshr(Res, SignBitConst); SetRFLAG(LshrOp); } @@ -2552,7 +2545,7 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde { auto PopCountOp = _Popcount(_And(Res, _Constant(0xFF))); - auto XorOp = _Xor(PopCountOp, OneConst); + auto XorOp = _Xor(PopCountOp, _Constant(1)); SetRFLAG(XorOp); } @@ -2561,7 +2554,7 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde auto Dst8 = _Bfe(Size, 0, Res); auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Dst8, ZeroConst, OneConst, ZeroConst); + Dst8, _Constant(0), _Constant(1), _Constant(0)); SetRFLAG(SelectOp); } @@ -2571,9 +2564,9 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde auto Dst8 = _Bfe(Size, 0, Res); auto Src8 = _Bfe(Size, 0, Src2); - auto SelectOpLT = _Select(FEXCore::IR::COND_LT, Dst8, Src8, OneConst, ZeroConst); - auto SelectOpLE = _Select(FEXCore::IR::COND_LE, Dst8, Src8, OneConst, ZeroConst); - auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, OneConst, SelectOpLE, SelectOpLT); + auto SelectOpLT = _Select(FEXCore::IR::COND_LT, Dst8, Src8, _Constant(1), _Constant(0)); + auto SelectOpLE = _Select(FEXCore::IR::COND_LE, Dst8, Src8, _Constant(1), _Constant(0)); + auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, _Constant(1), SelectOpLE, SelectOpLT); SetRFLAG(SelectCF); } @@ -2605,9 +2598,6 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde } void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - // AF { OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); @@ -2617,9 +2607,9 @@ void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, Orde // SF { - auto ThirtyOneConst = _Constant(GetSrcSize(Op) * 8 - 1); + auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); - auto LshrOp = _Lshr(Res, ThirtyOneConst); + auto LshrOp = _Lshr(Res, SignBitConst); SetRFLAG(LshrOp); } @@ -2627,14 +2617,14 @@ void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, Orde { auto PopCountOp = _Popcount(_And(Res, _Constant(0xFF))); - auto XorOp = _Xor(PopCountOp, OneConst); + auto XorOp = _Xor(PopCountOp, _Constant(1)); SetRFLAG(XorOp); } // ZF { auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, ZeroConst, OneConst, ZeroConst); + Res, _Constant(0), _Constant(1), _Constant(0)); SetRFLAG(SelectOp); } @@ -2644,9 +2634,9 @@ void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, Orde auto Dst8 = _Bfe(GetSrcSize(Op) * 8, 0, Res); auto Src8_1 = _Bfe(GetSrcSize(Op) * 8, 0, Src1); - auto SelectOpLT = _Select(FEXCore::IR::COND_GT, Dst8, Src8_1, OneConst, ZeroConst); - auto SelectOpLE = _Select(FEXCore::IR::COND_GE, Dst8, Src8_1, OneConst, ZeroConst); - auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, OneConst, SelectOpLE, SelectOpLT); + auto SelectOpLT = _Select(FEXCore::IR::COND_GT, Dst8, Src8_1, _Constant(1), _Constant(0)); + auto SelectOpLE = _Select(FEXCore::IR::COND_GE, Dst8, Src8_1, _Constant(1), _Constant(0)); + auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, _Constant(1), SelectOpLE, SelectOpLT); SetRFLAG(SelectCF); } @@ -2677,8 +2667,6 @@ void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, Orde } void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); // AF { OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); @@ -2698,12 +2686,14 @@ void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, Orde { auto EightBitMask = _Constant(0xFF); auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, OneConst); + auto XorOp = _Xor(PopCountOp, _Constant(1)); SetRFLAG(XorOp); } // ZF { + auto ZeroConst = _Constant(0); + auto OneConst = _Constant(1); auto Bfe8 = _Bfe(GetSrcSize(Op) * 8, 0, Res); auto SelectOp = _Select(FEXCore::IR::COND_EQ, Bfe8, ZeroConst, OneConst, ZeroConst); @@ -2712,6 +2702,9 @@ void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, Orde // CF { + auto ZeroConst = _Constant(0); + auto OneConst = _Constant(1); + auto SelectOp = _Select(FEXCore::IR::COND_LT, Src1, Src2, OneConst, ZeroConst); @@ -2743,9 +2736,6 @@ void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, Orde } void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - // AF { OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); @@ -2765,14 +2755,14 @@ void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, Orde { auto EightBitMask = _Constant(0xFF); auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, OneConst); + auto XorOp = _Xor(PopCountOp, _Constant(1)); SetRFLAG(XorOp); } // ZF { auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, ZeroConst, OneConst, ZeroConst); + Res, _Constant(0), _Constant(1), _Constant(0)); SetRFLAG(SelectOp); } // CF @@ -2780,7 +2770,7 @@ void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, Orde auto Dst8 = _Bfe(GetSrcSize(Op) * 8, 0, Res); auto Src8 = _Bfe(GetSrcSize(Op) * 8, 0, Src2); - auto SelectOp = _Select(FEXCore::IR::COND_LT, Dst8, Src8, OneConst, ZeroConst); + auto SelectOp = _Select(FEXCore::IR::COND_LT, Dst8, Src8, _Constant(1), _Constant(0)); SetRFLAG(SelectOp); } @@ -2813,17 +2803,15 @@ void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, Orde } void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *High) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); // PF/AF/ZF/SF // Undefined { - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); } // CF/OF @@ -2833,7 +2821,7 @@ void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, Orde auto SignBit = _Ashr(Res, SignBitConst); - auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, SignBit, ZeroConst, OneConst); + auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, SignBit, _Constant(0), _Constant(1)); SetRFLAG(SelectOp); SetRFLAG(SelectOp); @@ -2841,16 +2829,13 @@ void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, Orde } void OpDispatchBuilder::GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *High) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - // AF/SF/PF/ZF // Undefined { - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); } // CF/OF @@ -2858,7 +2843,7 @@ void OpDispatchBuilder::GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, Ord // CF and OF are set if the result of the operation can't be fit in to the destination register // The result register will be all zero if it can't fit due to how multiplication behaves - auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, ZeroConst, ZeroConst, OneConst); + auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, _Constant(0), _Constant(0), _Constant(1)); SetRFLAG(SelectOp); SetRFLAG(SelectOp); @@ -2866,13 +2851,11 @@ void OpDispatchBuilder::GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, Ord } void OpDispatchBuilder::GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); // AF { // Undefined // Set to zero anyway - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); } // SF @@ -2887,35 +2870,32 @@ void OpDispatchBuilder::GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, { auto EightBitMask = _Constant(0xFF); auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, OneConst); + auto XorOp = _Xor(PopCountOp, _Constant(1)); SetRFLAG(XorOp); } // ZF { auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, ZeroConst, OneConst, ZeroConst); + Res, _Constant(0), _Constant(1), _Constant(0)); SetRFLAG(SelectOp); } // CF/OF { - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); } } void OpDispatchBuilder::GenerateFlags_Shift(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - - auto CmpResult = _Select(FEXCore::IR::COND_EQ, Src2, ZeroConst, OneConst, ZeroConst); + auto CmpResult = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), _Constant(1), _Constant(0)); auto CondJump = _CondJump(CmpResult); // AF { // Undefined // Set to zero anyway - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); } // SF @@ -2930,36 +2910,35 @@ void OpDispatchBuilder::GenerateFlags_Shift(FEXCore::X86Tables::DecodedOp Op, Or { auto EightBitMask = _Constant(0xFF); auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, OneConst); + auto XorOp = _Xor(PopCountOp, _Constant(1)); SetRFLAG(XorOp); } // ZF { auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, ZeroConst, OneConst, ZeroConst); + Res, _Constant(0), _Constant(1), _Constant(0)); SetRFLAG(SelectOp); } // CF/OF { - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); } - _EndBlock(0); - auto NewBlock = _BeginBlock(); + CreateNewEndBlock(0); + + auto NewBlock = CreateNewBeginBlock(); SetJumpTarget(CondJump, NewBlock); } void OpDispatchBuilder::GenerateFlags_Rotate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - auto ZeroConst = _Constant(0); - // CF/OF // XXX: These are wrong { - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); } } @@ -3091,10 +3070,10 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) { // If condition doesn't hold then keep going auto CondJump = _CondJump(_Xor(Flag, _Constant(1))); _Break(Reason, Literal); - _EndBlock(0); + CreateNewEndBlock(0); // Make sure to start a new block after ending this one - auto JumpTarget = _BeginBlock(); + auto JumpTarget = CreateNewBeginBlock(); SetJumpTarget(CondJump, JumpTarget); } else { @@ -3240,8 +3219,40 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) { } } +void OpDispatchBuilder::PAlignrOp(OpcodeArgs) { + OrderedNode *Src1 = LoadSource(Op, Op->Dest, Op->Flags); + OrderedNode *Src2 = LoadSource(Op, Op->Src1, Op->Flags); + + uint8_t Index = Op->Src2.TypeLiteral.Literal; + OrderedNode *Res = _VExtr(GetDstSize(Op), 1, Src1, Src2, Index); + StoreResult(Op, Res); +} + #undef OpcodeArgs + +void OpDispatchBuilder::ReplaceAllUsesWithInclusive(OrderedNode *Node, OrderedNode *NewNode, IR::NodeWrapperIterator After, IR::NodeWrapperIterator End) { + uintptr_t ListBegin = ListData.Begin(); + uintptr_t DataBegin = Data.Begin(); + + while (After != End) { + OrderedNodeWrapper *WrapperOp = After(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); + FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + + for (uint8_t i = 0; i < IROp->NumArgs; ++i) { + if (IROp->Args[i].ID() == Node->Wrapped(ListBegin).ID()) { + LogMan::Msg::D("\tAt %%ssa%d: Replacing ID %%ssa%d with %%ssa%d", WrapperOp->ID(), IROp->Args[i].ID(), NewNode->Wrapped(ListBegin).ID()); + Node->RemoveUse(); + NewNode->AddUse(); + IROp->Args[i].NodeOffset = NewNode->Wrapped(ListBegin).NodeOffset; + } + } + + ++After; + } +} + void InstallOpcodeHandlers() { const std::vector> BaseOpTable = { // Instructions @@ -3279,6 +3290,7 @@ void InstallOpcodeHandlers() { {0x9F, 1, &OpDispatchBuilder::LAHFOp}, {0xA0, 4, &OpDispatchBuilder::MOVOffsetOp}, {0xA4, 2, &OpDispatchBuilder::MOVSOp}, + // XXX: Causes issues with ld.so {0xA6, 2, &OpDispatchBuilder::CMPSOp}, {0xA8, 2, &OpDispatchBuilder::TESTOp}, {0xAA, 2, &OpDispatchBuilder::STOSOp}, @@ -3396,37 +3408,37 @@ void InstallOpcodeHandlers() { {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 0), 1, &OpDispatchBuilder::ROLOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 1), 1, &OpDispatchBuilder::ROROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 4), 1, &OpDispatchBuilder::SHLOp}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 5), 1, &OpDispatchBuilder::SHROp}, + {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 5), 1, &OpDispatchBuilder::SHROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 7), 1, &OpDispatchBuilder::ASHROp}, // SAR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 0), 1, &OpDispatchBuilder::ROLOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 1), 1, &OpDispatchBuilder::ROROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 4), 1, &OpDispatchBuilder::SHLOp}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 5), 1, &OpDispatchBuilder::SHROp}, + {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 5), 1, &OpDispatchBuilder::SHROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 7), 1, &OpDispatchBuilder::ASHROp}, // SAR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 0), 1, &OpDispatchBuilder::ROLOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 1), 1, &OpDispatchBuilder::ROROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 4), 1, &OpDispatchBuilder::SHLOp}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 5), 1, &OpDispatchBuilder::SHROp}, // 1Bit SHR + {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 5), 1, &OpDispatchBuilder::SHROp}, // 1Bit SHR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 7), 1, &OpDispatchBuilder::ASHROp}, // SAR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 0), 1, &OpDispatchBuilder::ROLOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 1), 1, &OpDispatchBuilder::ROROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 4), 1, &OpDispatchBuilder::SHLOp}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 5), 1, &OpDispatchBuilder::SHROp}, // 1Bit SHR + {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 5), 1, &OpDispatchBuilder::SHROp}, // 1Bit SHR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 7), 1, &OpDispatchBuilder::ASHROp}, // SAR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, &OpDispatchBuilder::ROLOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, &OpDispatchBuilder::ROROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, &OpDispatchBuilder::SHLOp}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL + {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 7), 1, &OpDispatchBuilder::ASHROp}, // SAR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, &OpDispatchBuilder::ROLOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, &OpDispatchBuilder::ROROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, &OpDispatchBuilder::SHLOp}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL + {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, &OpDispatchBuilder::ASHROp}, // SAR // GROUP 3 @@ -3459,7 +3471,6 @@ void InstallOpcodeHandlers() { {OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 6), 1, &OpDispatchBuilder::PUSHOp}, // GROUP 11 - // XXX: LLVM hangs when commented out? {OPD(FEXCore::X86Tables::TYPE_GROUP_11, OpToIndex(0xC6), 0), 1, &OpDispatchBuilder::MOVOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, &OpDispatchBuilder::MOVOp}, #undef OPD @@ -3467,9 +3478,7 @@ void InstallOpcodeHandlers() { const std::vector> RepModOpTable = { {0x19, 7, &OpDispatchBuilder::NOPOp}, - {0x6F, 1, &OpDispatchBuilder::MOVUPSOp}, - // XXX: Causes LLVM to crash if commented out? {0x7E, 1, &OpDispatchBuilder::MOVQOp}, {0x7F, 1, &OpDispatchBuilder::MOVUPSOp}, }; @@ -3545,7 +3554,8 @@ constexpr uint16_t PF_F2 = 3; {OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 2), 1, &OpDispatchBuilder::PSRLD<4>}, {OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 6), 1, &OpDispatchBuilder::PSLL<8, true>}, {OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 3), 1, &OpDispatchBuilder::PSRLDQ}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 7), 1, &OpDispatchBuilder::PSLL<16, true>}, + // XXX: Causes issues with ld.so + // {OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 7), 1, &OpDispatchBuilder::PSLL<16, true>}, // GROUP 15 {OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 0), 1, &OpDispatchBuilder::FXSaveOp}, @@ -3564,6 +3574,18 @@ constexpr uint16_t PF_F2 = 3; const std::vector> X87OpTable = { }; +#define OPD(REX, prefix, opcode) ((REX << 9) | (prefix << 8) | opcode) +#define PF_3A_NONE 0 +#define PF_3A_66 1 + const std::vector> H0F3ATable = { + {OPD(0, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp}, + }; +#undef PF_3A_NONE +#undef PF_3A_66 + +#undef OPD + + uint64_t NumInsts{}; auto InstallToTable = [&NumInsts](auto& FinalTable, auto& LocalTable) { for (auto Op : LocalTable) { @@ -3600,6 +3622,8 @@ constexpr uint16_t PF_F2 = 3; InstallToTable(FEXCore::X86Tables::X87Ops, X87OpTable); + InstallToTable(FEXCore::X86Tables::H0F3ATableOps, H0F3ATable); + // Useful for debugging // CheckTable(FEXCore::X86Tables::BaseOps); printf("We installed %ld instructions to the tables\n", NumInsts); diff --git a/Source/Interface/Core/OpcodeDispatcher.h b/Source/Interface/Core/OpcodeDispatcher.h index 715b56661..fc50495ac 100644 --- a/Source/Interface/Core/OpcodeDispatcher.h +++ b/Source/Interface/Core/OpcodeDispatcher.h @@ -28,9 +28,9 @@ public: void ResetWorkingList(); bool HadDecodeFailure() { return DecodeFailure; } - void BeginBlock(); - void EndBlock(uint64_t RIPIncrement); + void BeginFunction(uint64_t RIP); void ExitFunction(); + void Finalize(); // Dispatch builder functions #define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op @@ -70,6 +70,7 @@ public: void CMOVOp(OpcodeArgs); void CPUIDOp(OpcodeArgs); void SHLOp(OpcodeArgs); + template void SHROp(OpcodeArgs); void ASHROp(OpcodeArgs); void ROROp(OpcodeArgs); @@ -128,6 +129,8 @@ public: void FXSaveOp(OpcodeArgs); void FXRStoreOp(OpcodeArgs); + void PAlignrOp(OpcodeArgs); + #undef OpcodeArgs /** @@ -215,7 +218,9 @@ public: IRPair _VUShr(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) { return _VUShr(ssa0, ssa1, RegisterSize, ElementSize); } - + IRPair _VExtr(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1, uint8_t Index) { + return _VExtr(ssa0, ssa1, RegisterSize, ElementSize, Index); + } IRPair _Jump() { return _Jump(InvalidNode); } @@ -232,8 +237,8 @@ public: /** @} */ - bool IsValueConstant(NodeWrapper ssa, uint64_t *Constant) { - OrderedNode *RealNode = reinterpret_cast(ssa.GetPtr(ListData.Begin())); + bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t *Constant) { + OrderedNode *RealNode = ssa.GetNode(ListData.Begin()); FEXCore::IR::IROp_Header *IROp = RealNode->Op(Data.Begin()); if (IROp->Op == OP_CONSTANT) { auto Op = IROp->C(); @@ -259,6 +264,8 @@ public: return Node; } + void ReplaceAllUsesWithInclusive(OrderedNode *Node, OrderedNode *NewNode, IR::NodeWrapperIterator After, IR::NodeWrapperIterator End); + void Unlink(OrderedNode *Node) { Node->Unlink(ListData.Begin()); } @@ -271,8 +278,55 @@ public: LogMan::Throw::A(rhs.ListData.BackingSize() <= ListData.BackingSize(), "Trying to take ownership of data that is too large"); Data.CopyData(rhs.Data); ListData.CopyData(rhs.ListData); + InvalidNode = rhs.InvalidNode; + CurrentWriteCursor = rhs.CurrentWriteCursor; + CodeBlocks = rhs.CodeBlocks; } + void SetWriteCursor(OrderedNode *Node) { + CurrentWriteCursor = Node; + } + + OrderedNode *GetWriteCursor() { + return CurrentWriteCursor; + } + + /** + * @brief This creates an orphaned code node + * The IROp backing is in the correct list but the OrderedNode lives outside of the list + * + * XXX: This is because we don't want code blocks to interleave with current instruction IR ops currently + * We can change this behaviour once we remove the old BeginBlock/EndBlock types + * + * @return OrderedNode + */ + OrderedNode *CreateCodeNode() { + OrderedNode *CodeNode = _CodeBlock(InvalidNode, InvalidNode, InvalidNode); + CodeBlocks.emplace_back(CodeNode); + return CodeNode; + } + + void SetCodeNodeBegin(OrderedNode *CodeNode, OrderedNode *Begin) { + FEXCore::IR::IROp_CodeBlock *IROp = CodeNode->Op(Data.Begin())->CW(); + LogMan::Throw::A(IROp->Header.Op == IROps::OP_CODEBLOCK, "Invalid"); + IROp->Begin = Begin->Wrapped(ListData.Begin()); + } + + void SetCodeNodeLast(OrderedNode *CodeNode, OrderedNode *Last) { + FEXCore::IR::IROp_CodeBlock *IROp = CodeNode->Op(Data.Begin())->CW(); + LogMan::Throw::A(IROp->Header.Op == IROps::OP_CODEBLOCK, "Invalid"); + IROp->Last = Last->Wrapped(ListData.Begin()); + } + + void LinkCodeBlocks(OrderedNode *CodeNode, OrderedNode *Next) { + FEXCore::IR::IROp_CodeBlock *IROp = CodeNode->Op(Data.Begin())->CW(); + LogMan::Throw::A(IROp->Header.Op == IROps::OP_CODEBLOCK, "Invalid"); + IROp->Next = Next->Wrapped(ListData.Begin()); + } + + IRPair CreateNewBeginBlock(); + IRPair CreateNewEndBlock(uint64_t RIPIncrement); + private: void TestFunction(); bool DecodeFailure{false}; @@ -314,12 +368,17 @@ private: return Node; } - void SetWriteCursor(OrderedNode *Node) { - CurrentWriteCursor = Node; + OrderedNode *GetNode(uint32_t SSANode) { + uintptr_t ListBegin = ListData.Begin(); + OrderedNode *Node = reinterpret_cast(ListBegin + SSANode * sizeof(OrderedNode)); + return Node; } - OrderedNode *GetWriteCursor() { - return CurrentWriteCursor; + OrderedNode *EmplaceOrphanedNode(OrderedNode *OldNode) { + size_t Size = sizeof(OrderedNode); + OrderedNode *Ptr = reinterpret_cast(ListData.Allocate(Size)); + memcpy(Ptr, OldNode, Size); + return Ptr; } OrderedNode *CurrentWriteCursor = nullptr; @@ -329,6 +388,8 @@ private: IntrusiveAllocator ListData; OrderedNode *InvalidNode; + OrderedNode *CurrentCodeBlock{}; + std::vector CodeBlocks; }; void InstallOpcodeHandlers(); diff --git a/Source/Interface/Core/RegisterAllocation.cpp b/Source/Interface/Core/RegisterAllocation.cpp deleted file mode 100644 index 8563640b4..000000000 --- a/Source/Interface/Core/RegisterAllocation.cpp +++ /dev/null @@ -1,230 +0,0 @@ -#include "Common/BitSet.h" -#include "Interface/Core/RegisterAllocation.h" - -#include - -#include - -constexpr uint32_t INVALID_REG = ~0U; -constexpr uint32_t INVALID_CLASS = ~0U; - - -namespace FEXCore::RA { - - struct Register { - }; - - struct RegisterClass { - uint32_t RegisterBase; - uint32_t NumberOfRegisters{0}; - BitSet Registers; - }; - - struct RegisterNode { - uint32_t RegisterClass; - uint32_t Register; - uint32_t InterferenceCount; - uint32_t InterferenceListSize; - uint32_t *InterferenceList; - BitSet Interference; - }; - - static_assert(std::is_pod::value, "We want this to be POD"); - - struct RegisterSet { - Register *Registers; - RegisterClass *RegisterClasses; - uint32_t RegisterCount; - uint32_t ClassCount; - }; - - struct SpillStackUnit { - uint32_t Node; - uint32_t Class; - }; - - struct RegisterGraph { - RegisterSet *Set; - RegisterNode *Nodes; - uint32_t NodeCount; - uint32_t MaxNodeCount; - std::vector SpillStack; - }; - - RegisterSet *AllocateRegisterSet(uint32_t RegisterCount, uint32_t ClassCount) { - RegisterSet *Set = new RegisterSet; - - Set->RegisterCount = RegisterCount; - Set->ClassCount = ClassCount; - - Set->Registers = static_cast(calloc(RegisterCount, sizeof(Register))); - Set->RegisterClasses = static_cast(calloc(ClassCount, sizeof(RegisterClass))); - - for (uint32_t i = 0; i < ClassCount; ++i) { - Set->RegisterClasses[i].Registers.Allocate(RegisterCount); - } - - return Set; - } - - void FreeRegisterSet(RegisterSet *Set) { - for (uint32_t i = 0; i < Set->ClassCount; ++i) { - Set->RegisterClasses[i].Registers.Free(); - } - free(Set->RegisterClasses); - free(Set->Registers); - delete Set; - } - - void AddRegisters(RegisterSet *Set, uint32_t Class, uint32_t RegistersBase, uint32_t RegisterCount) { - for (uint32_t i = 0; i < RegisterCount; ++i) { - Set->RegisterClasses[Class].Registers.Set(RegistersBase + i); - } - Set->RegisterClasses[Class].RegisterBase = RegistersBase; - Set->RegisterClasses[Class].NumberOfRegisters += RegisterCount; - } - - RegisterGraph *AllocateRegisterGraph(RegisterSet *Set, uint32_t NodeCount) { - RegisterGraph *Graph = new RegisterGraph; - Graph->Set = Set; - Graph->NodeCount = NodeCount; - Graph->MaxNodeCount = NodeCount; - Graph->Nodes = static_cast(calloc(NodeCount, sizeof(RegisterNode))); - - // Initialize nodes - for (uint32_t i = 0; i < NodeCount; ++i) { - Graph->Nodes[i].Register = INVALID_REG; - Graph->Nodes[i].RegisterClass = INVALID_CLASS; - Graph->Nodes[i].InterferenceListSize = 32; - Graph->Nodes[i].InterferenceList = reinterpret_cast(calloc(Graph->Nodes[i].InterferenceListSize, sizeof(uint32_t))); - Graph->Nodes[i].InterferenceCount = 0; - Graph->Nodes[i].Interference.Allocate(NodeCount); - Graph->Nodes[i].Interference.Clear(NodeCount); - } - - return Graph; - } - - void ResetRegisterGraph(RegisterGraph *Graph, uint32_t NodeCount) { - if (NodeCount > Graph->MaxNodeCount) { - uint32_t OldNodeCount = Graph->MaxNodeCount; - Graph->NodeCount = NodeCount; - Graph->MaxNodeCount = NodeCount; - Graph->Nodes = static_cast(realloc(Graph->Nodes, NodeCount * sizeof(RegisterNode))); - - // Initialize nodes - for (uint32_t i = 0; i < OldNodeCount; ++i) { - Graph->Nodes[i].Register = INVALID_REG; - Graph->Nodes[i].RegisterClass = INVALID_CLASS; - Graph->Nodes[i].InterferenceCount = 0; - Graph->Nodes[i].Interference.Realloc(NodeCount); - Graph->Nodes[i].Interference.Clear(NodeCount); - } - - for (uint32_t i = OldNodeCount; i < NodeCount; ++i) { - Graph->Nodes[i].Register = INVALID_REG; - Graph->Nodes[i].RegisterClass = INVALID_CLASS; - Graph->Nodes[i].InterferenceListSize = 32; - Graph->Nodes[i].InterferenceList = reinterpret_cast(calloc(Graph->Nodes[i].InterferenceListSize, sizeof(uint32_t))); - Graph->Nodes[i].InterferenceCount = 0; - Graph->Nodes[i].Interference.Allocate(NodeCount); - Graph->Nodes[i].Interference.Clear(NodeCount); - } - } - else { - // We are only handling a node count of this size right now - Graph->NodeCount = NodeCount; - - // Initialize nodes - for (uint32_t i = 0; i < NodeCount; ++i) { - Graph->Nodes[i].Register = INVALID_REG; - Graph->Nodes[i].RegisterClass = INVALID_CLASS; - Graph->Nodes[i].InterferenceCount = 0; - Graph->Nodes[i].Interference.Clear(NodeCount); - } - } - } - - void FreeRegisterGraph(RegisterGraph *Graph) { - for (uint32_t i = 0; i < Graph->MaxNodeCount; ++i) { - RegisterNode *Node = &Graph->Nodes[i]; - Node->InterferenceCount = 0; - Node->InterferenceListSize = 0; - free(Node->InterferenceList); - Node->Interference.Free(); - } - - free(Graph->Nodes); - Graph->NodeCount = 0; - Graph->MaxNodeCount = 0; - delete Graph; - } - - void SetNodeClass(RegisterGraph *Graph, uint32_t Node, uint32_t Class) { - Graph->Nodes[Node].RegisterClass = Class; - } - - void AddNodeInterference(RegisterGraph *Graph, uint32_t Node1, uint32_t Node2) { - auto AddInterference = [&Graph](uint32_t Node1, uint32_t Node2) { - RegisterNode *Node = &Graph->Nodes[Node1]; - Node->Interference.Set(Node2); - if (Node->InterferenceListSize <= Node->InterferenceCount) { - Node->InterferenceListSize *= 2; - Node->InterferenceList = reinterpret_cast(realloc(Node->InterferenceList, Node->InterferenceListSize * sizeof(uint32_t))); - } - Node->InterferenceList[Node->InterferenceCount] = Node2; - ++Node->InterferenceCount; - }; - - AddInterference(Node1, Node2); - AddInterference(Node2, Node1); - } - - uint32_t GetNodeRegister(RegisterGraph *Graph, uint32_t Node) { - return Graph->Nodes[Node].Register; - } - - static bool HasInterference(RegisterGraph *Graph, RegisterNode *Node, uint32_t Register) { - for (uint32_t i = 0; i < Node->InterferenceCount; ++i) { - RegisterNode *IntNode = &Graph->Nodes[Node->InterferenceList[i]]; - if (IntNode->Register == Register) { - return true; - } - } - - return false; - } - - bool AllocateRegisters(RegisterGraph *Graph) { - Graph->SpillStack.clear(); - for (uint32_t i = 0; i < Graph->NodeCount; ++i) { - RegisterNode *CurrentNode = &Graph->Nodes[i]; - if (CurrentNode->RegisterClass == INVALID_CLASS) - continue; - - uint32_t Reg = ~0U; - RegisterClass *RAClass = &Graph->Set->RegisterClasses[CurrentNode->RegisterClass]; - for (uint32_t ri = 0; ri < RAClass->NumberOfRegisters; ++ri) { - if (!HasInterference(Graph, CurrentNode, RAClass->RegisterBase + ri)) { - Reg = ri; - break; - } - } - - if (Reg == ~0U) { - Graph->SpillStack.emplace_back(SpillStackUnit{i, CurrentNode->RegisterClass}); - } - else { - CurrentNode->Register = RAClass->RegisterBase + Reg; - } - } - - if (!Graph->SpillStack.empty()) { - printf("Couldn't allocate %ld registers\n", Graph->SpillStack.size()); - return false; - } - return true; - } - -} - diff --git a/Source/Interface/Core/RegisterAllocation.h b/Source/Interface/Core/RegisterAllocation.h deleted file mode 100644 index d7c044d5e..000000000 --- a/Source/Interface/Core/RegisterAllocation.h +++ /dev/null @@ -1,30 +0,0 @@ -#pragma once -#include -#include - -namespace FEXCore::RA { -struct RegisterSet; -struct RegisterGraph; -using CrappyBitset = std::vector; - -RegisterSet *AllocateRegisterSet(uint32_t RegisterCount, uint32_t ClassCount); -void FreeRegisterSet(RegisterSet *Set); -void AddRegisters(RegisterSet *Set, uint32_t Class, uint32_t RegistersBase, uint32_t RegisterCount); - -/** - * @name Inference graph handling - * @{ */ - -RegisterGraph *AllocateRegisterGraph(RegisterSet *Set, uint32_t NodeCount); -void FreeRegisterGraph(RegisterGraph *Graph); -void ResetRegisterGraph(RegisterGraph *Graph, uint32_t NodeCount); -void SetNodeClass(RegisterGraph *Graph, uint32_t Node, uint32_t Class); -void AddNodeInterference(RegisterGraph *Graph, uint32_t Node1, uint32_t Node2); -uint32_t GetNodeRegister(RegisterGraph *Graph, uint32_t Node); - -bool AllocateRegisters(RegisterGraph *Graph); - -/** @} */ - -} - diff --git a/Source/Interface/HLE/Syscalls.cpp b/Source/Interface/HLE/Syscalls.cpp index 68d28a074..0b5e041ae 100644 --- a/Source/Interface/HLE/Syscalls.cpp +++ b/Source/Interface/HLE/Syscalls.cpp @@ -8,6 +8,7 @@ #include #include +#include constexpr uint64_t PAGE_SIZE = 4096; @@ -128,7 +129,7 @@ void SyscallHandler::DefaultProgramBreak(FEXCore::Core::InternalThreadState *Thr DefaultProgramBreakAddress = Addr; // Just allocate 1GB of data memory past the default program break location at this point - CTX->MapRegion(Thread, Addr, 0x1000'0000); + CTX->MapRegion(Thread, Addr, 0x1000'0000, true); } uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args) { @@ -154,8 +155,11 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa } // Memory management case SYSCALL_BRK: { - LogMan::Msg::D("\tBRK: 0x%lx - 0x%lx", Args->Argument[1], DataSpace); if (Args->Argument[1] == 0) { // Just wants to get the location of the program break atm + if (DataSpace == 0) { + // XXX: We need to setup our default BRK space first + DefaultProgramBreak(Thread, 0xe000'0000); + } Result = DataSpace + DataSpaceSize; } else { @@ -175,12 +179,8 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa break; } case SYSCALL_MMAP: { - LogMan::Msg::D("\tMMAP( %p, 0x%lx, %d, 0x%x, %d, 0x%lx)", - Args->Argument[1], Args->Argument[2], - Args->Argument[3], Args->Argument[4], - Args->Argument[5], Args->Argument[6]); int Flags = Args->Argument[4]; - int GuestFD = Args->Argument[5]; + int GuestFD = static_cast(Args->Argument[5]); int HostFD = -1; if (GuestFD != -1) { @@ -193,13 +193,13 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa uint64_t Prot = Args->Argument[3]; #ifdef DEBUG_MMAP - FileSizeToUse = Size; - Prot = PROT_READ | PROT_WRITE | PROT_EXEC; +// FileSizeToUse = Size; #endif + Prot = PROT_READ | PROT_WRITE | PROT_EXEC; if (Flags & MAP_FIXED) { Base = Args->Argument[1]; - void *HostPtr = CTX->MemoryMapper.GetPointer(Base); + void *HostPtr = CTX->MemoryMapper.GetPointerSizeCheck(Base, FileSizeToUse); if (!HostPtr) { HostPtr = CTX->MapRegion(Thread, Base, Size, true); } @@ -232,7 +232,13 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa } else { // XXX: MMAP should map memory regions for all threads - void *HostPtr = CTX->MapRegion(Thread, Base, Size, true); + void *HostPtr = CTX->MemoryMapper.GetPointerSizeCheck(Base, FileSizeToUse); + if (!HostPtr) { + HostPtr = CTX->MapRegion(Thread, Base, Size, true); + } + else { + LogMan::Msg::D("\tMapping Fixed pointer in already mapped space: 0x%lx -> %p", Base, HostPtr); + } if (HostFD != -1) { #ifdef DEBUG_MMAP @@ -258,14 +264,12 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa break; } case SYSCALL_MPROTECT: { - LogMan::Msg::D("\tMPROTECT: 0x%x, 0x%lx, 0x%lx", Args->Argument[1], Args->Argument[2], Args->Argument[3]); void *HostPtr = CTX->MemoryMapper.GetPointer(Args->Argument[1]); - Result = mprotect(HostPtr, Args->Argument[2], Args->Argument[3]); +// Result = mprotect(HostPtr, Args->Argument[2], Args->Argument[3]); break; } case SYSCALL_ARCH_PRCTL: { - LogMan::Msg::D("\tPRTCL: 0x%x: 0x%lx", Args->Argument[1], Args->Argument[2]); switch (Args->Argument[1]) { case 0x1001: // ARCH_SET_GS Thread->State.State.gs = Args->Argument[2]; @@ -508,7 +512,21 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa Args->Argument[3], Args->Argument[4]); break; + case SYSCALL_IOCTL: + Result = FM.Ioctl( + Args->Argument[1], + Args->Argument[2], + CTX->MemoryMapper.GetPointer(Args->Argument[3])); + break; + case SYSCALL_TIME: { + time_t *ClockResult = CTX->MemoryMapper.GetPointer(Args->Argument[2]); + Result = time(ClockResult); + // XXX: Debug + // memset(ClockResult, 0, sizeof(time_t)); + // Result = 0; + } + break; case SYSCALL_CLOCK_GETTIME: { timespec *ClockResult = CTX->MemoryMapper.GetPointer(Args->Argument[2]); Result = clock_gettime(Args->Argument[1], ClockResult); @@ -533,25 +551,31 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa break; } case SYSCALL_PRLIMIT64: { - LogMan::Throw::A(Args->Argument[3] == 0, "Guest trying to set limit for %d", Args->Argument[2]); - struct rlimit { - uint64_t rlim_cur; - uint64_t rlim_max; - }; - switch (Args->Argument[2]) { - case 3: { // Stack limits - rlimit *old_limit = CTX->MemoryMapper.GetPointer(Args->Argument[3]); - // Default size - old_limit->rlim_cur = 8 * 1024; - old_limit->rlim_max = ~0ULL; - break; - } - default: LogMan::Msg::A("Unknown PRLimit: %d", Args->Argument[2]); - } - Result = 0; - + LogMan::Throw::A(Args->Argument[3] == 0, "Guest trying to set limit for %d", Args->Argument[2]); + struct rlimit { + uint64_t rlim_cur; + uint64_t rlim_max; + }; + switch (Args->Argument[2]) { + case 3: { // Stack limits + rlimit *old_limit = CTX->MemoryMapper.GetPointer(Args->Argument[3]); + // Default size + old_limit->rlim_cur = 8 * 1024; + old_limit->rlim_max = ~0ULL; + break; + } + default: LogMan::Msg::A("Unknown PRLimit: %d", Args->Argument[2]); + } + Result = 0; break; } + case SYSCALL_UMASK: + // Just say that the mask has always matched what was passed in + Result = Args->Argument[1]; + break; + case SYSCALL_CHDIR: + Result = chdir(CTX->MemoryMapper.GetPointer(Args->Argument[1])); + break; // Currently unhandled // Return fake result case SYSCALL_RT_SIGACTION: diff --git a/Source/Interface/IR/IR.cpp b/Source/Interface/IR/IR.cpp index 3966a5628..7663dd45a 100644 --- a/Source/Interface/IR/IR.cpp +++ b/Source/Interface/IR/IR.cpp @@ -8,12 +8,20 @@ namespace FEXCore::IR { static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, uint64_t Arg) { *out << "0x" << std::hex << Arg; } +static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, RegisterClassType Arg) { + if (Arg == 0) + *out << "GPR"; + else if (Arg == 1) + *out << "FPR"; + else + *out << "Unknown Registerclass " << Arg; +} -static void PrintArg(std::stringstream *out, IRListView const* IR, NodeWrapper Arg) { +static void PrintArg(std::stringstream *out, IRListView const* IR, OrderedNodeWrapper Arg) { uintptr_t Data = IR->GetData(); uintptr_t ListBegin = IR->GetListData(); - OrderedNode *RealNode = reinterpret_cast(Arg.GetPtr(ListBegin)); + OrderedNode *RealNode = Arg.GetNode(ListBegin); auto IROp = RealNode->Op(Data); *out << "%ssa" << std::to_string(Arg.ID()) << " i" << std::dec << (IROp->Size * 8); @@ -23,44 +31,95 @@ static void PrintArg(std::stringstream *out, IRListView const* IR, NodeWr } void Dump(std::stringstream *out, IRListView const* IR) { - uintptr_t Data = IR->GetData(); uintptr_t ListBegin = IR->GetListData(); + uintptr_t DataBegin = IR->GetData(); auto Begin = IR->begin(); - auto End = IR->end(); - while (Begin != End) { - auto Op = Begin(); - OrderedNode *RealNode = reinterpret_cast(Op->GetPtr(ListBegin)); - auto IROp = RealNode->Op(Data); + auto Op = Begin(); - auto Name = FEXCore::IR::GetName(IROp->Op); + OrderedNode *RealNode = Op->GetNode(ListBegin); + auto HeaderOp = RealNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); - if (IROp->HasDest) { - *out << "%ssa" << std::to_string(Op->ID()) << " i" << std::dec << (IROp->Size * 8); - if (IROp->Elements > 1) { - *out << "v" << std::dec << IROp->Elements; + OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); + uint8_t CurrentIndent = 0; + auto AddIndent = [&out, &CurrentIndent]() { + for (uint8_t i = 0; i < CurrentIndent; ++i) { + *out << "\t"; + } + }; + + *out << "(%%ssa" << std::to_string(RealNode->Wrapped(ListBegin).ID()) << ") " << "IRHeader "; + *out << "0x" << std::hex << HeaderOp->Entry << ", "; + *out << "%%ssa" << HeaderOp->Blocks.ID() << ", "; + *out << std::dec << HeaderOp->BlockCount << std::endl; + + while (1) { + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = IR->at(BlockIROp->Begin); + auto CodeLast = IR->at(BlockIROp->Last); + *out << "(%%ssa" << std::to_string(BlockNode->Wrapped(ListBegin).ID()) << ") " << "CodeBlock "; + + *out << "%%ssa" << std::to_string(BlockIROp->Begin.ID()) << ", "; + *out << "%%ssa" << std::to_string(BlockIROp->Last.ID()) << ", "; + *out << "%%ssa" << std::to_string(BlockIROp->Next.ID()) << std::endl; + + while (1) { + OrderedNodeWrapper *CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); + + auto Name = FEXCore::IR::GetName(IROp->Op); + + AddIndent(); + if (IROp->HasDest) { + *out << "%ssa" << std::to_string(CodeOp->ID()) << " i" << std::dec << (IROp->Size * 8); + if (IROp->Elements > 1) { + *out << "v" << std::dec << IROp->Elements; + } + *out << " = "; } - *out << " = "; + else { + *out << "(%%ssa" << std::to_string(CodeOp->ID()) << ") "; + } + + *out << Name; + switch (IROp->Op) { + case IR::OP_BEGINBLOCK: + *out << " %ssa" << std::to_string(CodeOp->ID()); + ++CurrentIndent; + break; + case IR::OP_ENDBLOCK: + --CurrentIndent; + break; + default: break; + } + + #define IROP_ARGPRINTER_HELPER + #include "IRDefines.inc" + default: *out << ""; break; + } + + *out << "\n"; + printf("%s", out->str().c_str()); + *out = std::stringstream{}; + + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; } - *out << Name; - switch (IROp->Op) { - case IR::OP_BEGINBLOCK: - *out << " %ssa" << std::to_string(Op->ID()); + if (BlockIROp->Next.ID() == 0) { break; - default: break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); } - -#define IROP_ARGPRINTER_HELPER -#include "IRDefines.inc" - default: *out << ""; break; - } - - *out << "\n"; - - ++Begin; } - } } diff --git a/Source/Interface/IR/IR.json b/Source/Interface/IR/IR.json index 3e17763ab..bdb31cb25 100644 --- a/Source/Interface/IR/IR.json +++ b/Source/Interface/IR/IR.json @@ -19,6 +19,21 @@ "Ops": { "Dummy": { }, + "IRHeader": { + "Args": [ + "uint64_t", "Entry", + "OrderedNodeWrapper", "Blocks", + "uint32_t", "BlockCount" + ] + }, + "CodeBlock": { + "SSAArgs": "3", + "SSANames": [ + "Begin", + "Last", + "Next" + ] + }, "Constant": { "HasDest": true, "FixedDestSize": "8", @@ -79,6 +94,20 @@ "uint32_t", "Offset" ] }, + "SpillRegister": { + "SSAArgs": "1", + "Args": [ + "uint32_t", "Slot", + "RegisterClassType", "Class" + ] + }, + "FillRegister": { + "HasDest": true, + "Args": [ + "uint32_t", "Slot", + "RegisterClassType", "Class" + ] + }, "LoadFlag": { "HasDest": true, @@ -343,6 +372,11 @@ "SSAArgs": "3" }, + "Print": { + "DispatcherUnary": true, + "SSAArgs": "1" + }, + "CreateVector2": { "HasDest": true, "DestSize": "GetOpSize(ssa0) * 2", @@ -513,11 +547,15 @@ ] }, - "Print": { - "DispatcherUnary": true, - "SSAArgs": "1" + "VExtr": { + "HasDest": true, + "SSAArgs": "2", + "Args": [ + "uint8_t", "RegisterSize", + "uint8_t", "ElementSize", + "uint8_t", "Index" + ] }, - "Last": { "Last": true, "Args": [] diff --git a/Source/Interface/IR/PassManager.cpp b/Source/Interface/IR/PassManager.cpp index bf77e6f24..9b341b508 100644 --- a/Source/Interface/IR/PassManager.cpp +++ b/Source/Interface/IR/PassManager.cpp @@ -1,14 +1,19 @@ #include "Interface/IR/Passes.h" +#include "Interface/IR/Passes/RegisterAllocationPass.h" #include "Interface/IR/PassManager.h" namespace FEXCore::IR { + void PassManager::AddDefaultPasses() { Passes.emplace_back(std::unique_ptr(CreateConstProp())); - Passes.emplace_back(std::unique_ptr(CreateRedundantContextLoadElimination())); + // XXX: Causes corrupted output in test app + // Passes.emplace_back(std::unique_ptr(CreateRedundantContextLoadElimination())); Passes.emplace_back(std::unique_ptr(CreateRedundantFlagCalculationEliminination())); Passes.emplace_back(std::unique_ptr(CreateSyscallOptimization())); Passes.emplace_back(std::unique_ptr(CreatePassDeadContextStoreElimination())); + // If the IR is compacted post-RA then the node indexing gets messed up and the backend isn't able to find the register assigned to a node + // Compact before IR, don't worry about RA generating spills/fills Passes.emplace_back(std::unique_ptr(CreateIRCompaction())); } diff --git a/Source/Interface/IR/PassManager.h b/Source/Interface/IR/PassManager.h index 379e5cbb8..f07a06714 100644 --- a/Source/Interface/IR/PassManager.h +++ b/Source/Interface/IR/PassManager.h @@ -18,6 +18,9 @@ class PassManager final { public: void AddDefaultPasses(); void AddDefaultValidationPasses(); + void InsertPass(Pass *Pass) { + Passes.emplace_back(Pass); + } bool Run(OpDispatchBuilder *Disp); private: diff --git a/Source/Interface/IR/Passes/ConstProp.cpp b/Source/Interface/IR/Passes/ConstProp.cpp index 7793f48b7..979cd7cc4 100644 --- a/Source/Interface/IR/Passes/ConstProp.cpp +++ b/Source/Interface/IR/Passes/ConstProp.cpp @@ -14,33 +14,60 @@ bool ConstProp::Run(OpDispatchBuilder *Disp) { uintptr_t ListBegin = CurrentIR.GetListData(); uintptr_t DataBegin = CurrentIR.GetData(); - IR::NodeWrapperIterator Begin = CurrentIR.begin(); - IR::NodeWrapperIterator End = CurrentIR.end(); + auto Begin = CurrentIR.begin(); + auto Op = Begin(); - while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + OrderedNode *RealNode = Op->GetNode(ListBegin); + auto HeaderOp = RealNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); - switch (IROp->Op) { - case OP_ZEXT: { - auto Op = IROp->C(); - uint64_t Constant; - if (Disp->IsValueConstant(Op->Header.Args[0], &Constant)) { - uint64_t NewConstant = Constant & ((1ULL << Op->SrcSize) - 1); - auto ConstantVal = Disp->_Constant(NewConstant); - Disp->ReplaceAllUsesWith(RealNode, ConstantVal); - Changed = true; + OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); + + auto OriginalWriteCursor = Disp->GetWriteCursor(); + + while (1) { + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + while (1) { + auto CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); + switch (IROp->Op) { + case OP_ZEXT: { + auto Op = IROp->C(); + uint64_t Constant; + if (Disp->IsValueConstant(Op->Header.Args[0], &Constant)) { + uint64_t NewConstant = Constant & ((1ULL << Op->SrcSize) - 1); + Disp->SetWriteCursor(CodeNode); + auto ConstantVal = Disp->_Constant(NewConstant); + Disp->ReplaceAllUsesWith(CodeNode, ConstantVal); + Changed = true; + } + break; + } + default: break; } - break; - } - default: break; + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; } - ++Begin; + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + } } + Disp->SetWriteCursor(OriginalWriteCursor); + return Changed; } diff --git a/Source/Interface/IR/Passes/DeadContextStoreElimination.cpp b/Source/Interface/IR/Passes/DeadContextStoreElimination.cpp index f373a13d3..bb431e261 100644 --- a/Source/Interface/IR/Passes/DeadContextStoreElimination.cpp +++ b/Source/Interface/IR/Passes/DeadContextStoreElimination.cpp @@ -42,68 +42,88 @@ static bool IsGPR(uint32_t Offset, uint8_t *greg) { bool RCLE::Run(OpDispatchBuilder *Disp) { bool Changed = false; auto CurrentIR = Disp->ViewIR(); + std::array LastValidGPRStores{}; + auto OriginalWriteCursor = Disp->GetWriteCursor(); + uintptr_t ListBegin = CurrentIR.GetListData(); uintptr_t DataBegin = CurrentIR.GetData(); - IR::NodeWrapperIterator Begin = CurrentIR.begin(); - IR::NodeWrapperIterator End = CurrentIR.end(); + auto Begin = CurrentIR.begin(); + auto Op = Begin(); - std::array LastValidGPRStores{}; + OrderedNode *RealNode = Op->GetNode(ListBegin); + auto HeaderOp = RealNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); - while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); - if (IROp->Op == OP_BEGINBLOCK || - IROp->Op == OP_ENDBLOCK || - IROp->Op == OP_JUMP || - IROp->Op == OP_CONDJUMP || - IROp->Op == OP_EXITFUNCTION) { - // We don't track across block boundaries - LastValidGPRStores.fill(nullptr); - } + while (1) { + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); - if (IROp->Op == OP_STORECONTEXT) { - auto Op = IROp->CW(); - // Make sure we are within GREG state - uint8_t greg = ~0; - if (IsAlignedGPR(Op->Size, Op->Offset, &greg)) { - FEXCore::IR::IROp_Header *ArgOp = reinterpret_cast(Op->Header.Args[0].GetPtr(ListBegin))->Op(DataBegin); - // Ensure we aren't doing a mismatched store - // XXX: We should really catch this in IR validation - if (ArgOp->Size == 8) { - LastValidGPRStores[greg] = &Op->Header.Args[0]; - } - else { - LastValidGPRStores[greg] = nullptr; - } - } else if (IsGPR(Op->Offset, &greg)) { - // If we aren't overwriting the whole state then we don't want to track this value - LastValidGPRStores[greg] = nullptr; - } - } + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + while (1) { + auto CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); - if (IROp->Op == OP_LOADCONTEXT) { - auto Op = IROp->C(); - - // Make sure we are within GREG state - uint8_t greg = ~0; - if (IsAlignedGPR(Op->Size, Op->Offset, &greg)) { - if (LastValidGPRStores[greg] != nullptr) { - // If the last store matches this load value then we can replace the loaded value with the previous valid one - auto MovVal = Disp->_Mov(reinterpret_cast(LastValidGPRStores[greg]->GetPtr(ListBegin))); - Disp->ReplaceAllUsesWith(RealNode, MovVal); - Changed = true; - } - } else if (IsGPR(Op->Offset, &greg)) { + if (IROp->Op == OP_STORECONTEXT) { + auto Op = IROp->CW(); + // Make sure we are within GREG state + uint8_t greg = ~0; + if (IsAlignedGPR(Op->Size, Op->Offset, &greg)) { + LastValidGPRStores[greg] = Op->Header.Args[0].NodeOffset; + } else if (IsGPR(Op->Offset, &greg)) { // If we aren't overwriting the whole state then we don't want to track this value - LastValidGPRStores[greg] = nullptr; // 0 is invalid + LastValidGPRStores[greg] = 0; + } } + + if (IROp->Op == OP_LOADCONTEXT) { + auto Op = IROp->C(); + + // Make sure we are within GREG state + uint8_t greg = ~0; + if (IsAlignedGPR(Op->Size, Op->Offset, &greg)) { + if (LastValidGPRStores[greg] != 0) { + // If the last store matches this load value then we can replace the loaded value with the previous valid one + if (1) { + Disp->SetWriteCursor(CodeNode); + auto MovVal = Disp->_Mov(OrderedNodeWrapper::WrapOffset(LastValidGPRStores[greg]).GetNode(ListBegin)); + Disp->ReplaceAllUsesWith(CodeNode, MovVal); + } + else { + Disp->ReplaceAllUsesWith(CodeNode, OrderedNodeWrapper::WrapOffset(LastValidGPRStores[greg]).GetNode(ListBegin)); + } + Changed = true; + } + } else if (IsGPR(Op->Offset, &greg)) { + // If we aren't overwriting the whole state then we don't want to track this value + LastValidGPRStores[greg] = 0; // 0 is invalid + } + } + + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; } - ++Begin; + + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + } + + // We don't track across block boundaries + LastValidGPRStores.fill(0); } + Disp->SetWriteCursor(OriginalWriteCursor); + return Changed; } diff --git a/Source/Interface/IR/Passes/IRCompaction.cpp b/Source/Interface/IR/Passes/IRCompaction.cpp index e36a3ad62..81dd2d65e 100644 --- a/Source/Interface/IR/Passes/IRCompaction.cpp +++ b/Source/Interface/IR/Passes/IRCompaction.cpp @@ -12,7 +12,7 @@ public: private: OpDispatchBuilder LocalBuilder; - std::vector NodeLocationRemapper; + std::vector NodeLocationRemapper; }; IRCompaction::IRCompaction() { @@ -21,15 +21,16 @@ IRCompaction::IRCompaction() { bool IRCompaction::Run(OpDispatchBuilder *Disp) { auto CurrentIR = Disp->ViewIR(); - auto LocalIR = LocalBuilder.ViewIR(); - uint32_t NodeCount = LocalIR.GetListSize() / sizeof(OrderedNode); + uint32_t NodeCount = CurrentIR.GetSSACount(); - // Reset our local working list - LocalBuilder.ResetWorkingList(); if (NodeLocationRemapper.size() < NodeCount) { NodeLocationRemapper.resize(NodeCount); } - memset(&NodeLocationRemapper.at(0), 0xFF, NodeCount * sizeof(IR::NodeWrapper::NodeOffsetType)); + memset(&NodeLocationRemapper.at(0), 0xFF, NodeCount * sizeof(IR::OrderedNodeWrapper::NodeOffsetType)); + + // Reset our local working list + LocalBuilder.ResetWorkingList(); + auto LocalIR = LocalBuilder.ViewIR(); uintptr_t LocalListBegin = LocalIR.GetListData(); uintptr_t LocalDataBegin = LocalIR.GetData(); @@ -37,89 +38,201 @@ bool IRCompaction::Run(OpDispatchBuilder *Disp) { uintptr_t ListBegin = CurrentIR.GetListData(); uintptr_t DataBegin = CurrentIR.GetData(); - IR::NodeWrapperIterator Begin = CurrentIR.begin(); - IR::NodeWrapperIterator End = CurrentIR.end(); + auto HeaderIterator = CurrentIR.begin(); + OrderedNodeWrapper *HeaderNodeWrapper = HeaderIterator(); + OrderedNode *HeaderNode = HeaderNodeWrapper->GetNode(ListBegin); + auto HeaderOp = HeaderNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); - // This compaction pass is something that we need to ensure correct ordering and distances between IROps\ + // This compaction pass is something that we need to ensure correct ordering and distances between IROps // Later on we assume that an IROp's SSA value live range is its Node locations // // RA distance calculation is calculated purely on the Node locations - // So we just need to reorder those + // So we need to reorder those // // Additionally there may be some dead ops hanging out in the IR list that are orphaned. // These can also be dropped during this pass - while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + // First thing is first, we need to do some housekeeping + // Create the IRHeader op + // Create the codeblocks + // Then create all the ops inside the code blocks - if (IROp->HasDest && RealNode->GetUses() == 0) { - // Should this be in a dedicated DCE pass? - ++Begin; - continue; + auto LocalHeaderOp = LocalBuilder._IRHeader(HeaderOp->Entry, OrderedNodeWrapper::WrapOffset(0), HeaderOp->BlockCount); + NodeLocationRemapper[HeaderNode->Wrapped(ListBegin).ID()] = LocalHeaderOp.Node->Wrapped(LocalListBegin).ID(); + + struct CodeBlockData { + OrderedNode *OldNode; + OrderedNode *NewNode; + }; + std::vector GeneratedCodeBlocks{}; + + { + // Generate our codeblocks and link them together + OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); + OrderedNode* PrevCodeBlock{}; + while (1) { + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); + + auto LocalBlockIRNode = LocalBuilder.CreateCodeNode(); + NodeLocationRemapper[BlockNode->Wrapped(ListBegin).ID()] = LocalBlockIRNode->Wrapped(LocalListBegin).ID(); + GeneratedCodeBlocks.emplace_back(CodeBlockData{BlockNode, LocalBlockIRNode}); + + if (PrevCodeBlock) { + auto PrevLocalBlockIROp = PrevCodeBlock->Op(LocalDataBegin)->CW(); + PrevLocalBlockIROp->Next = LocalBlockIRNode->Wrapped(LocalListBegin); + } + + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + PrevCodeBlock = LocalBlockIRNode; + } } - size_t OpSize = FEXCore::IR::GetSize(IROp->Op); - // Allocate the ops locally for our local dispatch - auto LocalPair = LocalBuilder.AllocateRawOp(OpSize); - IR::NodeWrapper LocalNodeWrapper = LocalPair.Node->Wrapped(LocalListBegin); - - // Copy over the op - memcpy(LocalPair.first, IROp, OpSize); - - // Set our map remapper to map the new location - // Even nodes that don't have a destination need to be in this map - // Need to be able to remap branch targets any other bits - NodeLocationRemapper[WrapperOp->ID()] = LocalNodeWrapper.ID(); - ++Begin; + // Link the IRHeader to the first code block + LocalHeaderOp.first->Blocks = GeneratedCodeBlocks[0].NewNode->Wrapped(LocalListBegin); } - Begin = CurrentIR.begin(); - while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + { + // Copy all of our IR ops over to the new location + for (auto &Block : GeneratedCodeBlocks) { + auto BlockIROp = Block.OldNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); - if (IROp->HasDest && RealNode->GetUses() == 0) { - // Should this be in a dedicated DCE pass? - ++Begin; - continue; + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + + CodeBlockData FirstNode{}; + CodeBlockData LastNode{}; + uint32_t i {}; + while (1) { + auto CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); + LogMan::Throw::A(IROp->Op != OP_IRHEADER, "%%ssa%d ended up being IRHeader. Shouldn't have hit this", CodeOp->ID()); + + size_t OpSize = FEXCore::IR::GetSize(IROp->Op); + + // Allocate the ops locally for our local dispatch + auto LocalPair = LocalBuilder.AllocateRawOp(OpSize); + IR::OrderedNodeWrapper LocalNodeWrapper = LocalPair.Node->Wrapped(LocalListBegin); + + // Copy over the op + memcpy(LocalPair.first, IROp, OpSize); + LogMan::Throw::A(LocalPair.first->Op == IROp->Op, "What. How did this fail"); + + // Set our map remapper to map the new location + // Even nodes that don't have a destination need to be in this map + // Need to be able to remap branch targets any other bits + NodeLocationRemapper[CodeOp->ID()] = LocalNodeWrapper.ID(); + if (i == 0) { + FirstNode.OldNode = CodeNode; + FirstNode.NewNode = LocalPair.Node; + } + + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + LastNode.OldNode = CodeNode; + LastNode.NewNode = LocalPair.Node; + break; + } + ++CodeBegin; + ++i; + } + + // Set the code block's begin and end correctly + auto NewBlockIROp = Block.NewNode->Op(LocalDataBegin)->CW(); + NewBlockIROp->Begin = FirstNode.NewNode->Wrapped(LocalListBegin); + NewBlockIROp->Last = LastNode.NewNode->Wrapped(LocalListBegin); } - - NodeWrapper LocalNodeWrapper = NodeWrapper::WrapOffset(NodeLocationRemapper[WrapperOp->ID()] * sizeof(OrderedNode)); - OrderedNode *LocalNode = reinterpret_cast(LocalNodeWrapper.GetPtr(LocalListBegin)); - FEXCore::IR::IROp_Header *LocalIROp = LocalNode->Op(LocalDataBegin); - - // Now that we have the op copied over, we need to modify SSA values to point to the new correct locations - for (uint8_t i = 0; i < IROp->NumArgs; ++i) { - NodeWrapper OldArg = IROp->Args[i]; - LogMan::Throw::A(NodeLocationRemapper[OldArg.ID()] != ~0U, "Tried remapping unfound node"); - LocalIROp->Args[i].NodeOffset = NodeLocationRemapper[OldArg.ID()] * sizeof(OrderedNode); - } - ++Begin; } -// uintptr_t OldListSize = CurrentIR.GetListSize(); -// uintptr_t OldDataSize = CurrentIR.GetDataSize(); + { + // Fixup the arguments of all the IROps + for (auto &Block : GeneratedCodeBlocks) { + auto BlockIROp = Block.OldNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + while (1) { + auto CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); + + OrderedNodeWrapper LocalNodeWrapper = OrderedNodeWrapper::WrapOffset(NodeLocationRemapper[CodeOp->ID()] * sizeof(OrderedNode)); + OrderedNode *LocalNode = LocalNodeWrapper.GetNode(LocalListBegin); + FEXCore::IR::IROp_Header *LocalIROp = LocalNode->Op(LocalDataBegin); + + // Now that we have the op copied over, we need to modify SSA values to point to the new correct locations + for (uint8_t i = 0; i < IROp->NumArgs; ++i) { + uint32_t OldArg = IROp->Args[i].ID(); + LogMan::Throw::A(NodeLocationRemapper[OldArg] != ~0U, "Tried remapping unfound node %%ssa%d", OldArg); + LocalIROp->Args[i].NodeOffset = NodeLocationRemapper[OldArg] * sizeof(OrderedNode); + } + + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; + } + } + } + +// XXX: Example for iterating blocks +// OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); +// while (1) { +// auto BlockIROp = BlockNode->Op(DataBegin)->CW(); +// LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); // -// uintptr_t NewListSize = LocalIR.GetListSize(); -// uintptr_t NewDataSize = LocalIR.GetDataSize(); +// // We grab these nodes this way so we can iterate easily +// auto CodeBegin = CurrentIR.at(BlockIROp->Begin); +// auto CodeLast = CurrentIR.at(BlockIROp->Last); +// while (1) { +// auto CodeOp = CodeBegin(); +// OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); +// auto IROp = CodeNode->Op(DataBegin); // -// if (NewListSize < OldListSize || -// NewDataSize < OldDataSize) { -// if (NewListSize < OldListSize) { -// LogMan::Msg::D("Shaved %ld bytes off the list size", OldListSize - NewListSize); +// // CodeLast is inclusive. So we still need to dump the CodeLast op as well +// if (CodeBegin == CodeLast) { +// break; +// } +// ++CodeBegin; // } -// if (NewDataSize < OldDataSize) { -// LogMan::Msg::D("Shaved %ld bytes off the data size", OldDataSize - NewDataSize); +// +// if (BlockIROp->Next.ID() == 0) { +// break; +// } else { +// BlockNode = BlockIROp->Next.GetNode(ListBegin); // } // } -// if (NewListSize > OldListSize || -// NewDataSize > OldDataSize) { -// LogMan::Msg::A("Whoa. Compaction made the IR a different size when it shouldn't have. 0x%lx > 0x%lx or 0x%lx > 0x%lx",NewListSize, OldListSize, NewDataSize, OldDataSize); -// } + // uintptr_t OldListSize = CurrentIR.GetListSize(); + // uintptr_t OldDataSize = CurrentIR.GetDataSize(); + + // uintptr_t NewListSize = LocalIR.GetListSize(); + // uintptr_t NewDataSize = LocalIR.GetDataSize(); + + // if (NewListSize < OldListSize || + // NewDataSize < OldDataSize) { + // if (NewListSize < OldListSize) { + // LogMan::Msg::D("Shaved %ld bytes off the list size", OldListSize - NewListSize); + // } + // if (NewDataSize < OldDataSize) { + // LogMan::Msg::D("Shaved %ld bytes off the data size", OldDataSize - NewDataSize); + // } + // } + + // if (NewListSize > OldListSize || + // NewDataSize > OldDataSize) { + // LogMan::Msg::A("Whoa. Compaction made the IR a different size when it shouldn't have. 0x%lx > 0x%lx or 0x%lx > 0x%lx",NewListSize, OldListSize, NewDataSize, OldDataSize); + // } Disp->CopyData(LocalBuilder); diff --git a/Source/Interface/IR/Passes/IRValidation.cpp b/Source/Interface/IR/Passes/IRValidation.cpp index 7e2bc710b..a7b6b7d70 100644 --- a/Source/Interface/IR/Passes/IRValidation.cpp +++ b/Source/Interface/IR/Passes/IRValidation.cpp @@ -6,13 +6,13 @@ namespace FEXCore::IR::Validation { struct BlockInfo { - IR::NodeWrapper *Begin; - IR::NodeWrapper *End; + IR::OrderedNodeWrapper *Begin; + IR::OrderedNodeWrapper *End; bool HasExit; - std::vector Predecessors; - std::vector Successors; + std::vector Predecessors; + std::vector Successors; }; class IRValidation final : public FEXCore::IR::Pass { @@ -20,7 +20,7 @@ public: bool Run(OpDispatchBuilder *Disp) override; private: - std::unordered_map OffsetToBlockMap; + std::unordered_map OffsetToBlockMap; }; bool IRValidation::Run(OpDispatchBuilder *Disp) { @@ -37,8 +37,8 @@ bool IRValidation::Run(OpDispatchBuilder *Disp) { std::ostringstream Errors; while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); uint8_t OpSize = IROp->Size; @@ -56,7 +56,7 @@ bool IRValidation::Run(OpDispatchBuilder *Disp) { } for (uint8_t i = 0; i < IROp->NumArgs; ++i) { - NodeWrapper Arg = IROp->Args[i]; + OrderedNodeWrapper Arg = IROp->Args[i]; if (Arg.ID() == 0) { HadError |= true; Errors << "Op" << WrapperOp->ID() <<": Arg[" << i << "] has invalid target of %ssa0" << std::endl; @@ -70,7 +70,7 @@ bool IRValidation::Run(OpDispatchBuilder *Disp) { Errors << "BasicBlock " << WrapperOp->ID() << ": Begin in middle of block" << std::endl; } - auto Block = OffsetToBlockMap.try_emplace(WrapperOp->ID(), BlockInfo{}).first; + auto Block = OffsetToBlockMap.try_emplace(WrapperOp->ID()).first; CurrentBlock = &Block->second; CurrentBlock->Begin = WrapperOp; InBlock = true; @@ -109,14 +109,14 @@ bool IRValidation::Run(OpDispatchBuilder *Disp) { CurrentBlock->Successors.emplace_back(IterLocation()); } - OrderedNode *TargetNode = reinterpret_cast(IterLocation()->GetPtr(ListBegin)); + OrderedNode *TargetNode = IterLocation()->GetNode(ListBegin); FEXCore::IR::IROp_Header *TargetOp = TargetNode->Op(DataBegin); HadError |= TargetOp->Op != OP_BEGINBLOCK; if (TargetOp->Op != OP_BEGINBLOCK) { Errors << "CondJump " << WrapperOp->ID() << ": CondJump to Op that isn't the begining of a block" << std::endl; } else { - auto Block = OffsetToBlockMap.try_emplace(IterLocation()->NodeOffset, BlockInfo{}).first; + auto Block = OffsetToBlockMap.try_emplace(IterLocation()->NodeOffset).first; Block->second.Predecessors.emplace_back(CurrentBlock->Begin); } @@ -130,14 +130,14 @@ bool IRValidation::Run(OpDispatchBuilder *Disp) { CurrentBlock->Successors.emplace_back(IterLocation()); } - OrderedNode *TargetNode = reinterpret_cast(IterLocation()->GetPtr(ListBegin)); + OrderedNode *TargetNode = IterLocation()->GetNode(ListBegin); FEXCore::IR::IROp_Header *TargetOp = TargetNode->Op(DataBegin); HadError |= TargetOp->Op != OP_BEGINBLOCK; if (TargetOp->Op != OP_BEGINBLOCK) { Errors << "Jump " << WrapperOp->ID() << ": Jump to Op that isn't the begining of a block" << std::endl; } else { - auto Block = OffsetToBlockMap.try_emplace(IterLocation()->NodeOffset, BlockInfo{}).first; + auto Block = OffsetToBlockMap.try_emplace(IterLocation()->NodeOffset).first; Block->second.Predecessors.emplace_back(CurrentBlock->Begin); } break; diff --git a/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index 51df22214..22dff4862 100644 --- a/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -9,51 +9,70 @@ public: }; bool RedundantFlagCalculationEliminination::Run(OpDispatchBuilder *Disp) { + std::array LastValidFlagStores{}; + bool Changed = false; auto CurrentIR = Disp->ViewIR(); uintptr_t ListBegin = CurrentIR.GetListData(); uintptr_t DataBegin = CurrentIR.GetData(); - IR::NodeWrapperIterator Begin = CurrentIR.begin(); - IR::NodeWrapperIterator End = CurrentIR.end(); + auto Begin = CurrentIR.begin(); + auto Op = Begin(); - std::array LastValidFlagStores{}; + OrderedNode *RealNode = Op->GetNode(ListBegin); + auto HeaderOp = RealNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); - while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); - if (IROp->Op == OP_BEGINBLOCK || - IROp->Op == OP_ENDBLOCK || - IROp->Op == OP_JUMP || - IROp->Op == OP_CONDJUMP || - IROp->Op == OP_EXITFUNCTION) { - // We don't track across block boundaries - LastValidFlagStores.fill(nullptr); - } + while (1) { + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); - if (IROp->Op == OP_STOREFLAG) { - auto Op = IROp->CW(); + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + while (1) { + auto CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); - // If we have had a valid flag store previously and it hasn't been touched until this new store - // Then just delete the old one and let DCE to take care of the rest - if (LastValidFlagStores[Op->Flag] != nullptr) { - Disp->Unlink(LastValidFlagStores[Op->Flag]); - Changed = true; + if (IROp->Op == OP_STOREFLAG) { + auto Op = IROp->CW(); + + // If we have had a valid flag store previously and it hasn't been touched until this new store + // Then just delete the old one and let DCE to take care of the rest + if (LastValidFlagStores[Op->Flag] != nullptr) { + Disp->Unlink(LastValidFlagStores[Op->Flag]); + Changed = true; + } + + // Set this node as the last one valid for this flag + LastValidFlagStores[Op->Flag] = RealNode; + } + else if (IROp->Op == OP_LOADFLAG) { + auto Op = IROp->CW(); + + // If we loaded a flag then we can't track past this + LastValidFlagStores[Op->Flag] = nullptr; } - // Set this node as the last one valid for this flag - LastValidFlagStores[Op->Flag] = RealNode; - } - else if (IROp->Op == OP_LOADFLAG) { - auto Op = IROp->CW(); - // If we loaded a flag then we can't track past this - LastValidFlagStores[Op->Flag] = nullptr; + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; } - ++Begin; + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + } + + // We don't track across block boundaries + LastValidFlagStores.fill(nullptr); } return Changed; diff --git a/Source/Interface/IR/Passes/RegisterAllocationPass.cpp b/Source/Interface/IR/Passes/RegisterAllocationPass.cpp index 84f391d0a..8d8bdd53d 100644 --- a/Source/Interface/IR/Passes/RegisterAllocationPass.cpp +++ b/Source/Interface/IR/Passes/RegisterAllocationPass.cpp @@ -1,15 +1,13 @@ #include "Common/BitSet.h" -#include "Common/Profiler.h" #include "Interface/IR/Passes/RegisterAllocationPass.h" #include "Interface/Core/OpcodeDispatcher.h" #include -PROFILER_DEFINE(RA, "Passes", "RA", 0); - namespace FEXCore::IR { constexpr uint32_t INVALID_REG = ~0U; constexpr uint32_t INVALID_CLASS = ~0U; + constexpr uint32_t DEFAULT_INTERFERENCE_LIST_SIZE = 128; struct Register { }; @@ -17,7 +15,7 @@ namespace FEXCore::IR { struct RegisterClass { uint32_t RegisterBase; uint32_t NumberOfRegisters{0}; - BitSet Registers; + BitSet Registers; }; struct RegisterAllocationPass::RegisterNode { @@ -26,7 +24,7 @@ namespace FEXCore::IR { uint32_t InterferenceCount; uint32_t InterferenceListSize; uint32_t *InterferenceList; - BitSet Interference; + BitSet Interference; }; static_assert(std::is_pod::value, "We want this to be POD"); @@ -96,7 +94,7 @@ namespace FEXCore::IR { for (uint32_t i = 0; i < NodeCount; ++i) { Graph->Nodes[i].Register = INVALID_REG; Graph->Nodes[i].RegisterClass = INVALID_CLASS; - Graph->Nodes[i].InterferenceListSize = 32; + Graph->Nodes[i].InterferenceListSize = DEFAULT_INTERFERENCE_LIST_SIZE; Graph->Nodes[i].InterferenceList = reinterpret_cast(calloc(Graph->Nodes[i].InterferenceListSize, sizeof(uint32_t))); Graph->Nodes[i].InterferenceCount = 0; Graph->Nodes[i].Interference.Allocate(NodeCount); @@ -125,7 +123,7 @@ namespace FEXCore::IR { for (uint32_t i = OldNodeCount; i < NodeCount; ++i) { Graph->Nodes[i].Register = INVALID_REG; Graph->Nodes[i].RegisterClass = INVALID_CLASS; - Graph->Nodes[i].InterferenceListSize = 32; + Graph->Nodes[i].InterferenceListSize = DEFAULT_INTERFERENCE_LIST_SIZE; Graph->Nodes[i].InterferenceList = reinterpret_cast(calloc(Graph->Nodes[i].InterferenceListSize, sizeof(uint32_t))); Graph->Nodes[i].InterferenceCount = 0; Graph->Nodes[i].Interference.Allocate(NodeCount); @@ -165,22 +163,6 @@ namespace FEXCore::IR { Graph->Nodes[Node].RegisterClass = Class; } - void RegisterAllocationPass::AddNodeInterference(uint32_t Node1, uint32_t Node2) { - auto AddInterference = [&](uint32_t Node1, uint32_t Node2) { - RegisterNode *Node = &Graph->Nodes[Node1]; - Node->Interference.Set(Node2); - if (Node->InterferenceListSize <= Node->InterferenceCount) { - Node->InterferenceListSize *= 2; - Node->InterferenceList = reinterpret_cast(realloc(Node->InterferenceList, Node->InterferenceListSize * sizeof(uint32_t))); - } - Node->InterferenceList[Node->InterferenceCount] = Node2; - ++Node->InterferenceCount; - }; - - AddInterference(Node1, Node2); - AddInterference(Node2, Node1); - } - uint32_t RegisterAllocationPass::GetNodeRegister(uint32_t Node) { return Graph->Nodes[Node].Register; } @@ -210,8 +192,8 @@ namespace FEXCore::IR { while (Begin != End) { using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); if (IROp->HasDest) { @@ -266,7 +248,8 @@ namespace FEXCore::IR { } } - void RegisterAllocationPass::CalculateLiveRange(IRListView *CurrentIR, uint32_t Nodes) { + void RegisterAllocationPass::CalculateLiveRange(IRListView *CurrentIR) { + size_t Nodes = CurrentIR->GetSSACount(); if (Nodes > LiveRanges.size()) { LiveRanges.resize(Nodes); } @@ -281,14 +264,14 @@ namespace FEXCore::IR { constexpr uint32_t DEFAULT_REMAT_COST = 1000; while (Begin != End) { using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); - uint32_t Node = WrapperOp->ID(); // If the destination hasn't yet been set then set it now - if (IROp->HasDest && LiveRanges[Node].Begin == ~0U) { + if (IROp->HasDest) { + LogMan::Throw::A(LiveRanges[Node].Begin == ~0U, "Node begin already defined?"); LiveRanges[Node].Begin = Node; // Default to ending right where it starts LiveRanges[Node].End = Node; @@ -312,13 +295,27 @@ namespace FEXCore::IR { ++Begin; } + } + + void RegisterAllocationPass::CalculateNodeInterference(uint32_t NodeCount) { + auto AddInterference = [&](uint32_t Node1, uint32_t Node2) { + RegisterNode *Node = &Graph->Nodes[Node1]; + Node->Interference.Set(Node2); + if (Node->InterferenceListSize <= Node->InterferenceCount) { + Node->InterferenceListSize *= 2; + Node->InterferenceList = reinterpret_cast(realloc(Node->InterferenceList, Node->InterferenceListSize * sizeof(uint32_t))); + } + Node->InterferenceList[Node->InterferenceCount] = Node2; + ++Node->InterferenceCount; + }; // Now that we have all the live ranges calculated we need to add them to our interference graph - for (uint32_t i = 0; i < Nodes; ++i) { - for (uint32_t j = i + 1; j < Nodes; ++j) { + for (uint32_t i = 0; i < NodeCount; ++i) { + for (uint32_t j = i + 1; j < NodeCount; ++j) { if (!(LiveRanges[i].Begin >= LiveRanges[j].End || LiveRanges[j].Begin >= LiveRanges[i].End)) { - AddNodeInterference(i, j); + AddInterference(i, j); + AddInterference(j, i); } } } @@ -342,10 +339,10 @@ namespace FEXCore::IR { if (Reg == ~0U) { auto RegisterNode = GetRegisterNode(i); - LogMan::Msg::E("\t%%ssa%d with no-RA has live range [%d, %d): Remat cost: %d", i, LiveRanges[i].Begin, LiveRanges[i].End, LiveRanges[i].RematCost); + // LogMan::Msg::E("\t%%ssa%d with no-RA has live range [%d, %d): Remat cost: %d", i, LiveRanges[i].Begin, LiveRanges[i].End, LiveRanges[i].RematCost); for (uint32_t j = 0; j < RegisterNode->InterferenceCount; ++j) { uint32_t InterferenceNode = RegisterNode->InterferenceList[j]; - LogMan::Msg::E("\t\tInterferes with %%ssa%d: live range[%d, %d): Remat cost: %d", InterferenceNode, LiveRanges[InterferenceNode].Begin, LiveRanges[InterferenceNode].End, LiveRanges[InterferenceNode].RematCost); + // LogMan::Msg::E("\t\tInterferes with %%ssa%d: live range[%d, %d): Remat cost: %d", InterferenceNode, LiveRanges[InterferenceNode].Begin, LiveRanges[InterferenceNode].End, LiveRanges[InterferenceNode].RematCost); } Graph->SpillStack.emplace_back(SpillStackUnit{i, CurrentNode->RegisterClass}); } @@ -375,8 +372,8 @@ namespace FEXCore::IR { while (Begin != End) { using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); LogMan::Msg::D("\t\t%%ssa%d Do we have %%ssa%d", WrapperOp->ID(), Node->Wrapped(ListBegin).ID()); @@ -406,8 +403,8 @@ namespace FEXCore::IR { auto LastCursor = Disp->GetWriteCursor(); while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); auto Iter = IsInSpillStack(&Graph->SpillStack, WrapperOp->ID()); if (Iter != Graph->SpillStack.end()) { @@ -424,8 +421,8 @@ namespace FEXCore::IR { // We want to end the live range of this value here and continue it on first use auto ConstantRegisterNode = GetRegisterNode(InterferenceNode); auto *ConstantLiveRange = &LiveRanges[InterferenceNode]; - NodeWrapper ConstantOp = NodeWrapper::WrapOffset(InterferenceNode * sizeof(OrderedNode)); - OrderedNode *ConstantNode = reinterpret_cast(ConstantOp.GetPtr(ListBegin)); + OrderedNodeWrapper ConstantOp = OrderedNodeWrapper::WrapOffset(InterferenceNode * sizeof(OrderedNode)); + OrderedNode *ConstantNode = ConstantOp.GetNode(ListBegin); FEXCore::IR::IROp_Constant const *ConstantIROp = ConstantNode->Op(DataBegin)->C(); LogMan::Throw::A(ConstantIROp->Header.Op == OP_CONSTANT, "This needs to be const"); @@ -435,8 +432,8 @@ namespace FEXCore::IR { auto FirstUseLocation = FindFirstUse(Disp, ConstantNode, NextIter, End); if (FirstUseLocation != End) { // LogMan::Throw::A(FirstUseLocation != End, "Failure to find op use"); - NodeWrapper *FirstUseOp = FirstUseLocation(); - OrderedNode *FirstUseOrderedNode = reinterpret_cast(FirstUseOp->GetPtr(ListBegin)); + OrderedNodeWrapper *FirstUseOp = FirstUseLocation(); + OrderedNode *FirstUseOrderedNode = FirstUseOp->GetNode(ListBegin); Disp->SetWriteCursor(FirstUseOrderedNode); auto FilledConstant = Disp->_Constant(ConstantIROp->Constant); Disp->ReplaceAllUsesWithInclusive(ConstantNode, FilledConstant, FirstUseLocation, End); @@ -507,13 +504,13 @@ namespace FEXCore::IR { auto *InterferenceLiveRange = &LiveRanges[InterferenceNode]; // If the interference's live range is past this op's live range then we can dump it if (1) { - NodeWrapper InterferenceOp = NodeWrapper::WrapOffset(InterferenceNode * sizeof(OrderedNode)); - OrderedNode *InterferenceOrderedNode = reinterpret_cast(InterferenceOp.GetPtr(ListBegin)); + OrderedNodeWrapper InterferenceOp = OrderedNodeWrapper::WrapOffset(InterferenceNode * sizeof(OrderedNode)); + OrderedNode *InterferenceOrderedNode = InterferenceOp.GetNode(ListBegin); FEXCore::IR::IROp_Header *InterferenceIROp = InterferenceOrderedNode->Op(DataBegin); auto PrevIter = Begin; --PrevIter; - Disp->SetWriteCursor(reinterpret_cast(PrevIter()->GetPtr(ListBegin))); + Disp->SetWriteCursor(PrevIter()->GetNode(ListBegin)); auto SpillOp = Disp->_SpillRegister(InterferenceOrderedNode, SpillSlotCount, {InterferenceRegisterNode->RegisterClass}); SpillOp.first->Header.Size = InterferenceIROp->Size; SpillOp.first->Header.Elements = InterferenceIROp->Elements; @@ -525,8 +522,8 @@ namespace FEXCore::IR { auto FirstUseLocation = FindFirstUse(Disp, InterferenceOrderedNode, NextIter, End); if (FirstUseLocation != End) { // LogMan::Throw::A(FirstUseLocation != End, "Failure to find op use"); - NodeWrapper *FirstUseOp = FirstUseLocation(); - OrderedNode *FirstUseOrderedNode = reinterpret_cast(FirstUseOp->GetPtr(ListBegin)); + OrderedNodeWrapper *FirstUseOp = FirstUseLocation(); + OrderedNode *FirstUseOrderedNode = FirstUseOp->GetNode(ListBegin); Disp->SetWriteCursor(FirstUseOrderedNode); auto FilledInterference = Disp->_FillRegister(SpillSlotCount, {InterferenceRegisterNode->RegisterClass}); @@ -558,32 +555,39 @@ namespace FEXCore::IR { } bool RegisterAllocationPass::Run(OpDispatchBuilder *Disp) { - PROFILER_SCOPE(RA); - PROFILER_COUNTER_ADD("RA/counter", 1); bool Changed = false; constexpr uint32_t RATries = 1; SpillSlotCount = 0; HasSpills = false; + HadFullRA = false; for (uint32_t i = 0; i < RATries; ++i) { auto CurrentIR = Disp->ViewIR(); uintptr_t ListSize = CurrentIR.GetListSize(); - uint32_t SSACount = ListSize / sizeof(IR::OrderedNode); + uint32_t SSACount = CurrentIR.GetSSACount(); ResetRegisterGraph(SSACount); FindNodeClasses(&CurrentIR); - CalculateLiveRange(&CurrentIR, SSACount); + CalculateLiveRange(&CurrentIR); + CalculateNodeInterference(SSACount); AllocateRegisters(); if (!Graph->SpillStack.empty()) { - Disp->ShouldDump = true; - Changed = true; - //ClearSpillList(Disp); + if (Config_SupportsSpills) { + Disp->ShouldDump = true; + Changed = true; + ClearSpillList(Disp); + } + else { + HadFullRA = false; + SpillSlotCount = 0; + } return Changed; } else { // We managed to RA, leave now + HadFullRA = true; Disp->ShouldDump = false; return Changed; } diff --git a/Source/Interface/IR/Passes/RegisterAllocationPass.h b/Source/Interface/IR/Passes/RegisterAllocationPass.h index e444679a7..ab6bdbcb8 100644 --- a/Source/Interface/IR/Passes/RegisterAllocationPass.h +++ b/Source/Interface/IR/Passes/RegisterAllocationPass.h @@ -30,16 +30,18 @@ public: void FreeRegisterGraph(); void ResetRegisterGraph(uint32_t NodeCount); void SetNodeClass(uint32_t Node, uint32_t Class); - void AddNodeInterference(uint32_t Node1, uint32_t Node2); uint32_t GetNodeRegister(uint32_t Node); void AllocateRegisters(); /** @} */ + bool HasFullRA() const { return HadFullRA; } bool HadSpills() const { return HasSpills; } uint32_t SpillSlots() const { return SpillSlotCount; } + void SetSupportsSpills(bool Supports) { Config_SupportsSpills = Supports; } + private: RegisterGraph *Graph; void FindNodeClasses(IRListView *CurrentIR); @@ -52,7 +54,8 @@ private: std::vector LiveRanges; - void CalculateLiveRange(IRListView *CurrentIR, uint32_t Nodes); + void CalculateLiveRange(IRListView *CurrentIR); + void CalculateNodeInterference(uint32_t NodeCount); void ClearSpillList(OpDispatchBuilder *Disp); @@ -61,6 +64,9 @@ private: bool HasSpills {}; uint32_t SpillSlotCount {}; + bool HadFullRA {}; + + bool Config_SupportsSpills {true}; }; } diff --git a/Source/Interface/IR/Passes/SyscallOptimization.cpp b/Source/Interface/IR/Passes/SyscallOptimization.cpp index 98b79d74d..34e87fbaf 100644 --- a/Source/Interface/IR/Passes/SyscallOptimization.cpp +++ b/Source/Interface/IR/Passes/SyscallOptimization.cpp @@ -16,24 +16,51 @@ bool SyscallOptimization::Run(OpDispatchBuilder *Disp) { uintptr_t ListBegin = CurrentIR.GetListData(); uintptr_t DataBegin = CurrentIR.GetData(); - IR::NodeWrapperIterator Begin = CurrentIR.begin(); - IR::NodeWrapperIterator End = CurrentIR.end(); + auto Begin = CurrentIR.begin(); + auto Op = Begin(); - while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + OrderedNode *RealNode = Op->GetNode(ListBegin); + auto HeaderOp = RealNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); + + OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); + + while (1) { + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + while (1) { + auto CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); + + if (IROp->Op == FEXCore::IR::OP_SYSCALL) { + Disp->ShouldDump = true; + // Is the first argument a constant? + uint64_t Constant; + if (Disp->IsValueConstant(IROp->Args[0], &Constant)) { + LogMan::Msg::D("Whoa. Syscall argument is constant: %ld", Constant); + Changed = true; + } - if (IROp->Op == FEXCore::IR::OP_SYSCALL) { - // Is the first argument a constant? - uint64_t Constant; - if (Disp->IsValueConstant(IROp->Args[0], &Constant)) { - // LogMan::Msg::A("Whoa. Syscall argument is constant: %ld", Constant); - Changed = true; } + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; } - ++Begin; + + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + } + } return Changed; diff --git a/include/FEXCore/Core/CodeLoader.h b/include/FEXCore/Core/CodeLoader.h index c0d750035..a4f1a8865 100644 --- a/include/FEXCore/Core/CodeLoader.h +++ b/include/FEXCore/Core/CodeLoader.h @@ -34,6 +34,9 @@ public: */ virtual uint64_t DefaultRIP() const = 0; + virtual void GetInitLocations(std::vector *Locations) {} + virtual uint64_t InitializeThreadSlot(std::function Writer) const { return 0; }; + using MemoryLayout = std::tuple; /** * @brief Gets the default memory layout of the memory object being loaded diff --git a/include/FEXCore/IR/IR.h b/include/FEXCore/IR/IR.h index a699ae636..04dc97de9 100644 --- a/include/FEXCore/IR/IR.h +++ b/include/FEXCore/IR/IR.h @@ -6,8 +6,19 @@ namespace FEXCore::IR { +/** + * @brief The IROp_Header is an dynamically sized array + * At the end it contains a uint8_t for the number of arguments that Op has + * Then there is an unsized array of NodeWrapper arguments for the number of arguments this op has + * The op structures that are including the header must ensure that they pad themselves correctly to the number of arguments used + */ +struct IROp_Header; + +class OrderedNode; /** * @brief This is a very simple wrapper for our node pointers + * You probably don't want to use this directly + * Use OpNodeWrapper and OrderedNodeWrapper types below instead * * This is necessary to allow two things * - Reduce memory usage by having the pointer be an 32bit offset rather than the whole 64bit pointer @@ -19,46 +30,53 @@ namespace FEXCore::IR { * - We have to have the base offset live somewhere else * - Has to be POD and trivially copyable * - Makes every real node access turn in to a [Base + Offset] access + * - Can be confusing if you're mixing OpNodeWrapper and OrderedNodeWrapper usage */ -struct NodeWrapper final { +template +struct NodeWrapperBase final { // On x86-64 using a uint64_t type is more efficient since RIP addressing gives you [ + + ] // On AArch64 using uint32_t is just more memory efficient. 32bit or 64bit offset doesn't matter // We use uint32_t to be more memory efficient (Cuts our node list size in half) using NodeOffsetType = uint32_t; NodeOffsetType NodeOffset; - static NodeWrapper WrapOffset(NodeOffsetType Offset) { - NodeWrapper Wrapped; + static NodeWrapperBase WrapOffset(NodeOffsetType Offset) { + NodeWrapperBase Wrapped; Wrapped.NodeOffset = Offset; return Wrapped; } - static NodeWrapper WrapPtr(uintptr_t Base, uintptr_t Value) { - NodeWrapper Wrapped; + static NodeWrapperBase WrapPtr(uintptr_t Base, uintptr_t Value) { + NodeWrapperBase Wrapped; Wrapped.SetOffset(Base, Value); return Wrapped; } - static void *UnwrapNode(uintptr_t Base, NodeWrapper Node) { - return Node.GetPtr(Base); + static void *UnwrapNode(uintptr_t Base, NodeWrapperBase Node) { + return Node.GetNode(Base); } uint32_t ID() const; - explicit NodeWrapper() = default; - void *GetPtr(uintptr_t Base) { return reinterpret_cast(Base + NodeOffset); } - void const *GetPtr(uintptr_t Base) const { return reinterpret_cast(Base + NodeOffset); } + explicit NodeWrapperBase() = default; + + Type *GetNode(uintptr_t Base) { return reinterpret_cast(Base + NodeOffset); } + Type const *GetNode(uintptr_t Base) const { return reinterpret_cast(Base + NodeOffset); } + void SetOffset(uintptr_t Base, uintptr_t Value) { NodeOffset = Value - Base; } - bool operator==(NodeWrapper const &rhs) { return NodeOffset == rhs.NodeOffset; } + bool operator==(NodeWrapperBase const &rhs) { return NodeOffset == rhs.NodeOffset; } }; -static_assert(std::is_pod::value); -static_assert(sizeof(NodeWrapper) == sizeof(uint32_t)); +static_assert(std::is_pod>::value); +static_assert(sizeof(NodeWrapperBase) == sizeof(uint32_t)); + +using OpNodeWrapper = NodeWrapperBase; +using OrderedNodeWrapper = NodeWrapperBase; struct OrderedNodeHeader { - NodeWrapper Value; - NodeWrapper Next; - NodeWrapper Previous; + OpNodeWrapper Value; + OrderedNodeWrapper Next; + OrderedNodeWrapper Previous; }; static_assert(sizeof(OrderedNodeHeader) == sizeof(uint32_t) * 3); @@ -70,7 +88,7 @@ static_assert(sizeof(OrderedNodeHeader) == sizeof(uint32_t) * 3); */ class NodeWrapperIterator final { public: - using value_type = NodeWrapper; + using value_type = OrderedNodeWrapper; using size_type = std::size_t; using difference_type = std::ptrdiff_t; using reference = value_type&; @@ -83,9 +101,9 @@ public: using const_reverse_iterator = const_iterator; using iterator_category = std::bidirectional_iterator_tag; - using NodeType = NodeWrapper; - using NodePtr = NodeWrapper*; - using NodeRef = NodeWrapper&; + using NodeType = value_type; + using NodePtr = value_type*; + using NodeRef = value_type&; NodeWrapperIterator(uintptr_t Base) : BaseList {Base} {} explicit NodeWrapperIterator(uintptr_t Base, NodeType Ptr) : BaseList {Base}, Node {Ptr} {} @@ -99,13 +117,13 @@ public: } NodeWrapperIterator operator++() { - OrderedNodeHeader *RealNode = reinterpret_cast(Node.GetPtr(BaseList)); + OrderedNodeHeader *RealNode = reinterpret_cast(Node.GetNode(BaseList)); Node = RealNode->Next; return *this; } NodeWrapperIterator operator--() { - OrderedNodeHeader *RealNode = reinterpret_cast(Node.GetPtr(BaseList)); + OrderedNodeHeader *RealNode = reinterpret_cast(Node.GetNode(BaseList)); Node = RealNode->Previous; return *this; } @@ -123,14 +141,6 @@ private: NodeType Node{}; }; -/** - * @brief The IROp_Header is an dynamically sized array - * At the end it contains a uint8_t for the number of arguments that Op has - * Then there is an unsized array of NodeWrapper arguments for the number of arguments this op has - * The op structures that are including the header must ensure that they pad themselves correctly to the number of arguments used - */ -struct IROp_Header; - /** * @brief This is a node in our IR representation * Is a doubly linked list node that lives in a representation of a linearly allocated node list @@ -153,6 +163,8 @@ class OrderedNode final { OrderedNodeHeader Header; uint32_t NumUses; + using value_type = OrderedNodeWrapper; + OrderedNode() = default; /** @@ -163,7 +175,7 @@ class OrderedNode final { * * @return Pointer to the node being added */ - NodeWrapper append(uintptr_t Base, NodeWrapper Node) { + value_type append(uintptr_t Base, value_type Node) { // Set Next Node's Previous to incoming node SetPrevious(Base, Header.Next, Node); @@ -179,7 +191,7 @@ class OrderedNode final { } OrderedNode *append(uintptr_t Base, OrderedNode *Node) { - NodeWrapper WNode = Node->Wrapped(Base); + value_type WNode = Node->Wrapped(Base); // Set Next Node's Previous to incoming node SetPrevious(Base, Header.Next, WNode); @@ -201,7 +213,7 @@ class OrderedNode final { * * @return Pointer to the node being added */ - NodeWrapper prepend(uintptr_t Base, NodeWrapper Node) { + value_type prepend(uintptr_t Base, value_type Node) { // Set the previous node's next to the incoming node SetNext(Base, Header.Previous, Node); @@ -217,7 +229,7 @@ class OrderedNode final { } OrderedNode *prepend(uintptr_t Base, OrderedNode *Node) { - NodeWrapper WNode = Node->Wrapped(Base); + value_type WNode = Node->Wrapped(Base); // Set the previous node's next to the incoming node SetNext(Base, Header.Previous, WNode); @@ -241,10 +253,10 @@ class OrderedNode final { size_t size(uintptr_t Base) const { size_t Size = 1; // Walk the list forward until we hit a sentinal - NodeWrapper Current = Header.Next; + value_type Current = Header.Next; while (Current.NodeOffset != 0) { ++Size; - OrderedNode *RealNode = reinterpret_cast(Current.GetPtr(Base)); + OrderedNode *RealNode = Current.GetNode(Base); Current = RealNode->Header.Next; } return Size; @@ -258,41 +270,36 @@ class OrderedNode final { SetPrevious(Base, Header.Next, Header.Previous); } - IROp_Header const* Op(uintptr_t Base) const { return reinterpret_cast(Header.Value.GetPtr(Base)); } - IROp_Header *Op(uintptr_t Base) { return reinterpret_cast(Header.Value.GetPtr(Base)); } + IROp_Header const* Op(uintptr_t Base) const { return Header.Value.GetNode(Base); } + IROp_Header *Op(uintptr_t Base) { return Header.Value.GetNode(Base); } uint32_t GetUses() const { return NumUses; } void AddUse() { ++NumUses; } void RemoveUse() { --NumUses; } - using iterator = NodeWrapperIterator; - - iterator begin(uint64_t Base) noexcept { return iterator(Base, Wrapped(Base)); } - iterator end(uint64_t Base, uint64_t End) noexcept { return iterator(Base, WrappedOffset(End)); } - - NodeWrapper Wrapped(uintptr_t Base) { - NodeWrapper Tmp; + value_type Wrapped(uintptr_t Base) { + value_type Tmp; Tmp.SetOffset(Base, reinterpret_cast(this)); return Tmp; } private: - NodeWrapper WrappedOffset(uint32_t Offset) { - NodeWrapper Tmp; + value_type WrappedOffset(uint32_t Offset) { + value_type Tmp; Tmp.NodeOffset = Offset; return Tmp; } - static void SetPrevious(uintptr_t Base, NodeWrapper Node, NodeWrapper New) { + static void SetPrevious(uintptr_t Base, value_type Node, value_type New) { if (Node.NodeOffset == 0) return; - OrderedNode *RealNode = reinterpret_cast(Node.GetPtr(Base)); + OrderedNode *RealNode = Node.GetNode(Base); RealNode->Header.Previous = New; } - static void SetNext(uintptr_t Base, NodeWrapper Node, NodeWrapper New) { + static void SetNext(uintptr_t Base, value_type Node, value_type New) { if (Node.NodeOffset == 0) return; - OrderedNode *RealNode = reinterpret_cast(Node.GetPtr(Base)); + OrderedNode *RealNode = Node.GetNode(Base); RealNode->Header.Next = New; } @@ -304,26 +311,24 @@ static_assert(std::is_trivially_copyable::value); static_assert(offsetof(OrderedNode, Header) == 0); static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t))); +struct RegisterClassType final { + uint32_t Val; + operator uint32_t() { + return Val; + } +}; + #define IROP_ENUM #define IROP_STRUCTS #define IROP_SIZES #include "IRDefines.inc" -template -struct Wrapper final { - T *first; - OrderedNode *Node; ///< Actual offset of this IR in ths list - - operator Wrapper() const { return Wrapper {reinterpret_cast(first), Node}; } - operator OrderedNode *() { return Node; } - operator NodeWrapper () { return Node->Header.Value; } -}; - template class IRListView; void Dump(std::stringstream *out, IRListView const* IR); -inline uint32_t NodeWrapper::ID() const { return NodeOffset / sizeof(IR::OrderedNode); } +template +inline uint32_t NodeWrapperBase::ID() const { return NodeOffset / sizeof(IR::OrderedNode); } }; diff --git a/include/FEXCore/IR/IntrusiveIRList.h b/include/FEXCore/IR/IntrusiveIRList.h index f3fcd78e9..e1adc3a8e 100644 --- a/include/FEXCore/IR/IntrusiveIRList.h +++ b/include/FEXCore/IR/IntrusiveIRList.h @@ -22,7 +22,7 @@ class IntrusiveAllocator final { IntrusiveAllocator(IntrusiveAllocator &&) = delete; IntrusiveAllocator(size_t Size) : MemorySize {Size} { - Data = reinterpret_cast(calloc(Size, 1)); + Data = reinterpret_cast(malloc(Size)); } ~IntrusiveAllocator() { @@ -35,7 +35,7 @@ class IntrusiveAllocator final { } void *Allocate(size_t Size) { - assert(CheckSize(Size) && "Failure"); + LogMan::Throw::A(CheckSize(Size), "Ran out of space in IntrusiveAllocator during allocation"); size_t NewOffset = CurrentOffset + Size; uintptr_t NewPointer = Data + CurrentOffset; CurrentOffset = NewOffset; @@ -95,12 +95,13 @@ public: size_t GetDataSize() const { return DataSize; } size_t GetListSize() const { return ListSize; } + size_t GetSSACount() const { return ListSize / sizeof(OrderedNode); } using iterator = NodeWrapperIterator; iterator begin() const noexcept { - NodeWrapper Wrapped; + OrderedNodeWrapper Wrapped; Wrapped.NodeOffset = sizeof(OrderedNode); return iterator(reinterpret_cast(ListData), Wrapped); } @@ -112,11 +113,19 @@ public: */ iterator end() const noexcept { - NodeWrapper Wrapped; + OrderedNodeWrapper Wrapped; Wrapped.NodeOffset = 0; return iterator(reinterpret_cast(ListData), Wrapped); } + /** + * @brief Convert a OrderedNodeWrapper to an interator that we can iterate over + * @return Iterator for this op + */ + iterator at(OrderedNodeWrapper Node) const noexcept { + return iterator(reinterpret_cast(ListData), Node); + } + private: void *IRData; void *ListData;