diff --git a/Scripts/json_ir_generator.py b/Scripts/json_ir_generator.py index 23cd2ee95..a85437107 100644 --- a/Scripts/json_ir_generator.py +++ b/Scripts/json_ir_generator.py @@ -36,7 +36,7 @@ def print_ir_structs(ops, defines): output_file.write("\ttemplate\n") output_file.write("\tT* CW() { return reinterpret_cast(Data); }\n") - output_file.write("\tNodeWrapper Args[0];\n") + output_file.write("\tOrderedNodeWrapper Args[0];\n") output_file.write("};\n\n"); @@ -48,6 +48,7 @@ def print_ir_structs(ops, defines): for op_key, op_vals in ops.items(): SSAArgs = 0 HasArgs = False + HasSSANames = False if ("SSAArgs" in op_vals): SSAArgs = int(op_vals["SSAArgs"]) @@ -55,16 +56,24 @@ def print_ir_structs(ops, defines): if ("Args" in op_vals and len(op_vals["Args"]) != 0): HasArgs = True + if ("SSANames" in op_vals and len(op_vals["SSANames"]) != 0): + HasSSANames = True + if (HasArgs or SSAArgs != 0): output_file.write("struct __attribute__((packed)) IROp_%s {\n" % op_key) output_file.write("\tIROp_Header Header;\n\n") # SSA arguments have a hard requirement to appear after the header if (SSAArgs != 0): - output_file.write("private:\n") - for i in range(0, SSAArgs): - output_file.write("\tuint64_t : (sizeof(NodeWrapper) * 8);\n"); - output_file.write("public:\n") + if (HasSSANames): + for i in range(0, SSAArgs): + output_file.write("\tOrderedNodeWrapper %s;\n" % (op_vals["SSANames"][i])); + + else: + output_file.write("private:\n") + for i in range(0, SSAArgs): + output_file.write("\tuint64_t : (sizeof(OrderedNodeWrapper) * 8);\n"); + output_file.write("public:\n") if (HasArgs): output_file.write("\t// User defined data\n") @@ -182,13 +191,32 @@ def print_ir_allocator_helpers(ops, defines): output_file.write("#ifdef IROP_ALLOCATE_HELPERS\n") output_file.write("\ttemplate \n") - output_file.write("\tusing IRPair = FEXCore::IR::Wrapper;\n\n") + output_file.write("\tstruct Wrapper final {\n") + output_file.write("\t\tT *first;\n") + output_file.write("\t\tOrderedNode *Node; ///< Actual offset of this IR in ths list\n") + output_file.write("\t\t\n") + output_file.write("\t\toperator Wrapper() const { return Wrapper {reinterpret_cast(first), Node}; }\n") + output_file.write("\t\toperator OrderedNode *() { return Node; }\n") + output_file.write("\t\toperator OpNodeWrapper () { return Node->Header.Value; }\n") + output_file.write("\t};\n") + + output_file.write("\ttemplate \n") + output_file.write("\tusing IRPair = Wrapper;\n\n") output_file.write("\tIRPair AllocateRawOp(size_t HeaderSize) {\n") output_file.write("\t\tauto Op = reinterpret_cast(Data.Allocate(HeaderSize));\n") output_file.write("\t\tmemset(Op, 0, HeaderSize);\n") output_file.write("\t\tOp->Op = IROps::OP_DUMMY;\n") - output_file.write("\t\treturn FEXCore::IR::Wrapper{Op, CreateNode(Op)};\n") + output_file.write("\t\treturn IRPair{Op, CreateNode(Op)};\n") + output_file.write("\t}\n\n") + + output_file.write("\ttemplate\n") + output_file.write("\tT *AllocateOrphanOp() {\n") + output_file.write("\t\tsize_t Size = FEXCore::IR::GetSize(T2);\n") + output_file.write("\t\tauto Op = reinterpret_cast(Data.Allocate(Size));\n") + output_file.write("\t\tmemset(Op, 0, Size);\n") + output_file.write("\t\tOp->Header.Op = T2;\n") + output_file.write("\t\treturn Op;\n") output_file.write("\t}\n\n") output_file.write("\ttemplate\n") @@ -197,23 +225,23 @@ def print_ir_allocator_helpers(ops, defines): output_file.write("\t\tauto Op = reinterpret_cast(Data.Allocate(Size));\n") output_file.write("\t\tmemset(Op, 0, Size);\n") output_file.write("\t\tOp->Header.Op = T2;\n") - output_file.write("\t\treturn FEXCore::IR::Wrapper{Op, CreateNode(&Op->Header)};\n") + output_file.write("\t\treturn IRPair{Op, CreateNode(&Op->Header)};\n") output_file.write("\t}\n\n") output_file.write("\tuint8_t GetOpSize(OrderedNode *Op) const {\n") - output_file.write("\t\tauto HeaderOp = reinterpret_cast(Op->Header.Value.GetPtr(Data.Begin()));\n") + output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(Data.Begin());\n") output_file.write("\t\tLogMan::Throw::A(HeaderOp->HasDest, \"Op %s has no dest\\n\", GetName(HeaderOp->Op));\n") output_file.write("\t\treturn HeaderOp->Size;\n") output_file.write("\t}\n\n") output_file.write("\tuint8_t GetOpElements(OrderedNode *Op) const {\n") - output_file.write("\t\tauto HeaderOp = reinterpret_cast(Op->Header.Value.GetPtr(Data.Begin()));\n") + output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(Data.Begin());\n") output_file.write("\t\tLogMan::Throw::A(HeaderOp->HasDest, \"Op %s has no dest\\n\", GetName(HeaderOp->Op));\n") output_file.write("\t\treturn HeaderOp->Elements;\n") output_file.write("\t}\n\n") output_file.write("\tbool OpHasDest(OrderedNode *Op) const {\n") - output_file.write("\t\tauto HeaderOp = reinterpret_cast(Op->Header.Value.GetPtr(Data.Begin()));\n") + output_file.write("\t\tauto HeaderOp = Op->Header.Value.GetNode(Data.Begin());\n") output_file.write("\t\treturn HeaderOp->HasDest;\n") output_file.write("\t}\n\n") diff --git a/Source/CMakeLists.txt b/Source/CMakeLists.txt index ff5a263ed..b9058db6d 100644 --- a/Source/CMakeLists.txt +++ b/Source/CMakeLists.txt @@ -54,7 +54,6 @@ set (SRCS Interface/Core/CPUID.cpp Interface/Core/Frontend.cpp Interface/Core/OpcodeDispatcher.cpp - Interface/Core/RegisterAllocation.cpp Interface/Core/X86Tables.cpp Interface/Core/X86DebugInfo.cpp Interface/Core/Interpreter/InterpreterCore.cpp @@ -71,6 +70,7 @@ set (SRCS Interface/IR/Passes/IRCompaction.cpp Interface/IR/Passes/IRValidation.cpp Interface/IR/Passes/RedundantFlagCalculationElimination.cpp + Interface/IR/Passes/RegisterAllocationPass.cpp Interface/IR/Passes/SyscallOptimization.cpp ) @@ -108,7 +108,7 @@ add_custom_target(IR_INC add_library(${PROJECT_NAME} STATIC ${SRCS}) add_dependencies(${PROJECT_NAME} IR_INC) -target_link_libraries(${PROJECT_NAME} LLVM pthread rt ${JIT_LIBS}) +target_link_libraries(${PROJECT_NAME} LLVM pthread rt ${JIT_LIBS} dl) target_include_directories(${PROJECT_NAME} PUBLIC "${CMAKE_CURRENT_BINARY_DIR}") diff --git a/Source/Interface/Context/Context.h b/Source/Interface/Context/Context.h index f8c14b3bd..85e6a9205 100644 --- a/Source/Interface/Context/Context.h +++ b/Source/Interface/Context/Context.h @@ -14,6 +14,13 @@ namespace FEXCore { class SyscallHandler; +namespace CPU { + class JITCore; +} +} + +namespace FEXCore::IR { + class RegisterAllocationPass; } namespace FEXCore::Context { @@ -24,6 +31,8 @@ namespace FEXCore::Context { struct Context { friend class FEXCore::SyscallHandler; + friend class FEXCore::CPU::JITCore; + struct { bool Multiblock {false}; bool BreakOnFrontendFailure {true}; @@ -78,6 +87,11 @@ namespace FEXCore::Context { FEXCore::Core::ThreadState *GetThreadState(); void LoadEntryList(); + uintptr_t CompileBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP); + uintptr_t CompileFallbackBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP); + protected: + IR::RegisterAllocationPass *GetRegisterAllocatorPass(); + private: void WaitForIdle(); FEXCore::Core::InternalThreadState* CreateThread(FEXCore::Core::CPUState *NewThreadState, uint64_t ParentTID, uint64_t ChildTID); @@ -89,7 +103,6 @@ namespace FEXCore::Context { void ExecutionThread(FEXCore::Core::InternalThreadState *Thread); void RunThread(FEXCore::Core::InternalThreadState *Thread); - uintptr_t CompileBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP); uintptr_t AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr); FEXCore::CodeLoader *LocalLoader{}; @@ -99,5 +112,8 @@ namespace FEXCore::Context { void AddThreadRIPsToEntryList(FEXCore::Core::InternalThreadState *Thread); void SaveEntryList(); std::set EntryList; + std::vector InitLocations; + uint64_t StartingRIP; + IR::RegisterAllocationPass *RAPass {}; }; } diff --git a/Source/Interface/Core/Core.cpp b/Source/Interface/Core/Core.cpp index 373aa616d..ea9243bfc 100644 --- a/Source/Interface/Core/Core.cpp +++ b/Source/Interface/Core/Core.cpp @@ -8,6 +8,7 @@ #include "Interface/Core/Interpreter/InterpreterCore.h" #include "Interface/Core/JIT/JITCore.h" #include "Interface/Core/LLVMJIT/LLVMCore.h" +#include "Interface/IR/Passes/RegisterAllocationPass.h" #include #include @@ -20,7 +21,7 @@ constexpr uint64_t STACK_OFFSET = 0xc000'0000; constexpr uint64_t FS_OFFSET = 0xb000'0000; -constexpr uint64_t FS_SIZE = 0x1000; +constexpr uint64_t FS_SIZE = 0x1000'0000; namespace FEXCore::CPU { bool CreateCPUCore(FEXCore::Context::Context *CTX) { @@ -212,7 +213,7 @@ namespace FEXCore::Context { } memset(NewThreadState.flags, 0, 32); NewThreadState.gs = 0; - NewThreadState.fs = FS_OFFSET + FS_SIZE / 2; + NewThreadState.fs = FS_OFFSET; NewThreadState.flags[1] = 1; FEXCore::Core::InternalThreadState *Thread = CreateThread(&NewThreadState, 0, 0); @@ -220,17 +221,11 @@ namespace FEXCore::Context { // We are the parent thread ParentThread = Thread; - auto MemLayout = Loader->GetLayout(); - - uint64_t BasePtr = AlignDown(std::get<0>(MemLayout), PAGE_SIZE); - uint64_t BaseSize = AlignUp(std::get<2>(MemLayout), PAGE_SIZE); - - Thread->BlockCache->HintUsedRange(BasePtr, BaseSize); - - uintptr_t BaseRegion = reinterpret_cast(MapRegion(Thread, BasePtr, BaseSize, true)); + uintptr_t MemoryBase = MemoryMapper.GetBaseOffset(0); auto MemoryMapperFunction = [&](uint64_t Base, uint64_t Size) -> void* { - return MapRegion(Thread, Base, Size); + Thread->BlockCache->HintUsedRange(Base, Base); + return MapRegion(Thread, Base, Size, true); }; Loader->MapMemoryRegion(MemoryMapperFunction); @@ -244,15 +239,19 @@ namespace FEXCore::Context { // Now let the code loader setup memory auto MemoryWriterFunction = [&](void const *Data, uint64_t Addr, uint64_t Size) -> void { // Writes the machine code to be emulated in to memory - memcpy(reinterpret_cast(BaseRegion + Addr), Data, Size); + memcpy(reinterpret_cast(MemoryBase + Addr), Data, Size); }; Loader->LoadMemory(MemoryWriterFunction); + Loader->GetInitLocations(&InitLocations); - // Set the RIP to what the code loader wants - Thread->State.State.rip = Loader->DefaultRIP(); + auto TLSSlotWriter = [&](void const *Data, uint64_t Size) -> void { + memcpy(reinterpret_cast(MemoryBase + FS_OFFSET), Data, Size); + }; - LogMan::Msg::D("Memory Base: 0x%016lx", MemoryMapper.GetBaseOffset(0)); + // Offset next thread's FS_OFFSET by slot size + uint64_t SlotSize = Loader->InitializeThreadSlot(TLSSlotWriter); + StartingRIP = Loader->DefaultRIP(); InitializeThread(Thread); @@ -275,7 +274,7 @@ namespace FEXCore::Context { if (AllPaused) break; - PauseWait.WaitFor(std::chrono::seconds(1)); + PauseWait.WaitFor(std::chrono::milliseconds(10)); } while (true); } @@ -325,10 +324,11 @@ namespace FEXCore::Context { Thread->FallbackBackend->Initialize(); // Compile all of our cached entries - LogMan::Msg::D("Precompiling: %ld blocks", EntryList.size()); + LogMan::Msg::D("Precompiling: %ld blocks...", EntryList.size()); for (auto Entry : EntryList) { CompileRIP(Thread, Entry); } + LogMan::Msg::D("Done", EntryList.size()); // This will create the execution thread but it won't actually start executing Thread->ExecutionThread = std::thread(&Context::ExecutionThread, this, Thread); @@ -379,6 +379,15 @@ namespace FEXCore::Context { return Thread; } + IR::RegisterAllocationPass *Context::GetRegisterAllocatorPass() { + if (!RAPass) { + RAPass = new IR::RegisterAllocationPass(); + PassManager.InsertPass(RAPass); + } + + return RAPass; + } + uintptr_t Context::AddBlockMapping(FEXCore::Core::InternalThreadState *Thread, uint64_t Address, void *Ptr) { auto BlockMapPtr = Thread->BlockCache->AddBlockMapping(Address, Ptr); if (BlockMapPtr == 0) { @@ -406,7 +415,6 @@ namespace FEXCore::Context { bool HadDispatchError {false}; [[maybe_unused]] bool HadRIPSetter {false}; - Thread->OpDispatcher->BeginBlock(); if (!FrontendDecoder.DecodeInstructionsInBlock(&GuestCode[TotalInstructionsLength], GuestRIP + TotalInstructionsLength)) { if (Config.BreakOnFrontendFailure) { LogMan::Msg::E("Had Frontend decoder error"); @@ -415,6 +423,7 @@ namespace FEXCore::Context { return 0; } + Thread->OpDispatcher->BeginFunction(GuestRIP); auto DecodedOps = FrontendDecoder.GetDecodedInsts(); for (size_t i = 0; i < DecodedOps.second; ++i) { FEXCore::X86Tables::X86InstInfo const* TableInfo {nullptr}; @@ -515,15 +524,17 @@ namespace FEXCore::Context { if (!Thread->OpDispatcher->Information.HadUnconditionalExit) { - Thread->OpDispatcher->EndBlock(TotalInstructionsLength); + Thread->OpDispatcher->CreateNewEndBlock(TotalInstructionsLength); + Thread->OpDispatcher->CreateNewBeginBlock(); Thread->OpDispatcher->ExitFunction(); + Thread->OpDispatcher->CreateNewEndBlock(0); } + Thread->OpDispatcher->Finalize(); // Run the passmanager over the IR from the dispatcher PassManager.Run(Thread->OpDispatcher.get()); - if (Thread->OpDispatcher->ShouldDump) -// if (GuestRIP == 0x48b680) + // if (Thread->OpDispatcher->ShouldDump) { std::stringstream out; auto NewIR = Thread->OpDispatcher->ViewIR(); @@ -531,17 +542,15 @@ namespace FEXCore::Context { printf("IR 0x%lx:\n%s\n@@@@@\n", GuestRIP, out.str().c_str()); } - // Do RA on the IR right now? - // Create a copy of the IR and place it in this thread's IR cache - auto IR = Thread->IRLists.try_emplace(GuestRIP, Thread->OpDispatcher->CreateIRCopy()); + auto AddedIR = Thread->IRLists.try_emplace(GuestRIP, Thread->OpDispatcher->CreateIRCopy()); Thread->OpDispatcher->ResetWorkingList(); - auto Debugit = Thread->DebugData.try_emplace(GuestRIP, FEXCore::Core::DebugData{}); + auto Debugit = Thread->DebugData.try_emplace(GuestRIP); Debugit.first->second.GuestCodeSize = TotalInstructionsLength; Debugit.first->second.GuestInstructionCount = TotalInstructions; - IRList = IR.first->second.get(); + IRList = AddedIR.first->second.get(); DebugData = &Debugit.first->second; Thread->Stats.BlocksCompiled.fetch_add(1); } @@ -560,6 +569,20 @@ namespace FEXCore::Context { return 0; } + using BlockFn = void (*)(FEXCore::Core::InternalThreadState *Thread); + uintptr_t Context::CompileFallbackBlock(FEXCore::Core::InternalThreadState *Thread, uint64_t GuestRIP) { + // We have ONE more chance to try and fallback to the fallback CPU backend + // This will most likely fail since regular code use won't be using a fallback core. + // It's mainly for testing new instruction encodings + void *CodePtr = Thread->FallbackBackend->CompileCode(nullptr, nullptr); + if (CodePtr) { + uintptr_t Ptr = reinterpret_cast(AddBlockMapping(Thread, GuestRIP, CodePtr)); + return Ptr; + } + + return 0; + } + void Context::ExecutionThread(FEXCore::Core::InternalThreadState *Thread) { Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_WAITING; @@ -579,119 +602,143 @@ namespace FEXCore::Context { Thread->State.RunningEvents.Running = true; Thread->State.RunningEvents.ShouldPause = false; constexpr uint32_t CoreDebugLevel = 0; - while (!ShouldStop.load() && !Thread->State.RunningEvents.ShouldStop.load()) { - uint64_t GuestRIP = Thread->State.State.rip; - if (CoreDebugLevel >= 1) { - char const *Name = LocalLoader->FindSymbolNameInRange(GuestRIP); - LogMan::Msg::D(">>>>RIP: 0x%lx: '%s'", GuestRIP, Name ? Name : ""); - } + bool Initializing = false; - using BlockFn = void (*)(FEXCore::Core::InternalThreadState *Thread); + uint64_t InitializationStep = 0; + if (Initializing) { + Thread->State.State.rip = ~0ULL; + } + else { + Thread->State.State.rip = StartingRIP; + } - if (!Thread->CPUBackend->NeedsOpDispatch()) { - BlockFn Ptr = reinterpret_cast(Thread->CPUBackend->CompileCode(nullptr, nullptr)); - Ptr(Thread); - } - else { - // Do have have this block compiled? - auto it = Thread->BlockCache->FindBlock(GuestRIP); - if (it == 0) { - // If not compile it - it = CompileBlock(Thread, GuestRIP); + if (Thread->CPUBackend->HasCustomDispatch()) { + Thread->CPUBackend->ExecuteCustomDispatch(&Thread->State); + } + else { + while (!ShouldStop.load() && !Thread->State.RunningEvents.ShouldStop.load()) { + if (Initializing) { + if (Thread->State.State.rip == ~0ULL) { + if (InitializationStep < InitLocations.size()) { + Thread->State.State.gregs[X86State::REG_RSP] -= 8; + *MemoryMapper.GetPointer(Thread->State.State.gregs[X86State::REG_RSP]) = ~0ULL; + LogMan::Msg::D("Going down init path: 0x%lx", InitLocations[InitializationStep]); + Thread->State.State.rip = InitLocations[InitializationStep++]; + } + else { + Initializing = false; + Thread->State.State.rip = StartingRIP; + } + } + } + uint64_t GuestRIP = Thread->State.State.rip; + + if (CoreDebugLevel >= 1) { + char const *Name = LocalLoader->FindSymbolNameInRange(GuestRIP); + LogMan::Msg::D(">>>>RIP: 0x%lx: '%s'", GuestRIP, Name ? Name : ""); } - // Did we successfully compile this block? - if (it != 0) { - // Block is compiled, run it - BlockFn Ptr = reinterpret_cast(it); + if (!Thread->CPUBackend->NeedsOpDispatch()) { + BlockFn Ptr = reinterpret_cast(Thread->CPUBackend->CompileCode(nullptr, nullptr)); Ptr(Thread); } else { - // We have ONE more chance to try and fallback to the fallback CPU backend - // This will most likely fail since regular code use won't be using a fallback core. - // It's mainly for testing new instruction encodings - void *CodePtr = Thread->FallbackBackend->CompileCode(nullptr, nullptr); - if (CodePtr) { - BlockFn Ptr = reinterpret_cast(AddBlockMapping(Thread, GuestRIP, CodePtr)); - Ptr(Thread); + // Do have have this block compiled? + auto it = Thread->BlockCache->FindBlock(GuestRIP); + if (it == 0) { + // If not compile it + it = CompileBlock(Thread, GuestRIP); + } + + // Did we successfully compile this block? + if (it != 0) { + // Block is compiled, run it + BlockFn Ptr = reinterpret_cast(it); + Ptr(Thread); } else { - // Let the frontend know that something has happened that is unhandled - Thread->State.RunningEvents.ShouldPause = true; - Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_UNKNOWNERROR; + // We have ONE more chance to try and fallback to the fallback CPU backend + // This will most likely fail since regular code use won't be using a fallback core. + // It's mainly for testing new instruction encodings + uintptr_t CodePtr = CompileFallbackBlock(Thread, GuestRIP); + if (CodePtr) { + BlockFn Ptr = reinterpret_cast(CodePtr); + Ptr(Thread); + } + else { + // Let the frontend know that something has happened that is unhandled + Thread->State.RunningEvents.ShouldPause = true; + Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_UNKNOWNERROR; + } } } - } -// if (GuestRIP == 0x48c8dd) { -// fflush(stdout); -// __builtin_trap(); -// } - if (CoreDebugLevel >= 2) { - int i = 0; - LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); - i += 4; - LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); - i += 4; - LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); - i += 4; - LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); - uint64_t PackedFlags{}; - for (unsigned i = 0; i < 32; ++i) { - PackedFlags |= static_cast(Thread->State.State.flags[i]) << i; + if (CoreDebugLevel >= 2) { + int i = 0; + LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); + i += 4; + LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); + i += 4; + LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); + i += 4; + LogMan::Msg::D("\tGPR[%d]: %016lx %016lx %016lx %016lx", i, Thread->State.State.gregs[i + 0], Thread->State.State.gregs[i + 1], Thread->State.State.gregs[i + 2], Thread->State.State.gregs[i + 3]); + uint64_t PackedFlags{}; + for (i = 0; i < 32; ++i) { + PackedFlags |= static_cast(Thread->State.State.flags[i]) << i; + } + LogMan::Msg::D("\tFlags: %016lx", PackedFlags); } - LogMan::Msg::D("\tFlags: %016lx", PackedFlags); - } - if (CoreDebugLevel >= 3) { - int i = 0; - LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); - LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); + if (CoreDebugLevel >= 3) { + int i = 0; + LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); + LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); - i += 4; - LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); - LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); - i += 4; - LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); - LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); - i += 4; - LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); - LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); - uint64_t PackedFlags{}; - for (unsigned i = 0; i < 32; ++i) { - PackedFlags |= static_cast(Thread->State.State.flags[i]) << i; + i += 4; + LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); + LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); + i += 4; + LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); + LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); + i += 4; + LogMan::Msg::D("\tXMM[%d][0]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][0], Thread->State.State.xmm[i + 1][0], Thread->State.State.xmm[i + 2][0], Thread->State.State.xmm[i + 3][0]); + LogMan::Msg::D("\tXMM[%d][1]: %016lx %016lx %016lx %016lx", i, Thread->State.State.xmm[i + 0][1], Thread->State.State.xmm[i + 1][1], Thread->State.State.xmm[i + 2][1], Thread->State.State.xmm[i + 3][1]); + uint64_t PackedFlags{}; + for (i = 0; i < 32; ++i) { + PackedFlags |= static_cast(Thread->State.State.flags[i]) << i; + } + LogMan::Msg::D("\tFlags: %016lx", PackedFlags); } - LogMan::Msg::D("\tFlags: %016lx", PackedFlags); - } - if (Thread->State.RunningEvents.ShouldStop.load()) { - // If it is the parent thread that died then just leave - // XXX: This doesn't make sense when the parent thread doesn't outlive its children - if (Thread->State.ThreadManager.GetTID() == 1) { - ShouldStop = true; - Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_SHUTDOWN; + if (Thread->State.RunningEvents.ShouldStop.load()) { + // If it is the parent thread that died then just leave + // XXX: This doesn't make sense when the parent thread doesn't outlive its children + if (Thread->State.ThreadManager.GetTID() == 1) { + ShouldStop = true; + Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_SHUTDOWN; + } + break; } - break; - } - if (RunningMode == FEXCore::Context::CoreRunningMode::MODE_SINGLESTEP || Thread->State.RunningEvents.ShouldPause) { - Thread->State.RunningEvents.Running = false; - Thread->State.RunningEvents.WaitingToStart = false; + if (RunningMode == FEXCore::Context::CoreRunningMode::MODE_SINGLESTEP || Thread->State.RunningEvents.ShouldPause) { + Thread->State.RunningEvents.Running = false; + Thread->State.RunningEvents.WaitingToStart = false; - // If something previously hasn't set the exit state then set it now - if (Thread->ExitReason == FEXCore::Context::ExitReason::EXIT_NONE) - Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_DEBUG; + // If something previously hasn't set the exit state then set it now + if (Thread->ExitReason == FEXCore::Context::ExitReason::EXIT_NONE) + Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_DEBUG; - PauseWait.NotifyAll(); - Thread->StartRunning.Wait(); + PauseWait.NotifyAll(); + Thread->StartRunning.Wait(); - // If we set it to debug then set it back to none after this - // We want to retain the state if the frontend decides to leave - if (Thread->ExitReason == FEXCore::Context::ExitReason::EXIT_DEBUG) - Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_NONE; + // If we set it to debug then set it back to none after this + // We want to retain the state if the frontend decides to leave + if (Thread->ExitReason == FEXCore::Context::ExitReason::EXIT_DEBUG) + Thread->ExitReason = FEXCore::Context::ExitReason::EXIT_NONE; - Thread->State.RunningEvents.Running = true; + Thread->State.RunningEvents.Running = true; + } } } @@ -731,7 +778,7 @@ namespace FEXCore::Context { return MemoryMapper.GetMemoryBase(); } - void Context::CopyMemoryMapping([[maybe_unused]] FEXCore::Core::InternalThreadState *ParentThread, FEXCore::Core::InternalThreadState *ChildThread) { + void Context::CopyMemoryMapping([[maybe_unused]] FEXCore::Core::InternalThreadState*, FEXCore::Core::InternalThreadState *ChildThread) { auto Regions = MemoryMapper.MappedRegions; for (auto const& Region : Regions) { ChildThread->CPUBackend->MapRegion(Region.Ptr, Region.Offset, Region.Size); diff --git a/Source/Interface/Core/Interpreter/InterpreterCore.cpp b/Source/Interface/Core/Interpreter/InterpreterCore.cpp index a4eab635e..527ecf500 100644 --- a/Source/Interface/Core/Interpreter/InterpreterCore.cpp +++ b/Source/Interface/Core/Interpreter/InterpreterCore.cpp @@ -39,10 +39,10 @@ private: void *AllocateTmpSpace(size_t Size); template - Res GetDest(IR::NodeWrapper Op); + Res GetDest(IR::OrderedNodeWrapper Op); template - Res GetSrc(IR::NodeWrapper Src); + Res GetSrc(IR::OrderedNodeWrapper Src); std::vector TmpSpace; DestMapType DestMap; @@ -86,13 +86,13 @@ void *InterpreterCore::AllocateTmpSpace(size_t Size) { } template -Res InterpreterCore::GetDest(IR::NodeWrapper Op) { +Res InterpreterCore::GetDest(IR::OrderedNodeWrapper Op) { auto DstPtr = DestMap[Op.NodeOffset]; return reinterpret_cast(DstPtr); } template -Res InterpreterCore::GetSrc(IR::NodeWrapper Src) { +Res InterpreterCore::GetSrc(IR::OrderedNodeWrapper Src) { #if DESTMAP_AS_MAP LogMan::Throw::A(DestMap.find(Src.NodeOffset) != DestMap.end(), "Op had source but it wasn't in the destination map"); #endif @@ -138,8 +138,8 @@ void InterpreterCore::ExecuteCode(FEXCore::Core::InternalThreadState *Thread) { using namespace FEXCore::IR; using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); uint8_t OpSize = IROp->Size; @@ -287,15 +287,16 @@ void InterpreterCore::ExecuteCode(FEXCore::Core::InternalThreadState *Thread) { case IR::OP_STOREMEM: { #define STORE_DATA(x, y) \ case x: { \ - y *Data = Thread->CTX->MemoryMapper.GetBaseOffset(*GetSrc(Op->Header.Args[0])); \ + uint64_t SrcPtr = *GetSrc(Op->Header.Args[0]); \ + y *Data = Thread->CTX->MemoryMapper.GetPointer(SrcPtr); \ LogMan::Throw::A(Data != nullptr, "Couldn't Map pointer to 0x%lx for size %d store\n", *GetSrc(Op->Header.Args[0]), x);\ - *Data = *GetSrc(Op->Header.Args[1]); \ + memcpy(Data, GetSrc(Op->Header.Args[1]), sizeof(y)); \ } \ break auto Op = IROp->C(); - //LogMan::Msg::D("Storing guestmem: 0x%lx (%d)", *GetSrc(Op->Header.Args[0]), Op->Size); - //LogMan::Msg::D("\tStoring: 0x%016lx", (uint64_t)*GetSrc(Op->Header.Args[1])); + // LogMan::Msg::D("Storing guestmem: 0x%lx (%d)", *GetSrc(Op->Header.Args[0]), Op->Size); + // LogMan::Msg::D("\tStoring: 0x%016lx", (uint64_t)*GetSrc(Op->Header.Args[1])); switch (Op->Size) { STORE_DATA(1, uint8_t); @@ -1449,6 +1450,16 @@ void InterpreterCore::ExecuteCode(FEXCore::Core::InternalThreadState *Thread) { default: LogMan::Msg::A("Unknown LREM Size: %d", Size); break; } break; + } + case IR::OP_VEXTR: { + auto Op = IROp->C(); + __uint128_t Src1 = *GetSrc<__uint128_t*>(Op->Header.Args[0]); + __uint128_t Src2 = *GetSrc<__uint128_t*>(Op->Header.Args[1]); + + uint8_t Offset = Op->Index * 8; + __uint128_t Dst = (Src1 << (sizeof(__uint128_t) - Offset)) | (Src2 >> Offset); + memcpy(GDP, &Dst, 16); + break; } default: diff --git a/Source/Interface/Core/JIT/Arm64/JIT.cpp b/Source/Interface/Core/JIT/Arm64/JIT.cpp index b691281a0..0e03165d3 100644 --- a/Source/Interface/Core/JIT/Arm64/JIT.cpp +++ b/Source/Interface/Core/JIT/Arm64/JIT.cpp @@ -1,10 +1,12 @@ #include "Interface/Context/Context.h" -#include "Interface/Core/RegisterAllocation.h" + +#include "Interface/Core/BlockCache.h" #include "Interface/Core/InternalThreadState.h" -#include "Interface/Core/JIT/x86_64/JIT.h" #include "Interface/HLE/Syscalls.h" +#include "Interface/IR/Passes/RegisterAllocationPass.h" + #if _M_X86_64 #define VIXL_INCLUDE_SIMULATOR_AARCH64 #include "aarch64/simulator-aarch64.h" @@ -22,17 +24,17 @@ namespace FEXCore::CPU { using namespace vixl; using namespace vixl::aarch64; -#define STATE x0 -#define MEM_BASE x1 -#define TMP1 x2 -#define TMP2 x3 + +#define MEM_BASE x28 +#define STATE x27 +#define TMP1 x1 +#define TMP2 x2 #define VTMP1 v1 #define VTMP2 v2 #define VTMP3 v3 -static uint64_t SyscallThunk(FEXCore::Core::InternalThreadState *Thread, FEXCore::SyscallHandler *Handler, FEXCore::HLE::SyscallArguments *Args) -{ +static uint64_t SyscallThunk(FEXCore::SyscallHandler *Handler, FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args) { return Handler->HandleSyscall(Thread, Args); } @@ -41,6 +43,16 @@ static void CPUIDThunk(FEXCore::CPUIDEmu *CPUID, uint64_t Function, FEXCore::CPU memcpy(Results, &Res, sizeof(FEXCore::CPUIDEmu::FunctionResults)); } +static uint64_t CompileBlockThunk(FEXCore::Context::Context* CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) { + uint64_t Result = CTX->CompileBlock(Thread, RIP); + return Result; +} + +static uint64_t CompileFallbackBlockThunk(FEXCore::Context::Context* CTX, FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) { + uint64_t Result = CTX->CompileFallbackBlock(Thread, RIP); + return Result; +} + // XXX: Switch from MacroAssembler to Assembler once we drop the simulator class JITCore final : public CPUBackend, public vixl::aarch64::MacroAssembler { public: @@ -57,12 +69,22 @@ public: void SimulationExecution(FEXCore::Core::InternalThreadState *Thread); #endif + bool HasCustomDispatch() const override { return CustomDispatchGenerated; } + +#if _M_X86_64 + void ExecuteCustomDispatch(FEXCore::Core::ThreadState *Thread) override; +#else + void ExecuteCustomDispatch(FEXCore::Core::ThreadState *Thread) override { + DispatchPtr(reinterpret_cast(Thread->InternalState)); + } +#endif + private: FEXCore::Context::Context *CTX; FEXCore::Core::InternalThreadState *State; FEXCore::IR::IRListView const *CurrentIR; - std::unordered_map JumpTargets; + std::map JumpTargets; /** * @name Register Allocation @@ -73,21 +95,19 @@ private: constexpr static uint32_t RegisterClasses = 2; constexpr static uint32_t GPRBase = 0; - constexpr static uint32_t GPRClass = 0; + constexpr static uint32_t GPRClass = IR::RegisterAllocationPass::GPRClass; constexpr static uint32_t FPRBase = NumGPRs; - constexpr static uint32_t FPRClass = 1; + constexpr static uint32_t FPRClass = IR::RegisterAllocationPass::FPRClass; - RA::RegisterSet *RASet; + IR::RegisterAllocationPass::RegisterSet *RASet; /** @} */ - void FindNodeClasses(); - bool CalculateLiveRange(uint32_t Nodes); constexpr static uint8_t RA_32 = 0; constexpr static uint8_t RA_64 = 1; constexpr static uint8_t RA_FPR = 2; bool HasRA = false; - RA::RegisterGraph *Graph; + IR::RegisterAllocationPass::RegisterGraph *Graph; uint32_t GetPhys(uint32_t Node); template @@ -118,9 +138,27 @@ private: std::unordered_map> HostToGuest; #endif void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant); + + void CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread); + bool CustomDispatchGenerated {false}; + using CustomDispatch = void(*)(FEXCore::Core::InternalThreadState *Thread); + CustomDispatch DispatchPtr{}; + IR::RegisterAllocationPass *RAPass; + +#if _M_X86_64 + uint64_t CustomDispatchEnd; +#endif }; #if _M_X86_64 +void JITCore::ExecuteCustomDispatch(FEXCore::Core::ThreadState *Thread) { + PrintDisassembler PrintDisasm(stdout); + PrintDisasm.DisassembleBuffer(vixl::aarch64::Instruction::Cast(DispatchPtr), vixl::aarch64::Instruction::Cast(CustomDispatchEnd)); + + Sim.WriteXRegister(0, reinterpret_cast(Thread)); + Sim.RunFrom(vixl::aarch64::Instruction::Cast(DispatchPtr)); +} + static void SimulatorExecution(FEXCore::Core::InternalThreadState *Thread) { JITCore *Core = reinterpret_cast(Thread->CPUBackend.get()); Core->SimulationExecution(Thread); @@ -129,8 +167,8 @@ static void SimulatorExecution(FEXCore::Core::InternalThreadState *Thread) { void JITCore::SimulationExecution(FEXCore::Core::InternalThreadState *Thread) { using namespace vixl::aarch64; auto SimulatorAddress = HostToGuest[Thread->State.State.rip]; - //PrintDisassembler PrintDisasm(stdout); - //PrintDisasm.DisassembleBuffer(vixl::aarch64::Instruction::Cast(SimulatorAddress.first), vixl::aarch64::Instruction::Cast(SimulatorAddress.second)); + // PrintDisassembler PrintDisasm(stdout); + // PrintDisasm.DisassembleBuffer(vixl::aarch64::Instruction::Cast(SimulatorAddress.first), vixl::aarch64::Instruction::Cast(SimulatorAddress.second)); Sim.WriteXRegister(0, reinterpret_cast(Thread)); Sim.RunFrom(vixl::aarch64::Instruction::Cast(SimulatorAddress.first)); @@ -149,11 +187,14 @@ JITCore::JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadSt // XXX: Set this to a real minimum feature set in the future SetCPUFeatures(vixl::CPUFeatures::All()); - RASet = RA::AllocateRegisterSet(RegisterCount, RegisterClasses); - RA::AddRegisters(RASet, GPRClass, GPRBase, NumGPRs); - RA::AddRegisters(RASet, FPRClass, FPRBase, NumFPRs); + RAPass = CTX->GetRegisterAllocatorPass(); + RAPass->SetSupportsSpills(false); - Graph = RA::AllocateRegisterGraph(RASet, 9000); + RASet = RAPass->AllocateRegisterSet(RegisterCount, RegisterClasses); + RAPass->AddRegisters(RASet, GPRClass, GPRBase, NumGPRs); + RAPass->AddRegisters(RASet, FPRClass, FPRBase, NumFPRs); + + Graph = RAPass->AllocateRegisterGraph(RASet, 9000); LiveRanges.resize(9000); // Just set the entire range as executable @@ -165,11 +206,13 @@ JITCore::JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadSt #if _M_X86_64 Sim.SetCPUFeatures(vixl::CPUFeatures::All()); #endif + SetAllowAssembler(true); + CreateCustomDispatch(Thread); } JITCore::~JITCore() { - FreeRegisterGraph(Graph); - FreeRegisterSet(RASet); + RAPass->FreeRegisterGraph(); + RAPass->FreeRegisterSet(RASet); } void JITCore::LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant) { @@ -202,7 +245,7 @@ const std::array RAFPR = { v29, v30, v31}; uint32_t JITCore::GetPhys(uint32_t Node) { - uint32_t Reg = RA::GetNodeRegister(Graph, Node); + uint32_t Reg = RAPass->GetNodeRegister(Node); if (Reg < FPRBase) return Reg; @@ -242,114 +285,6 @@ aarch64::VRegister JITCore::GetDst(uint32_t Node) { return RAFPR[Reg]; } -void JITCore::FindNodeClasses() { - uintptr_t ListBegin = CurrentIR->GetListData(); - uintptr_t DataBegin = CurrentIR->GetData(); - - IR::NodeWrapperIterator Begin = CurrentIR->begin(); - IR::NodeWrapperIterator End = CurrentIR->end(); - - while (Begin != End) { - using namespace FEXCore::IR; - - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); - - if (IROp->HasDest) { - // XXX: This needs to be better - switch (IROp->Op) { - case OP_LOADCONTEXT: { - auto Op = IROp->C(); - if (Op->Size == 16) - RA::SetNodeClass(Graph, WrapperOp->ID(), FPRClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - case IR::OP_LOADMEM: { - auto Op = IROp->C(); - if (Op->Size == 16) - RA::SetNodeClass(Graph, WrapperOp->ID(), FPRClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - - case OP_ZEXT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - - if (Op->SrcSize == 64) { - RA::SetNodeClass(Graph, WrapperOp->ID(), FPRClass); - } - else { - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - } - break; - } - case OP_CPUID: RA::SetNodeClass(Graph, WrapperOp->ID(), FPRClass); break; - default: - if (IROp->Op >= IR::OP_VOR) - RA::SetNodeClass(Graph, WrapperOp->ID(), FPRClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - } - ++Begin; - } -} - -bool JITCore::CalculateLiveRange(uint32_t Nodes) { - if (Nodes > LiveRanges.size()) { - LiveRanges.resize(Nodes); - } - memset(&LiveRanges.at(0), 0xFF, Nodes * sizeof(LiveRange)); - - uintptr_t ListBegin = CurrentIR->GetListData(); - uintptr_t DataBegin = CurrentIR->GetData(); - - IR::NodeWrapperIterator Begin = CurrentIR->begin(); - IR::NodeWrapperIterator End = CurrentIR->end(); - - while (Begin != End) { - using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); - - uint32_t Node = WrapperOp->ID(); - - // If the destination hasn't yet been set then set it now - if (IROp->HasDest && LiveRanges[Node].Begin == ~0U) { - LiveRanges[Node].Begin = Node; - // Default to ending right where it starts - LiveRanges[Node].End = Node; - } - - for (uint8_t i = 0; i < IROp->NumArgs; ++i) { - uint32_t ArgNode = IROp->Args[i].ID(); - // Set the node end to be at least here - LiveRanges[ArgNode].End = Node; - } - - ++Begin; - } - - // Now that we have all the live ranges calculated we need to add them to our interference graph - for (uint32_t i = 0; i < Nodes; ++i) { - for (uint32_t j = i + 1; j < Nodes; ++j) { - if (!(LiveRanges[i].Begin >= LiveRanges[j].End || - LiveRanges[j].Begin >= LiveRanges[i].End)) { - RA::AddNodeInterference(Graph, i, j); - } - } - } - - return RA::AllocateRegisters(Graph); -} - void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData) { using namespace aarch64; JumpTargets.clear(); @@ -362,11 +297,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const IR::NodeWrapperIterator End = CurrentIR->end(); uintptr_t ListSize = CurrentIR->GetListSize(); - uint32_t SSACount = ListSize / sizeof(IR::OrderedNode); - - ResetRegisterGraph(Graph, SSACount); - FindNodeClasses(); - HasRA = CalculateLiveRange(SSACount); + HasRA = RAPass->HasFullRA(); LogMan::Throw::A(HasRA, "Arm64 JIT only works with RA"); @@ -393,14 +324,17 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const auto Buffer = GetBuffer(); auto Entry = Buffer->GetOffsetAddress(GetCursorOffset()); - void *Memory = CTX->MemoryMapper.GetMemoryBase(); - LoadConstant(MEM_BASE, (uint64_t)Memory); + if (!CustomDispatchGenerated) { + void *Memory = CTX->MemoryMapper.GetMemoryBase(); + LoadConstant(MEM_BASE, (uint64_t)Memory); + mov(STATE, x0); + } while (Begin != End) { using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); uint8_t OpSize = IROp->Size; uint32_t Node = WrapperOp->ID(); @@ -411,7 +345,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const auto Name = FEXCore::IR::GetName(IROp->Op); if (IROp->HasDest) { - uint32_t PhysReg = RA::GetNodeRegister(Graph, Node); + uint32_t PhysReg = RAPass->GetNodeRegister(Node); if (PhysReg >= FPRBase) Inst << "\tFPR" << GetPhys(Node) << " = " << Name << " "; else @@ -423,14 +357,12 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const for (uint8_t i = 0; i < IROp->NumArgs; ++i) { uint32_t ArgNode = IROp->Args[i].ID(); - uint32_t PhysReg = RA::GetNodeRegister(Graph, ArgNode); + uint32_t PhysReg = RAPass->GetNodeRegister(ArgNode); if (PhysReg >= FPRBase) Inst << "FPR" << GetPhys(ArgNode) << (i + 1 == IROp->NumArgs ? "" : ", "); else Inst << "Reg" << GetPhys(ArgNode) << (i + 1 == IROp->NumArgs ? "" : ", "); } - - LogMan::Msg::D("%s", Inst.str().c_str()); } } @@ -439,10 +371,10 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const auto IsTarget = JumpTargets.find(WrapperOp->ID()); if (IsTarget == JumpTargets.end()) { // XXX: This is a memory leak - JumpTargets.try_emplace(WrapperOp->ID(), new aarch64::Label); + JumpTargets.try_emplace(WrapperOp->ID()); } else { - bind(IsTarget->second); + bind(&IsTarget->second); } break; } @@ -467,7 +399,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const // X1: ThreadState // X2: Pointer to SyscallArguments - uint64_t SPOffset = AlignUp((2 + RA64.size() + 7 + 2) * 8, 16); + uint64_t SPOffset = AlignUp((RA64.size() + 7 + 1) * 8, 16); sub(sp, sp, SPOffset); for (uint32_t i = 0; i < 7; ++i) @@ -478,14 +410,26 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const str(RA, MemOperand(sp, 7 * 8 + i * 8)); i++; } - str(STATE, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 1 * 8)); - str(MEM_BASE, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 2 * 8)); - str(lr, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 3 * 8)); + str(lr, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 0 * 8)); - // x0 = threadstate already - LoadConstant(x1, reinterpret_cast(&CTX->SyscallHandler)); - mov (x2, sp); + LoadConstant(x0, reinterpret_cast(&CTX->SyscallHandler)); + mov(x1, STATE); + mov(x2, sp); + +#if _M_X86_64 CallRuntime(SyscallThunk); +#else + using ClassPtrType = uint64_t (FEXCore::SyscallHandler::*)(FEXCore::Core::InternalThreadState *, FEXCore::HLE::SyscallArguments *); + union PtrCast { + ClassPtrType ClassPtr; + uintptr_t Data; + }; + + PtrCast Ptr; + Ptr.ClassPtr = &FEXCore::SyscallHandler::HandleSyscall; + LoadConstant(x3, Ptr.Data); + blr(x3); +#endif // Result is now in x0 // Fix the stack and any values that were stepped on @@ -498,9 +442,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const // Move result to its destination register mov(GetDst(Node), x0); - ldr(STATE, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 1 * 8)); - ldr(MEM_BASE, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 2 * 8)); - ldr(lr, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 3 * 8)); + ldr(lr, MemOperand(sp, 7 * 8 + RA64.size() * 8 + 0 * 8)); add(sp, sp, SPOffset); break; @@ -517,9 +459,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const i++; } - str(STATE, MemOperand(sp, RA64.size() * 8 + 0 * 8)); - str(MEM_BASE, MemOperand(sp, RA64.size() * 8 + 1 * 8)); - str(lr, MemOperand(sp, RA64.size() * 8 + 2 * 8)); + str(lr, MemOperand(sp, RA64.size() * 8 + 0 * 8)); // x0 = CPUID Handler // x1 = CPUID Function @@ -540,9 +480,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const auto Dst = GetDst(Node); ldr(Dst, MemOperand(sp, RA64.size() * 8 + 3 * 8)); - ldr(STATE, MemOperand(sp, RA64.size() * 8 + 0 * 8)); - ldr(MEM_BASE, MemOperand(sp, RA64.size() * 8 + 1 * 8)); - ldr(lr, MemOperand(sp, RA64.size() * 8 + 2 * 8)); + ldr(lr, MemOperand(sp, RA64.size() * 8 + 0 * 8)); add(sp, sp, SPOffset); @@ -551,7 +489,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const case IR::OP_EXTRACTELEMENT: { auto Op = IROp->C(); - uint32_t PhysReg = RA::GetNodeRegister(Graph, Op->Header.Args[0].ID()); + uint32_t PhysReg = RAPass->GetNodeRegister(Op->Header.Args[0].ID()); if (PhysReg >= FPRBase) { switch (OpSize) { case 4: @@ -574,16 +512,15 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const Label *TargetLabel; auto IsTarget = JumpTargets.find(Op->Header.Args[0].ID()); if (IsTarget == JumpTargets.end()) { - TargetLabel = JumpTargets.try_emplace(Op->Header.Args[0].ID(), new aarch64::Label).first->second; + TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[0].ID()).first->second; } else { - TargetLabel = IsTarget->second; + TargetLabel = &IsTarget->second; } b(TargetLabel); break; } - case IR::OP_CONDJUMP: { auto Op = IROp->C(); @@ -591,10 +528,10 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const auto IsTarget = JumpTargets.find(Op->Header.Args[1].ID()); if (IsTarget == JumpTargets.end()) { // XXX: This is a memory leak - TargetLabel = JumpTargets.try_emplace(Op->Header.Args[1].ID(), new aarch64::Label).first->second; + TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[1].ID()).first->second; } else { - TargetLabel = IsTarget->second; + TargetLabel = &IsTarget->second; } cbnz(GetSrc(Op->Header.Args[0].ID()), TargetLabel); @@ -709,7 +646,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const lsrv(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); else lsrv(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; + break; } case IR::OP_ASHR: { auto Op = IROp->C(); @@ -717,7 +654,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const asrv(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); else asrv(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; + break; } case IR::OP_LSHL: { auto Op = IROp->C(); @@ -725,7 +662,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const lslv(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); else lslv(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; + break; } case IR::OP_ROR: { auto Op = IROp->C(); @@ -745,7 +682,6 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } break; } - case IR::OP_ROL: { auto Op = IROp->C(); uint8_t Mask = OpSize * 8 - 1; @@ -768,7 +704,6 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } break; } - case IR::OP_SEXT: { auto Op = IROp->C(); LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); @@ -794,7 +729,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const case IR::OP_ZEXT: { auto Op = IROp->C(); LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - uint32_t PhysReg = RA::GetNodeRegister(Graph, Op->Header.Args[0].ID()); + uint32_t PhysReg = RAPass->GetNodeRegister(Op->Header.Args[0].ID()); if (PhysReg >= FPRBase) { // FPR -> GPR transfer with free truncation switch (Op->SrcSize) { @@ -886,19 +821,20 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } break; } - case IR::OP_BFE: { auto Op = IROp->C(); LogMan::Throw::A(OpSize <= 16, "OpSize is too large for BFE: %d", OpSize); + LogMan::Throw::A(Op->Width != 0, "Invalid BFE width of 0"); + auto Dst = GetDst(Node); if (OpSize == 16) { LogMan::Throw::A(!(Op->lsb < 64 && (Op->lsb + Op->Width > 64)), "Trying to BFE an XMM across the 64bit split: Beginning at %d, ending at %d", Op->lsb, Op->lsb + Op->Width); uint8_t Offset = Op->lsb; if (Offset < 64) { - mov(Dst, GetSrc(Op->Header.Args[0].ID()), 0); + mov(Dst, GetSrc(Op->Header.Args[0].ID()).D(), 0); } else { - mov(Dst, GetSrc(Op->Header.Args[0].ID()), 1); + mov(Dst, GetSrc(Op->Header.Args[0].ID()).D(), 1); Offset -= 64; } @@ -912,18 +848,20 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } else { lsr(Dst, GetSrc(Op->Header.Args[0].ID()), Op->lsb); - and_(Dst, Dst, ((1ULL << Op->Width) - 1)); + if (Op->Width != 64) { + and_(Dst, Dst, ((1ULL << Op->Width) - 1)); + } } break; } case IR::OP_POPCOUNT: { auto Op = IROp->C(); auto Dst = GetDst(Node); - fmov(VTMP1, GetSrc(Op->Header.Args[0].ID())); + fmov(VTMP1.V1D(), GetSrc(Op->Header.Args[0].ID())); cnt(VTMP1.V8B(), VTMP1.V8B()); addv(VTMP1.B(), VTMP1.V8B()); - umov(Dst, VTMP1.B(), 0); - break; + umov(Dst.W(), VTMP1.B(), 0); + break; } case IR::OP_FINDLSB: { auto Op = IROp->C(); @@ -980,7 +918,6 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const mov(GetDst(Node), TMP2); break; } - case IR::OP_SELECT: { auto Op = IROp->C(); @@ -1162,7 +1099,6 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } break; } - case IR::OP_LUREM: { auto Op = IROp->C(); // Each source is OpSize in size @@ -1219,7 +1155,6 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } break; } - case IR::OP_VINSELEMENT: { auto Op = IROp->C(); mov(VTMP1, GetSrc(Op->Header.Args[0].ID())); @@ -1397,17 +1332,28 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const eor(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B()); break; } + case IR::OP_VEXTR: { + auto Op = IROp->C(); + // AArch64 ext op has bit arrangement as [Vm:Vn] so arguments need to be swapped + ext(GetDst(Node).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), Op->Index); + break; + } case IR::OP_VUSHLS: { auto Op = IROp->C(); switch (Op->ElementSize) { + case 1: { + dup(VTMP1.V16B(), GetSrc(Op->Header.Args[1].ID())); + ushl(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), VTMP1.V16B()); + break; + } case 2: { - dup(VTMP1.V8H(), GetSrc(Op->Header.Args[1].ID())); + dup(VTMP1.V8H(), GetSrc(Op->Header.Args[1].ID())); ushl(GetDst(Node).V8H(), GetSrc(Op->Header.Args[0].ID()).V8H(), VTMP1.V8H()); break; } case 4: { - dup(VTMP1.V4S(), GetSrc(Op->Header.Args[1].ID())); + dup(VTMP1.V4S(), GetSrc(Op->Header.Args[1].ID())); ushl(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S(), VTMP1.V4S()); break; } @@ -1420,11 +1366,58 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const } break; } + case IR::OP_VUMIN: { + auto Op = IROp->C(); + switch (Op->ElementSize) { + case 1: { + umin(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B()); + break; + } + case 2: { + umin(GetDst(Node).V8H(), GetSrc(Op->Header.Args[0].ID()).V8H(), GetSrc(Op->Header.Args[1].ID()).V8H()); + break; + } + case 4: { + umin(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S(), GetSrc(Op->Header.Args[1].ID()).V4S()); + break; + } + case 8: { + umin(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D(), GetSrc(Op->Header.Args[1].ID()).V2D()); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + break; + } + case IR::OP_VSMIN: { + auto Op = IROp->C(); + switch (Op->ElementSize) { + case 1: { + smin(GetDst(Node).V16B(), GetSrc(Op->Header.Args[0].ID()).V16B(), GetSrc(Op->Header.Args[1].ID()).V16B()); + break; + } + case 2: { + smin(GetDst(Node).V8H(), GetSrc(Op->Header.Args[0].ID()).V8H(), GetSrc(Op->Header.Args[1].ID()).V8H()); + break; + } + case 4: { + smin(GetDst(Node).V4S(), GetSrc(Op->Header.Args[0].ID()).V4S(), GetSrc(Op->Header.Args[1].ID()).V4S()); + break; + } + case 8: { + smin(GetDst(Node).V2D(), GetSrc(Op->Header.Args[0].ID()).V2D(), GetSrc(Op->Header.Args[1].ID()).V2D()); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + break; + } case IR::OP_CYCLECOUNTER: { - if (0) +#ifdef DEBUG_CYCLES movz(GetDst(Node), 0); - else +#else mrs(GetDst(Node), CNTVCT_EL0); +#endif break; } default: @@ -1437,14 +1430,174 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const FinalizeCode(); #if _M_X86_64 - auto CodeEnd = Buffer->GetOffsetAddress(GetCursorOffset()); - HostToGuest[State->State.State.rip] = std::make_pair(Entry, CodeEnd); - return (void*)SimulatorExecution; -#else - return reinterpret_cast(Entry); + if (!CustomDispatchGenerated) { + auto CodeEnd = Buffer->GetOffsetAddress(GetCursorOffset()); + HostToGuest[State->State.State.rip] = std::make_pair(Entry, CodeEnd); + return (void*)SimulatorExecution; + } #endif + + return reinterpret_cast(Entry); } +void JITCore::CreateCustomDispatch(FEXCore::Core::InternalThreadState *Thread) { + auto Buffer = GetBuffer(); + DispatchPtr = Buffer->GetOffsetAddress(GetCursorOffset()); + EmissionCheckScope(this, 0); + + // while (!Thread->State.RunningEvents.ShouldStop.load()) { + // Ptr = FindBlock(RIP) + // if (!Ptr) + // Ptr = CTX->CompileBlock(RIP); + // + // if (Ptr) + // Ptr(); + // else + // { + // Ptr = FallbackCore->CompileBlock() + // if (Ptr) + // Ptr() + // else { + // ShouldStop = true; + // } + // } + // } + + // Push all the register we need to save + PushCalleeSavedRegisters(); + + // Push our memory base to the correct register + void *Memory = CTX->MemoryMapper.GetMemoryBase(); + LoadConstant(MEM_BASE, (uint64_t)Memory); + // Move our thread pointer to the correct register + // This is passed in to parameter 0 (x0) + mov(STATE, x0); + + aarch64::Label LoopTop; + bind(&LoopTop); + + // Load in our RIP + ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::ThreadState, State.rip))); + LoadConstant(x0, Thread->BlockCache->GetPagePointer()); + + // Steal the page offset + and_(x1, x2, 0x0FFF); + // Offset the address and add to our page pointer + add(x3, x0, Operand(x2, LSR, 12)); + + // Load the pointer from the offset + ldr(x3, MemOperand(x3)); + aarch64::Label NoBlock; + + // If page pointer is zero then we have no block + cbz(x3, &NoBlock); + + // Now load from that pointer offset by the page offset to get our real block + ldr(x3, MemOperand(x3, x1)); + cbz(x3, &NoBlock); + + // If we've made it here then we have a real compiled block + { + blr(x3); + } + + aarch64::Label ExitCheck; + bind(&ExitCheck); + + constexpr uint64_t ShouldStopOffset = offsetof(FEXCore::Core::ThreadState, RunningEvents.ShouldStop); + // If we don't need to stop then keep going + add(x1, STATE, ShouldStopOffset); + ldarb(x0, MemOperand(x1)); + cbz(x0, &LoopTop); + + PopCalleeSavedRegisters(); + + // Return from the function + // LR is set to the correct return location now + ret(); + + aarch64::Label FallbackCore; + // Need to create the block + { + bind(&NoBlock); + + LoadConstant(x0, reinterpret_cast(CTX)); + mov(x1, STATE); + +#if _M_X86_64 + CallRuntime(CompileBlockThunk); +#else + using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::InternalThreadState *, uint64_t); + union PtrCast { + ClassPtrType ClassPtr; + uintptr_t Data; + }; + + PtrCast Ptr; + Ptr.ClassPtr = &FEXCore::Context::Context::CompileBlock; + LoadConstant(x3, Ptr.Data); + + // X2 contains our guest RIP + blr(x3); // { ThreadState, RIP} +#endif + // X0 now contains either nullptr or block pointer + cbz(x0, &FallbackCore); + blr(x0); + + b(&ExitCheck); + } + + aarch64::Label ExitError; + // We need to fallback to our fallback core + { + bind(&FallbackCore); + +#if _M_X86_64 + // XXX: Fallback core doesn't work on x86-64 + // We can't tell the difference between simulator entry points and not + b(&ExitError); +#else + LoadConstant(x0, reinterpret_cast(CTX)); + mov(x1, STATE); + + using ClassPtrType = uintptr_t (FEXCore::Context::Context::*)(FEXCore::Core::InternalThreadState *, uint64_t); + union PtrCast { + ClassPtrType ClassPtr; + uintptr_t Data; + }; + + PtrCast Ptr; + Ptr.ClassPtr = &FEXCore::Context::Context::CompileFallbackBlock; + LoadConstant(x3, Ptr.Data); + + // X2 contains our guest RIP + blr(x3); // {ThreadState, RIP} +#endif + // X0 now contains either nullptr or block pointer + cbz(x0, &ExitError); + blr(x0); + + b(&ExitCheck); + } + + // Exit error + { + bind(&ExitError); + LoadConstant(x0, 1); + add(x1, STATE, ShouldStopOffset); + stlrb(x0, MemOperand(x1)); + b(&ExitCheck); + } + +#if _M_X86_64 + CustomDispatchEnd = Buffer->GetOffsetAddress(GetCursorOffset()); +#endif + + FinalizeCode(); + // XXX: Crashes currently. + // Disabling will be useful for debugging ThreadState + // CustomDispatchGenerated = true; +} FEXCore::CPU::CPUBackend *CreateJITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) { return new JITCore(ctx, Thread); diff --git a/Source/Interface/Core/JIT/x86_64/JIT.cpp b/Source/Interface/Core/JIT/x86_64/JIT.cpp index b0468cce7..ea6c8deba 100644 --- a/Source/Interface/Core/JIT/x86_64/JIT.cpp +++ b/Source/Interface/Core/JIT/x86_64/JIT.cpp @@ -1,6 +1,6 @@ #include "Interface/Context/Context.h" -#include "Interface/Core/RegisterAllocation.h" #include "Interface/Core/InternalThreadState.h" +#include "Interface/IR/Passes/RegisterAllocationPass.h" #include "Interface/Core/JIT/x86_64/JIT.h" #include @@ -9,6 +9,7 @@ using namespace Xbyak; #include #include #include +// #define DEBUG_RA 1 namespace FEXCore::CPU { // Temp registers @@ -24,6 +25,14 @@ namespace FEXCore::CPU { // r11 assigned to temp state #define TEMP_STACK r11 #define STATE rdi +using namespace Xbyak::util; +const std::array RA64 = { rsi, r8, r9, r10, r11, rbx, rbp, r12, r13, r14, r15 }; +const std::array RA32 = { esi, r8d, r9d, r10d, r11d, ebx, ebp, r12d, r13d, r14d, r15d }; +const std::array RA16 = { si, r8w, r9w, r10w, r11w, bx, bp, r12w, r13w, r14w, r15w }; +const std::array RA8 = { sil, r8b, r9b, r10b, r11b, bl, bpl, r12b, r13b, r14b, r15b }; +const std::array RAXMM = { xmm0, xmm1, xmm2, xmm3, xmm4, xmm5, xmm6, xmm7, xmm8, xmm9, xmm10 }; +const std::array RAXMM_x = { xmm0, xmm1, xmm2, xmm3, xmm4, xmm5, xmm6, xmm7, xmm8, xmm9, xmm10 }; + class JITCore final : public CPUBackend, public Xbyak::CodeGenerator { public: @@ -36,10 +45,16 @@ public: bool NeedsOpDispatch() override { return true; } + bool HasCustomDispatch() const override { return CustomDispatchGenerated; } + + void ExecuteCustomDispatch(FEXCore::Core::ThreadState *Thread) override { + DispatchPtr(reinterpret_cast(Thread->InternalState)); + } + private: FEXCore::Context::Context *CTX; FEXCore::IR::IRListView const *CurrentIR; - std::unordered_map JumpTargets; + std::unordered_map JumpTargets; std::vector Stack; bool MemoryDebug = false; @@ -47,21 +62,19 @@ private: /** * @name Register Allocation * @{ */ - constexpr static uint32_t NumGPRs = 11; - constexpr static uint32_t NumXMMs = 11; + constexpr static uint32_t NumGPRs = RA64.size(); // 4 is the minimum required for GPR ops + constexpr static uint32_t NumXMMs = RAXMM.size(); constexpr static uint32_t RegisterCount = NumGPRs + NumXMMs; constexpr static uint32_t RegisterClasses = 2; constexpr static uint32_t GPRBase = 0; - constexpr static uint32_t GPRClass = 0; + constexpr static uint32_t GPRClass = IR::RegisterAllocationPass::GPRClass; constexpr static uint32_t XMMBase = NumGPRs; - constexpr static uint32_t XMMClass = 1; + constexpr static uint32_t XMMClass = IR::RegisterAllocationPass::FPRClass; - RA::RegisterSet *RASet; + IR::RegisterAllocationPass::RegisterSet *RASet; /** @} */ - void FindNodeClasses(); - bool CalculateLiveRange(uint32_t Nodes); constexpr static uint8_t RA_8 = 0; constexpr static uint8_t RA_16 = 1; constexpr static uint8_t RA_32 = 2; @@ -69,7 +82,7 @@ private: constexpr static uint8_t RA_XMM = 4; bool HasRA = false; - RA::RegisterGraph *Graph; + IR::RegisterAllocationPass::RegisterGraph *Graph; uint32_t GetPhys(uint32_t Node); template @@ -81,12 +94,11 @@ private: Xbyak::Xmm GetSrc(uint32_t Node); Xbyak::Xmm GetDst(uint32_t Node); - struct LiveRange { - uint32_t Begin; - uint32_t End; - }; - - std::vector LiveRanges; + void CreateCustomDispatch(); + bool CustomDispatchGenerated {false}; + using CustomDispatch = void(*)(FEXCore::Core::InternalThreadState *Thread); + CustomDispatch DispatchPtr{}; + IR::RegisterAllocationPass *RAPass; }; JITCore::JITCore(FEXCore::Context::Context *ctx) @@ -94,18 +106,21 @@ JITCore::JITCore(FEXCore::Context::Context *ctx) , CTX {ctx} { Stack.resize(9000 * 16 * 64); - RASet = RA::AllocateRegisterSet(RegisterCount, RegisterClasses); - RA::AddRegisters(RASet, GPRClass, GPRBase, NumGPRs); - RA::AddRegisters(RASet, XMMClass, XMMBase, NumXMMs); + RAPass = CTX->GetRegisterAllocatorPass(); + RAPass->SetSupportsSpills(false); - Graph = RA::AllocateRegisterGraph(RASet, 9000); - LiveRanges.resize(9000); + RASet = RAPass->AllocateRegisterSet(RegisterCount, RegisterClasses); + RAPass->AddRegisters(RASet, GPRClass, GPRBase, NumGPRs); + RAPass->AddRegisters(RASet, XMMClass, XMMBase, NumXMMs); + + Graph = RAPass->AllocateRegisterGraph(RASet, 9000); + CreateCustomDispatch(); } JITCore::~JITCore() { printf("Used %ld bytes for compiling\n", getCurr() - getCode()); - FreeRegisterGraph(Graph); - FreeRegisterSet(RASet); + RAPass->FreeRegisterSet(RASet); + RAPass->FreeRegisterGraph(); } static void LoadMem(uint64_t Addr, uint64_t Data, uint8_t Size) { @@ -118,124 +133,8 @@ static void StoreMem(uint64_t Addr, uint64_t Data, uint8_t Size) { LogMan::Msg::D("\tStoring: 0x%016lx", Data); } -void JITCore::FindNodeClasses() { - uintptr_t ListBegin = CurrentIR->GetListData(); - uintptr_t DataBegin = CurrentIR->GetData(); - - IR::NodeWrapperIterator Begin = CurrentIR->begin(); - IR::NodeWrapperIterator End = CurrentIR->end(); - - while (Begin != End) { - using namespace FEXCore::IR; - - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); - - if (IROp->HasDest) { - // XXX: This needs to be better - switch (IROp->Op) { - case OP_LOADCONTEXT: { - auto Op = IROp->C(); - if (Op->Size == 16) - RA::SetNodeClass(Graph, WrapperOp->ID(), XMMClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - case OP_STORECONTEXT: { - auto Op = IROp->C(); - if (Op->Size == 16) - RA::SetNodeClass(Graph, WrapperOp->ID(), XMMClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - case IR::OP_LOADMEM: { - auto Op = IROp->C(); - if (Op->Size == 16) - RA::SetNodeClass(Graph, WrapperOp->ID(), XMMClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - - case OP_ZEXT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - - if (Op->SrcSize == 64) { - RA::SetNodeClass(Graph, WrapperOp->ID(), XMMClass); - } - else { - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - } - break; - } - case OP_CPUID: RA::SetNodeClass(Graph, WrapperOp->ID(), XMMClass); break; - default: - if (IROp->Op >= IR::OP_VOR) - RA::SetNodeClass(Graph, WrapperOp->ID(), XMMClass); - else - RA::SetNodeClass(Graph, WrapperOp->ID(), GPRClass); - break; - } - } - ++Begin; - } -} - -bool JITCore::CalculateLiveRange(uint32_t Nodes) { - if (Nodes > LiveRanges.size()) { - LiveRanges.resize(Nodes); - } - memset(&LiveRanges.at(0), 0xFF, Nodes * sizeof(LiveRange)); - - uintptr_t ListBegin = CurrentIR->GetListData(); - uintptr_t DataBegin = CurrentIR->GetData(); - - IR::NodeWrapperIterator Begin = CurrentIR->begin(); - IR::NodeWrapperIterator End = CurrentIR->end(); - - while (Begin != End) { - using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); - - uint32_t Node = WrapperOp->ID(); - - // If the destination hasn't yet been set then set it now - if (IROp->HasDest && LiveRanges[Node].Begin == ~0U) { - LiveRanges[Node].Begin = Node; - // Default to ending right where it starts - LiveRanges[Node].End = Node; - } - - for (uint8_t i = 0; i < IROp->NumArgs; ++i) { - uint32_t ArgNode = IROp->Args[i].ID(); - // Set the node end to be at least here - LiveRanges[ArgNode].End = Node; - } - - ++Begin; - } - - // Now that we have all the live ranges calculated we need to add them to our interference graph - for (uint32_t i = 0; i < Nodes; ++i) { - for (uint32_t j = i + 1; j < Nodes; ++j) { - if (!(LiveRanges[i].Begin >= LiveRanges[j].End || - LiveRanges[j].Begin >= LiveRanges[i].End)) { - RA::AddNodeInterference(Graph, i, j); - } - } - } - - return RA::AllocateRegisters(Graph); -} - uint32_t JITCore::GetPhys(uint32_t Node) { - uint32_t Reg = RA::GetNodeRegister(Graph, Node); + uint32_t Reg = RAPass->GetNodeRegister(Node); if (Reg < XMMBase) return Reg; @@ -247,14 +146,6 @@ uint32_t JITCore::GetPhys(uint32_t Node) { return ~0U; } -using namespace Xbyak::util; -const std::array RA64 = { rsi, r8, r9, r10, r11, rbx, rbp, r12, r13, r14, r15 }; -const std::array RA32 = { esi, r8d, r9d, r10d, r11d, ebx, ebp, r12d, r13d, r14d, r15d }; -const std::array RA16 = { si, r8w, r9w, r10w, r11w, bx, bp, r12w, r13w, r14w, r15w }; -const std::array RA8 = { sil, r8b, r9b, r10b, r11b, bl, bpl, r12b, r13b, r14b, r15b }; -const std::array RAXMM = { xmm0, xmm1, xmm2, xmm3, xmm4, xmm5, xmm6, xmm7, xmm8, xmm9, xmm10 }; -const std::array RAXMM_x = { xmm0, xmm1, xmm2, xmm3, xmm4, xmm5, xmm6, xmm7, xmm8, xmm9, xmm10 }; - template Xbyak::Reg JITCore::GetSrc(uint32_t Node) { // rax, rcx, rdx, rsi, r8, r9, @@ -306,12 +197,7 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const uintptr_t ListBegin = CurrentIR->GetListData(); uintptr_t DataBegin = CurrentIR->GetData(); - IR::NodeWrapperIterator Begin = CurrentIR->begin(); - IR::NodeWrapperIterator End = CurrentIR->end(); - - uintptr_t ListSize = CurrentIR->GetListSize(); - - uint32_t SSACount = ListSize / sizeof(IR::OrderedNode); + uint32_t SSACount = CurrentIR->GetSSACount(); uint64_t ListStackSize = SSACount * 16; if (ListStackSize > Stack.size()) { Stack.resize(ListStackSize); @@ -319,10 +205,9 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const void *Entry = getCurr(); - ResetRegisterGraph(Graph, SSACount); - FindNodeClasses(); - HasRA = CalculateLiveRange(SSACount); + HasRA = RAPass->HasFullRA(); + uint32_t SpillSlots = RAPass->SpillSlots(); if (HasRA) { push(rbx); push(rbp); @@ -334,2464 +219,2691 @@ void *JITCore::CompileCode([[maybe_unused]] FEXCore::IR::IRListView const else { mov(TEMP_STACK, reinterpret_cast(&Stack.at(0))); } - while (Begin != End) { - using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); - uint8_t OpSize = IROp->Size; - uint32_t Node = WrapperOp->ID(); - - if (HasRA) { -#ifdef DEBUG_RA - std::stringstream Inst; - auto Name = FEXCore::IR::GetName(IROp->Op); - - if (IROp->HasDest) { - uint32_t PhysReg = RA::GetNodeRegister(Graph, Node); - if (PhysReg >= XMMBase) - Inst << "\tXMM" << GetPhys(Node) << " = " << Name << " "; - else - Inst << "\tReg" << GetPhys(Node) << " = " << Name << " "; - } - else { - Inst << "\t" << Name << " "; - } - - for (uint8_t i = 0; i < IROp->NumArgs; ++i) { - uint32_t ArgNode = IROp->Args[i].ID(); - uint32_t PhysReg = RA::GetNodeRegister(Graph, ArgNode); - if (PhysReg >= XMMBase) - Inst << "XMM" << GetPhys(ArgNode) << (i + 1 == IROp->NumArgs ? "" : ", "); - else - Inst << "Reg" << GetPhys(ArgNode) << (i + 1 == IROp->NumArgs ? "" : ", "); - } - - LogMan::Msg::D("%s", Inst.str().c_str()); -#endif - } - - switch (IROp->Op) { - case IR::OP_BEGINBLOCK: { - auto IsTarget = JumpTargets.find(WrapperOp->ID()); - if (IsTarget == JumpTargets.end()) { - JumpTargets[WrapperOp->ID()] = L(); - } - else { - L(IsTarget->second); - } - break; - } - case IR::OP_ENDBLOCK: { - auto Op = IROp->C(); - if (Op->RIPIncrement) { - add(qword [STATE + offsetof(FEXCore::Core::CPUState, rip)], Op->RIPIncrement); - } - break; - } - case IR::OP_EXITFUNCTION: - case IR::OP_ENDFUNCTION: { - if (HasRA) { - pop(r15); - pop(r14); - pop(r13); - pop(r12); - pop(rbp); - pop(rbx); - } - ret(); - break; - } - case IR::OP_BREAK: { - auto Op = IROp->C(); - switch (Op->Reason) { - case 4: // HLT - ud2(); - break; - default: LogMan::Msg::A("Unknown Break reason: %d", Op->Reason); - } - } - break; - case IR::OP_JUMP: { - auto Op = IROp->C(); - - Label *TargetLabel; - auto IsTarget = JumpTargets.find(Op->Header.Args[0].ID()); - if (IsTarget == JumpTargets.end()) { - TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[0].ID(), Label{}).first->second; - } - else { - TargetLabel = &IsTarget->second; - } - - jmp(*TargetLabel); - break; - } - default: break; - } - - if (HasRA) { - switch (IROp->Op) { - case IR::OP_CONDJUMP: { - auto Op = IROp->C(); - - Label *TargetLabel; - auto IsTarget = JumpTargets.find(Op->Header.Args[1].ID()); - if (IsTarget == JumpTargets.end()) { - TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[1].ID(), Label{}).first->second; - } - else { - TargetLabel = &IsTarget->second; - } - - cmp(GetSrc(Op->Header.Args[0].ID()), 0); - jne(*TargetLabel); - break; - } - case IR::OP_LOADCONTEXT: { - auto Op = IROp->C(); - switch (Op->Size) { - case 1: { - mov(GetDst(Node), byte [STATE + Op->Offset]); - } - break; - case 2: { - mov(GetDst(Node), word [STATE + Op->Offset]); - } - break; - case 4: { - mov(GetDst(Node), dword [STATE + Op->Offset]); - } - break; - case 8: { - mov(GetDst(Node), qword [STATE + Op->Offset]); - } - break; - case 16: { - if (Op->Offset % 16 == 0) - movaps(GetDst(Node), xword [STATE + Op->Offset]); - else - movups(GetDst(Node), xword [STATE + Op->Offset]); - } - break; - default: LogMan::Msg::A("Unhandled LoadContext size: %d", Op->Size); - } - break; - } - case IR::OP_STORECONTEXT: { - auto Op = IROp->C(); - - switch (Op->Size) { - case 1: { - mov(byte [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); - } - break; - - case 2: { - mov(word [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); - } - break; - case 4: { - mov(dword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); - } - break; - case 8: { - mov(qword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); - } - break; - case 16: { - if (Op->Offset % 16 == 0) - movaps(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); - else - movups(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); - } - break; - default: LogMan::Msg::A("Unhandled StoreContext size: %d", Op->Size); - } - break; - } - case IR::OP_ADD: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - mov(rax, GetSrc(Op->Header.Args[1].ID())); - add(rax, GetSrc(Op->Header.Args[0].ID())); - mov(Dst, rax); - break; - } - case IR::OP_SUB: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - mov(rax, GetSrc(Op->Header.Args[0].ID())); - sub(rax, GetSrc(Op->Header.Args[1].ID())); - mov(Dst, rax); - break; - } - case IR::OP_XOR: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - mov(rax, GetSrc(Op->Header.Args[1].ID())); - xor(rax, GetSrc(Op->Header.Args[0].ID())); - mov(Dst, rax); - break; - } - case IR::OP_AND: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - mov(rax, GetSrc(Op->Header.Args[1].ID())); - and(rax, GetSrc(Op->Header.Args[0].ID())); - mov(Dst, rax); - break; - } - case IR::OP_OR: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - mov(rax, GetSrc(Op->Header.Args[1].ID())); - or (rax, GetSrc(Op->Header.Args[0].ID())); - mov(Dst, rax); - break; - } - case IR::OP_MOV: { - auto Op = IROp->C(); - mov (GetDst(Node), GetSrc(Op->Header.Args[0].ID())); - break; - } - case IR::OP_CONSTANT: { - auto Op = IROp->C(); - mov(GetDst(Node), Op->Constant); - break; - } - case IR::OP_POPCOUNT: { - auto Op = IROp->C(); - auto Dst64 = GetDst(Node); - - switch (OpSize) { - case 1: - movzx(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); - popcnt(Dst64, Dst64); - break; - case 2: { - movzx(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); - popcnt(Dst64, Dst64); - break; - } - case 4: - popcnt(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); - break; - case 8: - popcnt(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); - break; - } - break; - } - case IR::OP_ZEXT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - - uint32_t PhysReg = RA::GetNodeRegister(Graph, Op->Header.Args[0].ID()); - if (PhysReg >= XMMBase) { - // XMM -> GPR transfer with free truncation - switch (Op->SrcSize) { - case 8: - pextrb(al, GetSrc(Op->Header.Args[0].ID()), 0); - break; - case 16: - pextrw(ax, GetSrc(Op->Header.Args[0].ID()), 0); - break; - case 32: - pextrd(eax, GetSrc(Op->Header.Args[0].ID()), 0); - break; - case 64: - pextrw(rax, GetSrc(Op->Header.Args[0].ID()), 0); - break; - default: LogMan::Msg::A("Unhandled Zext size: %d", Op->SrcSize); break; - } - auto Dst = GetDst(Node); - mov(Dst, rax); - } - else { - if (Op->SrcSize == 64) { - vmovq(xmm15, Reg64(GetSrc(Op->Header.Args[0].ID()).getIdx())); - movapd(GetDst(Node), xmm15); - } - else { - auto Dst = GetDst(Node); - mov(rax, uint64_t((1ULL << Op->SrcSize) - 1)); - and(rax, GetSrc(Op->Header.Args[0].ID())); - mov(Dst, rax); - } - } - break; - } - case IR::OP_SEXT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - auto Dst = GetDst(Node); - - switch (Op->SrcSize / 8) { - case 1: - movsx(Dst, GetSrc(Op->Header.Args[0].ID())); - break; - case 2: - movsx(Dst, GetSrc(Op->Header.Args[0].ID())); - break; - case 4: - movsxd(Reg64(Dst.getIdx()), GetSrc(Op->Header.Args[0].ID())); - break; - case 8: - mov(Dst, GetSrc(Op->Header.Args[0].ID())); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", Op->SrcSize / 8); - } - break; - } - case IR::OP_BFE: { - auto Op = IROp->C(); - LogMan::Throw::A(OpSize <= 16, "OpSize is too large for BFE: %d", OpSize); - if (OpSize == 16) { - LogMan::Throw::A(!(Op->lsb < 64 && (Op->lsb + Op->Width > 64)), "Trying to BFE an XMM across the 64bit split: Beginning at %d, ending at %d", Op->lsb, Op->lsb + Op->Width); - movups(xmm15, GetSrc(Op->Header.Args[0].ID())); - uint8_t Offset = Op->lsb; - if (Offset < 64) { - pextrq(rax, xmm15, 0); - } - else { - pextrq(rax, xmm15, 1); - Offset -= 64; - } - - if (Offset) { - shr(rax, Offset); - } - - if (Op->Width != 64) { - mov(rcx, uint64_t((1ULL << Op->Width) - 1)); - and(rax, rcx); - } - - mov (GetDst(Node), rax); - } - else { - auto Dst = GetDst(Node); - mov(rax, GetSrc(Op->Header.Args[0].ID())); - - if (Op->lsb != 0) - shr(rax, Op->lsb); - - if (Op->Width != 64) { - mov(rcx, uint64_t((1ULL << Op->Width) - 1)); - and(rax, rcx); - } - mov(Dst, rax); - } - break; - } - case IR::OP_LSHR: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - auto Dst = GetDst(Node); - mov (rcx, GetSrc(Op->Header.Args[1].ID())); - and(rcx, Mask); - - shrx(Reg32e(Dst.getIdx(), 64), GetSrc(Op->Header.Args[0].ID()), rcx); - break; - } - case IR::OP_LSHL: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - auto Dst = GetDst(Node); - mov (rcx, GetSrc(Op->Header.Args[1].ID())); - and(rcx, Mask); - - shlx(Reg32e(Dst.getIdx(), 64), GetSrc(Op->Header.Args[0].ID()), rcx); - break; - } - case IR::OP_ASHR: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov (rcx, GetSrc(Op->Header.Args[1].ID())); - and(rcx, Mask); - switch (OpSize) { - case 1: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - sar(al, cl); - movsx(GetDst(Node), al); - break; - case 2: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - sar(ax, cl); - movsx(GetDst(Node), ax); - break; - case 4: - sarx(Reg32e(GetDst(Node).getIdx(), 32), GetSrc(Op->Header.Args[0].ID()), ecx); - break; - case 8: - sarx(Reg32e(GetDst(Node).getIdx(), 64), GetSrc(Op->Header.Args[0].ID()), rcx); - break; - default: LogMan::Msg::A("Unknown ASHR Size: %d\n", OpSize); break; - }; - break; - } - case IR::OP_ROL: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov (rcx, GetSrc(Op->Header.Args[1].ID())); - and(rcx, Mask); - switch (OpSize) { - case 1: { - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - rol(al, cl); - break; - } - case 2: { - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - rol(ax, cl); - break; - } - case 4: { - mov(eax, GetSrc(Op->Header.Args[0].ID())); - rol(eax, cl); - break; - } - case 8: { - mov(rax, GetSrc(Op->Header.Args[0].ID())); - rol(rax, cl); - break; - } - } - mov(GetDst(Node), rax); - break; - } - case IR::OP_ROR: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov (rcx, GetSrc(Op->Header.Args[1].ID())); - and(rcx, Mask); - switch (OpSize) { - case 1: { - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - ror(al, cl); - break; - } - case 2: { - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - ror(ax, cl); - break; - } - case 4: { - mov(eax, GetSrc(Op->Header.Args[0].ID())); - ror(eax, cl); - break; - } - case 8: { - mov(rax, GetSrc(Op->Header.Args[0].ID())); - ror(rax, cl); - break; - } - } - mov(GetDst(Node), rax); - break; - } - case IR::OP_MUL: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - - switch (OpSize) { - case 1: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - movsx(rcx, GetSrc(Op->Header.Args[1].ID())); - imul(cl); - movsx(Dst, al); - break; - case 2: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - movsx(rcx, GetSrc(Op->Header.Args[1].ID())); - imul(cx); - movsx(Dst, ax); - break; - case 4: - movsxd(rax, GetSrc(Op->Header.Args[0].ID())); - imul(eax, GetSrc(Op->Header.Args[1].ID())); - movsx(Dst, eax); - break; - case 8: - mov(rax, GetSrc(Op->Header.Args[0].ID())); - imul(rax, GetSrc(Op->Header.Args[1].ID())); - mov(Dst, rax); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_MULH: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - movsx(rcx, GetSrc(Op->Header.Args[1].ID())); - imul(cl); - movsx(rax, ax); - mov(GetDst(Node), rax); - break; - case 2: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - movsx(rcx, GetSrc(Op->Header.Args[1].ID())); - imul(cx); - movsx(rax, dx); - mov(GetDst(Node), rax); - break; - case 4: - movsx(rax, GetSrc(Op->Header.Args[0].ID())); - imul(GetSrc(Op->Header.Args[1].ID())); - movsxd(rax, edx); - mov(GetDst(Node), rdx); - break; - case 8: - mov(rax, GetSrc(Op->Header.Args[0].ID())); - imul(GetSrc(Op->Header.Args[1].ID())); - mov(GetDst(Node), rdx); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_UMUL: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - movzx(rcx, GetSrc(Op->Header.Args[1].ID())); - mul(cl); - movzx(rax, al); - mov(GetDst(Node), rax); - break; - case 2: - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - movzx(rcx, GetSrc(Op->Header.Args[1].ID())); - mul(cx); - movzx(rax, ax); - mov(GetDst(Node), rax); - break; - case 4: - mov(rax, GetSrc(Op->Header.Args[0].ID())); - mul(GetSrc(Op->Header.Args[1].ID())); - mov(GetDst(Node), rax); - break; - case 8: - mov(rax, GetSrc(Op->Header.Args[0].ID())); - mul(GetSrc(Op->Header.Args[1].ID())); - mov(GetDst(Node), rax); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_UMULH: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - movzx(rcx, GetSrc(Op->Header.Args[1].ID())); - mul(cl); - movzx(rax, ax); - mov(GetDst(Node), rax); - break; - case 2: - movzx(rax, GetSrc(Op->Header.Args[0].ID())); - movzx(rcx, GetSrc(Op->Header.Args[1].ID())); - mul(cx); - movzx(rax, dx); - mov(GetDst(Node), rax); - break; - case 4: - mov(rax, GetSrc(Op->Header.Args[0].ID())); - mul(GetSrc(Op->Header.Args[1].ID())); - mov(GetDst(Node), rdx); - break; - case 8: - mov(rax, GetSrc(Op->Header.Args[0].ID())); - mul(GetSrc(Op->Header.Args[1].ID())); - mov(GetDst(Node), rdx); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_LDIV: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov(eax, GetSrc(Op->Header.Args[0].ID())); - mov(edx, GetSrc(Op->Header.Args[1].ID())); - mov(ecx, GetSrc(Op->Header.Args[2].ID())); - idiv(ecx); - mov(GetDst(Node), rax); - break; - } - case 8: { - mov(rax, GetSrc(Op->Header.Args[0].ID())); - mov(rdx, GetSrc(Op->Header.Args[1].ID())); - mov(rcx, GetSrc(Op->Header.Args[2].ID())); - idiv(rcx); - mov(GetDst(Node), rax); - break; - } - default: LogMan::Msg::A("Unknown LDIV Size: %d", Size); break; - } - break; - } - case IR::OP_LREM: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov(eax, GetSrc(Op->Header.Args[0].ID())); - mov(edx, GetSrc(Op->Header.Args[1].ID())); - mov(ecx, GetSrc(Op->Header.Args[2].ID())); - idiv(ecx); - mov(GetDst(Node), rdx); - break; - } - - case 8: { - mov(rax, GetSrc(Op->Header.Args[0].ID())); - mov(rdx, GetSrc(Op->Header.Args[1].ID())); - mov(rcx, GetSrc(Op->Header.Args[2].ID())); - idiv(rcx); - mov(GetDst(Node), rdx); - break; - } - default: LogMan::Msg::A("Unknown LREM Size: %d", Size); break; - } - break; - } - case IR::OP_LUDIV: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov (eax, GetSrc(Op->Header.Args[0].ID())); - mov (edx, GetSrc(Op->Header.Args[1].ID())); - mov (ecx, GetSrc(Op->Header.Args[2].ID())); - div(ecx); - mov(GetDst(Node), rax); - break; - } - case 8: { - mov (rax, GetSrc(Op->Header.Args[0].ID())); - mov (rdx, GetSrc(Op->Header.Args[1].ID())); - mov (rcx, GetSrc(Op->Header.Args[2].ID())); - div(rcx); - mov(GetDst(Node), rax); - break; - } - default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; - } - break; - } - case IR::OP_LUREM: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov (eax, GetSrc(Op->Header.Args[0].ID())); - mov (edx, GetSrc(Op->Header.Args[1].ID())); - mov (ecx, GetSrc(Op->Header.Args[2].ID())); - div(ecx); - mov(GetDst(Node), rdx); - break; - } - - case 8: { - mov (rax, GetSrc(Op->Header.Args[0].ID())); - mov (rdx, GetSrc(Op->Header.Args[1].ID())); - mov (rcx, GetSrc(Op->Header.Args[2].ID())); - div(rcx); - mov(GetDst(Node), rdx); - break; - } - default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; - } - break; - } - case IR::OP_LOADFLAG: { - auto Op = IROp->C(); - - auto Dst = GetDst(Node); - movzx(Dst, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]); - and(Dst, 1); - break; - } - case IR::OP_STOREFLAG: { - auto Op = IROp->C(); - - mov (rax, GetSrc(Op->Header.Args[0].ID())); - and(rax, 1); - mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al); - break; - } - case IR::OP_SELECT: { - auto Op = IROp->C(); - auto Dst = GetDst(Node); - - mov(rax, GetSrc(Op->Header.Args[0].ID())); - cmp(rax, GetSrc(Op->Header.Args[1].ID())); - - switch (Op->Cond) { - case FEXCore::IR::COND_EQ: - mov(rax, GetSrc(Op->Header.Args[3].ID())); - cmove(rax, GetSrc(Op->Header.Args[2].ID())); - break; - case FEXCore::IR::COND_NEQ: - mov(rax, GetSrc(Op->Header.Args[3].ID())); - cmovne(rax, GetSrc(Op->Header.Args[2].ID())); - break; - case FEXCore::IR::COND_GE: - mov(rax, GetSrc(Op->Header.Args[3].ID())); - cmovge(rax, GetSrc(Op->Header.Args[2].ID())); - break; - case FEXCore::IR::COND_LT: - mov(rax, GetSrc(Op->Header.Args[2].ID())); - cmovae(rax, GetSrc(Op->Header.Args[3].ID())); - break; - case FEXCore::IR::COND_GT: - mov(rax, GetSrc(Op->Header.Args[3].ID())); - cmovg(rax, GetSrc(Op->Header.Args[2].ID())); - break; - case FEXCore::IR::COND_LE: - mov(rax, GetSrc(Op->Header.Args[3].ID())); - cmovle(rax, GetSrc(Op->Header.Args[2].ID())); - break; - case FEXCore::IR::COND_CS: - case FEXCore::IR::COND_CC: - case FEXCore::IR::COND_MI: - case FEXCore::IR::COND_PL: - case FEXCore::IR::COND_VS: - case FEXCore::IR::COND_VC: - case FEXCore::IR::COND_HI: - case FEXCore::IR::COND_LS: - default: - LogMan::Msg::A("Unsupported compare type"); - break; - } - mov (Dst, rax); - break; - } - case IR::OP_LOADMEM: { - auto Op = IROp->C(); - uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); - - auto Dst = GetDst(Node); - mov(rax, Memory); - add(rax, GetSrc(Op->Header.Args[0].ID())); - switch (Op->Size) { - case 1: { - movzx (Dst, byte [rax]); - } - break; - case 2: { - movzx (Dst, word [rax]); - } - break; - case 4: { - mov(Dst, dword [rax]); - } - break; - case 8: { - mov(Dst, qword [rax]); - } - break; - case 16: { - movups(GetDst(Node), xword [rax]); - if (MemoryDebug) { - movq(rcx, GetDst(Node)); - } - } - break; - default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size); - } - break; - } - case IR::OP_STOREMEM: { - auto Op = IROp->C(); - uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); - - mov(rax, Memory); - add(rax, GetSrc(Op->Header.Args[0].ID())); - switch (Op->Size) { - case 1: - mov(byte [rax], GetSrc(Op->Header.Args[1].ID())); - break; - case 2: - mov(word [rax], GetSrc(Op->Header.Args[1].ID())); - break; - case 4: - mov(dword [rax], GetSrc(Op->Header.Args[1].ID())); - break; - case 8: - mov(qword [rax], GetSrc(Op->Header.Args[1].ID())); - break; - case 16: - movups(xword [rax], GetSrc(Op->Header.Args[1].ID())); - break; - default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size); - } - break; - } - case IR::OP_SYSCALL: { - auto Op = IROp->C(); - // XXX: This is very terrible, but I don't care for right now - - push(rdi); - const std::array RA64 = { rsi, r8, r9, r10, r11, rbx, rbp, r12, r13, r14, r15 }; - for (auto &Reg : RA64) - push(Reg); - - // Syscall ABI for x86-64 - // this: rdi - // Thread: rsi - // ArgPointer: rdx (Stack) - // - // Result: RAX - - // These are pushed in reverse order because stacks - for (uint32_t i = 7; i > 0; --i) - push(GetSrc(Op->Header.Args[i - 1].ID())); - - mov(rsi, rdi); // Move thread in to rsi - mov(rdi, reinterpret_cast(&CTX->SyscallHandler)); - mov(rdx, rsp); - - using PtrType = uint64_t (FEXCore::SyscallHandler::*)(FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args); - union { - PtrType ptr; - uint64_t Raw; - } PtrCast; - PtrCast.ptr = &FEXCore::SyscallHandler::HandleSyscall; - mov(rax, PtrCast.Raw); - call(rax); - - // Reload arguments just in case they are sill live after the fact - for (uint32_t i = 0; i < 7; ++i) - pop(GetSrc(Op->Header.Args[i].ID())); - - for (uint32_t i = RA64.size(); i > 0; --i) - pop(RA64[i - 1]); - - pop(rdi); - - mov (GetDst(Node), rax); - break; - } - case IR::OP_CPUID: { - auto Op = IROp->C(); - using ClassPtrType = FEXCore::CPUIDEmu::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t Function); - union { - ClassPtrType ClassPtr; - uint64_t Raw; - } Ptr; - Ptr.ClassPtr = &CPUIDEmu::RunFunction; - - const std::array RA64 = { rsi, r8, r9, r10, r11, rbx, rbp, r12, r13, r14, r15 }; - for (auto &Reg : RA64) - push(Reg); - - // CPUID ABI - // this: rdi - // Function: rsi - // - // Result: RAX, RDX. 4xi32 - push(rdi); - mov (rsi, GetSrc(Op->Header.Args[0].ID())); - mov (rdi, reinterpret_cast(&CTX->CPUID)); - - sub(rsp, 8); // Align - - mov(rax, Ptr.Raw); - call(rax); - - add(rsp, 8); // Align - - pop(rdi); - - for (uint32_t i = RA64.size(); i > 0; --i) - pop(RA64[i - 1]); - - auto Dst = GetDst(Node); - pinsrq(Dst, rax, 0); - pinsrd(Dst, rdx, 1); - break; - } - case IR::OP_EXTRACTELEMENT: { - auto Op = IROp->C(); - - uint32_t PhysReg = RA::GetNodeRegister(Graph, Op->Header.Args[0].ID()); - if (PhysReg >= XMMBase) { - switch (OpSize) { - case 1: - pextrb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); - break; - case 2: - pextrw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); - break; - case 4: - pextrd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); - break; - case 8: - pextrq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); - break; - default: LogMan::Msg::A("Unhandled ExtractElementSize: %d", OpSize); - } - } - else { - LogMan::Msg::A("Can't handle extract from GPR yet"); - } - break; - } - case IR::OP_VINSELEMENT: { - auto Op = IROp->C(); - movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); - - // Dst_d[Op->DestIdx] = Src2_d[Op->SrcIdx]; - - // pextrq reg64/mem64, xmm, imm - // pinsrq xmm, reg64/mem64, imm8 - switch (Op->ElementSize) { - case 1: { - pextrb(al, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); - pinsrb(xmm15, al, Op->DestIdx); - break; - } - case 2: { - pextrw(ax, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); - pinsrw(xmm15, ax, Op->DestIdx); - break; - } - case 4: { - pextrd(eax, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); - pinsrd(xmm15, eax, Op->DestIdx); - break; - } - case 8: { - pextrq(rax, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); - pinsrq(xmm15, rax, Op->DestIdx); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - - movapd(GetDst(Node), xmm15); - break; - } - case IR::OP_VADD: { - auto Op = IROp->C(); - switch (Op->ElementSize) { - case 1: { - vpaddb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case 2: { - vpaddw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case 4: { - vpaddd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case 8: { - vpaddq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - break; - } - case IR::OP_VSUB: { - auto Op = IROp->C(); - switch (Op->ElementSize) { - case 1: { - vpsubb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case 2: { - vpsubw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case 4: { - vpsubd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case 8: { - vpsubq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - break; - } - case IR::OP_VXOR: { - auto Op = IROp->C(); - vpxor(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case IR::OP_VOR: { - auto Op = IROp->C(); - vpor(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - } - case IR::OP_VCMPEQ: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); - - switch (Op->ElementSize) { - case 1: - vpcmpeqb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - case 2: - vpcmpeqw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - case 4: - vpcmpeqd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - case 8: - vpcmpeqq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); - } - break; - } - case IR::OP_VCMPGT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); - - switch (Op->ElementSize) { - case 1: - vpcmpgtb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - case 2: - vpcmpgtw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - case 4: - vpcmpgtd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - case 8: - vpcmpgtq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); - break; - default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); - } - break; - } - case IR::OP_VZIP: { - auto Op = IROp->C(); - movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); - - switch (Op->ElementSize) { - case 1: { - punpcklbw(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - case 2: { - punpcklwd(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - case 4: { - punpckldq(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - case 8: { - punpcklqdq(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movapd(GetDst(Node), xmm15); - break; - } - case IR::OP_VZIP2: { - auto Op = IROp->C(); - movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); - - switch (Op->ElementSize) { - case 1: { - punpckhbw(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - case 2: { - punpckhwd(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - case 4: { - punpckhdq(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - case 8: { - punpckhqdq(xmm15, GetSrc(Op->Header.Args[1].ID())); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movapd(GetDst(Node), xmm15); - break; - } - case IR::OP_VUSHLS: { - auto Op = IROp->C(); - movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); - vmovq(xmm14, Reg64(GetSrc(Op->Header.Args[1].ID()).getIdx())); - - switch (Op->ElementSize) { - case 2: { - psllw(xmm15, xmm14); - break; - } - case 4: { - pslld(xmm15, xmm14); - break; - } - case 8: { - psllq(xmm15, xmm14); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movapd(GetDst(Node), xmm15); - - break; - } - case IR::OP_CAS: { - auto Op = IROp->C(); - // Args[0]: Desired - // Args[1]: Expected - // Args[2]: Pointer - // DataSrc = *Src1 - // if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc - // This will write to memory! Careful! - // Third operand must be a calculated guest memory address - //OrderedNode *CASResult = _CAS(Src3, Src2, Src1); - uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); - - mov(rcx, Memory); - add(rcx, GetSrc(Op->Header.Args[2].ID())); - mov(rdx, GetSrc(Op->Header.Args[1].ID())); - mov(rax, GetSrc(Op->Header.Args[0].ID())); - - // RCX now contains pointer - // RAX contains our expected value - // RDX contains our desired - - lock(); - - switch (OpSize) { - case 1: { - cmpxchg(byte [rcx], dl); - movzx(rax, al); - break; - } - case 2: { - cmpxchg(word [rcx], dx); - movzx(rax, ax); - break; - } - case 4: { - cmpxchg(dword [rcx], edx); - break; - } - case 8: { - cmpxchg(qword [rcx], rdx); - break; - } - default: LogMan::Msg::A("Unsupported: %d", OpSize); - } - - // RAX now contains the result - mov (GetDst(Node), rax); - break; - } - case IR::OP_CYCLECOUNTER: { -#ifdef DEBUG_CYCLES - mov (GetDst(Node), 0); -#else - rdtsc(); - shl(rdx, 32); - or(rax, rdx); - mov (GetDst(Node), rax); -#endif - break; - } - case IR::OP_FINDLSB: { - auto Op = IROp->C(); - tzcnt(rcx, GetSrc(Op->Header.Args[0].ID())); - xor(rax, rax); - cmp(GetSrc(Op->Header.Args[0].ID()), 1); - sbb(rax, rax); - or(rax, rcx); - mov (GetDst(Node), rax); - break; - } - case IR::OP_FINDMSB: { - auto Op = IROp->C(); - mov(rax, OpSize * 8); - lzcnt(rcx, GetSrc(Op->Header.Args[0].ID())); - sub(rax, rcx); - mov (GetDst(Node), rax); - break; - } - default: break; - } - } - else { - switch (IROp->Op) { - case IR::OP_LOADCONTEXT: { - auto Op = IROp->C(); -#define LOAD_CTX(x, y) \ - case x: { \ - movzx(rax, y [STATE + Op->Offset]); \ - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); \ - } \ - break - switch (Op->Size) { - LOAD_CTX(1, byte); - LOAD_CTX(2, word); - case 4: { - mov(eax, dword [STATE + Op->Offset]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - } - break; - case 8: { - mov(rax, qword [STATE + Op->Offset]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - } - break; - case 16: { - if (Op->Offset % 16 == 0) { - movaps(xmm0, xword [STATE + Op->Offset]); - movaps(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - } - else { - movups(xmm0, xword [STATE + Op->Offset]); - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - } - } - break; - default: LogMan::Msg::A("Unhandled LoadContext size: %d", Op->Size); - } -#undef LOAD_CTX - break; - } - case IR::OP_STORECONTEXT: { - auto Op = IROp->C(); - - switch (Op->Size) { - case 1: { - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov(byte [STATE + Op->Offset], al); - } - break; - - case 2: { - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov(word [STATE + Op->Offset], ax); - } - break; - case 4: { - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov(dword [STATE + Op->Offset], eax); - } - break; - case 8: { - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov(qword [STATE + Op->Offset], rax); - } - break; - case 16: { - if (Op->Offset % 16 == 0) { - movaps(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movaps(xword [STATE + Op->Offset], xmm0); - } - else { - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xword [STATE + Op->Offset], xmm0); - } - } - break; - default: LogMan::Msg::A("Unhandled StoreContext size: %d", Op->Size); - } - - break; - } - - case IR::OP_SYSCALL: { - auto Op = IROp->C(); - - push(rdi); - push(r11); - - // Syscall ABI for x86-64 - // this: rdi - // Thread: rsi - // ArgPointer: rdx (Stack) - // - // Result: RAX - - mov(rsi, rdi); // Move thread in to rsi - mov(rdi, reinterpret_cast(&CTX->SyscallHandler)); - - // These are pushed in reverse order because stacks - push(qword [TEMP_STACK + (Op->Header.Args[6].ID() * 16)]); - push(qword [TEMP_STACK + (Op->Header.Args[5].ID() * 16)]); - push(qword [TEMP_STACK + (Op->Header.Args[4].ID() * 16)]); - push(qword [TEMP_STACK + (Op->Header.Args[3].ID() * 16)]); - push(qword [TEMP_STACK + (Op->Header.Args[2].ID() * 16)]); - push(qword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - push(qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov (rdx, rsp); - - using PtrType = uint64_t (FEXCore::SyscallHandler::*)(FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args); - union { - PtrType ptr; - uint64_t Raw; - } PtrCast; - PtrCast.ptr = &FEXCore::SyscallHandler::HandleSyscall; - mov(rax, PtrCast.Raw); - call(rax); - add(rsp, 7 * 8); - - pop(r11); - pop(rdi); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_CPUID: { - auto Op = IROp->C(); - using ClassPtrType = FEXCore::CPUIDEmu::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t Function); - union { - ClassPtrType ClassPtr; - uint64_t Raw; - } Ptr; - Ptr.ClassPtr = &CPUIDEmu::RunFunction; - - - // CPUID ABI - // this: rdi - // Function: rsi - // - // Result: RAX, RDX. 4xi32 - push(rdi); - push(r11); - mov (rsi, qword [TEMP_STACK + (Op->Header.Args[0].ID() *16)]); - mov (rdi, reinterpret_cast(&CTX->CPUID)); - - push(rax); // align - - mov(rax, Ptr.Raw); - call(rax); - - pop(r11); // align - - pop(r11); - pop(rdi); - - mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 0], eax); - shr(rax, 32); - mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 4], eax); - - mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 8], edx); - shr(rdx, 32); - mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 12], edx); - break; - } - case IR::OP_EXTRACTELEMENT: { - auto Op = IROp->C(); - - uint32_t Offset = Op->Header.Args[0].ID() * 16 + OpSize * Op->Idx; - switch (OpSize) { - case 1: - movzx(rax, byte [TEMP_STACK + Offset]); - break; - case 2: - movzx(rax, word [TEMP_STACK + Offset]); - break; - case 4: - mov(eax, dword [TEMP_STACK + Offset]); - break; - case 8: - mov(rax, qword [TEMP_STACK + Offset]); - break; - default: LogMan::Msg::A("Unhandled ExtractElementSize: %d", OpSize); - } - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_LOADFLAG: { - auto Op = IROp->C(); - - movzx(rax, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]); - and(rax, 1); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_STOREFLAG: { - auto Op = IROp->C(); - - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - and(rax, 1); - mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al); - break; - } - - case IR::OP_CONDJUMP: { - auto Op = IROp->C(); - - Label *TargetLabel; - auto IsTarget = JumpTargets.find(Op->Header.Args[1].ID()); - if (IsTarget == JumpTargets.end()) { - TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[1].ID(), Label{}).first->second; - } - else { - TargetLabel = &IsTarget->second; - } - - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - cmp(rax, 0); - jne(*TargetLabel); - break; - } - - case IR::OP_LOADMEM: { - auto Op = IROp->C(); - uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); - - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov(rcx, Memory); - add(rax, rcx); - switch (Op->Size) { - case 1: { - movzx (rcx, byte [rax]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); - } - break; - case 2: { - movzx (rcx, word [rax]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); - } - break; - case 4: { - mov(ecx, dword [rax]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); - } - break; - case 8: { - mov(rcx, qword [rax]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); - } - break; - case 16: { - movups(xmm0, xword [rax]); - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - if (MemoryDebug) { - movq(rcx, xmm0); - } - } - break; - default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size); - } - - if (MemoryDebug) { - push(rdi); - push(r11); - sub(rsp, 8); - - // Load the address in to Arg1 - mov(rdi, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - // Move the loaded value to Arg2 - mov(rsi, rcx); - mov (rdx, Op->Size); - - mov(rax, reinterpret_cast(LoadMem)); - call(rax); - - add(rsp, 8); - - pop(r11); - pop(rdi); - } - - break; - } - case IR::OP_STOREMEM: { - auto Op = IROp->C(); - uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); - - mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - mov(rcx, Memory); - add(rax, rcx); - switch (Op->Size) { - case 1: { - mov(cl, byte [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - mov(byte [rax], cl); - if (MemoryDebug) { - movzx(rcx, cl); - } - } - break; - case 2: { - mov(cx, word [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - mov(word [rax], cx); - - if (MemoryDebug) { - movzx(rcx, cx); - } - } - break; - case 4: { - mov(ecx, dword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - mov(dword [rax], ecx); - } - break; - case 8: { - mov(rcx, qword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - mov(qword [rax], rcx); - } - break; - case 16: { - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - movups(xword [rax], xmm0); - if (MemoryDebug) { - movq(rcx, xmm0); - } - } - break; - default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size); - } - - if (MemoryDebug) { - push(rdi); - push(r11); - sub(rsp, 8); - - // Load the address in to Arg1 - mov(rdi, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - // Load the value from RAX in to Arg2 - mov(rsi, rcx); - - mov (rdx, Op->Size); - - mov(rax, reinterpret_cast(StoreMem)); - call(rax); - - add(rsp, 8); - - pop(r11); - pop(rdi); - } - break; - } - case IR::OP_MOV: { - auto Op = IROp->C(); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_CONSTANT: { - auto Op = IROp->C(); - if (Op->Constant >> 31) { - mov(rax, Op->Constant); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - } - else { - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], Op->Constant); - } - break; - } - case IR::OP_POPCOUNT: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movzx(al, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - popcnt(eax, eax); - break; - case 2: - popcnt(ax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movzx(rax, ax); - break; - case 4: - popcnt(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - break; - case 8: - popcnt(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - break; - } - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_ADD: { - auto Op = IROp->C(); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - add(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_SUB: { - auto Op = IROp->C(); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - sub(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_XOR: { - auto Op = IROp->C(); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - xor(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_AND: { - auto Op = IROp->C(); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - and(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_OR: { - auto Op = IROp->C(); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - or(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_MUL: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movsx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - imul(cl); - movsx(rax, al); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 2: - movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movsx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - imul(cx); - movsx(rax, ax); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 4: - movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - imul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - movsxd(rax, eax); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 8: - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - imul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_UMUL: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movzx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mul(cl); - movzx(rax, al); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 2: - movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movzx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mul(cx); - movzx(rax, ax); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 4: - mov(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 8: - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - - case IR::OP_MULH: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movsx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - imul(cl); - movsx(rax, ax); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 2: - movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movsx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - imul(cx); - movsx(rax, dx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 4: - movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - imul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - movsxd(rax, edx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 8: - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - imul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_UMULH: { - auto Op = IROp->C(); - switch (OpSize) { - case 1: - movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movzx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mul(cl); - movzx(rax, ax); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 2: - movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movzx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mul(cx); - movzx(rax, dx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 4: - mov(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - case 8: - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); - } - break; - } - case IR::OP_LDIV: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - idiv(ecx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - - case 8: { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - idiv(rcx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - default: LogMan::Msg::A("Unknown LDIV Size: %d", Size); break; - } - break; - } - case IR::OP_LREM: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - idiv(ecx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - } - - case 8: { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - idiv(rcx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - } - default: LogMan::Msg::A("Unknown LREM Size: %d", Size); break; - } - break; - } - - - case IR::OP_LUDIV: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - div(ecx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - - case 8: { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - div(rcx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; - } - break; - } - case IR::OP_LUREM: { - auto Op = IROp->C(); - // Each source is OpSize in size - // So you can have up to a 128bit divide from x86-64 - auto Size = OpSize; - switch (Size) { - case 4: { - mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - div(ecx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - } - - case 8: { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - div(rcx); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); - break; - } - default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; - } - break; - } - - case IR::OP_ZEXT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - - if (Op->SrcSize == 64) { - movd(xmm0, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - movups(qword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - } - else { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(rcx, uint64_t((1ULL << Op->SrcSize) - 1)); - and(rax, rcx); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - } - break; - } - - case IR::OP_SEXT: { - auto Op = IROp->C(); - LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); - switch (Op->SrcSize / 8) { - case 1: - movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 2: - movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 4: - movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - case 8: - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - default: LogMan::Msg::A("Unknown Sext size: %d", Op->SrcSize / 8); - } - break; - } - case IR::OP_BFI: { - auto Op = IROp->C(); - LogMan::Throw::A(OpSize <= 8, "OpSize is too large for BFI: %d", OpSize); - - uint64_t SourceMask = (1ULL << Op->Width) - 1; - - uint64_t DestMask = ~(SourceMask << Op->lsb); - - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - - if (Op->Width != 64) { - mov(rdx, SourceMask); - and(rcx, rdx); - } - - mov(rdx, DestMask); - and(rax, rdx); - shl(rdx, Op->lsb); - or(rax, rdx); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - - break; - } - - case IR::OP_BFE: { - auto Op = IROp->C(); - LogMan::Throw::A(OpSize <= 16, "OpSize is too large for BFE: %d", OpSize); - // %ssa64 i128 = Bfe %ssa48 i128, 0x1, 0x7 - if (OpSize == 16) { - LogMan::Throw::A(!(Op->lsb < 64 && (Op->lsb + Op->Width > 64)), "Trying to BFE an XMM across the 64bit split: Beginning at %d, ending at %d", Op->lsb, Op->lsb + Op->Width); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - uint8_t Offset = Op->lsb; - if (Offset < 64) { - pextrq(rax, xmm0, 0); - } - else { - pextrq(rax, xmm0, 1); - Offset -= 64; - } - - if (Offset) { - shr(rax, Offset); - } - - if (Op->Width != 64) { - mov(rcx, uint64_t((1ULL << Op->Width) - 1)); - and(rax, rcx); - } - - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - } - else { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - if (Op->lsb != 0) - shr(rax, Op->lsb); - - if (Op->Width != 64) { - mov(rcx, uint64_t((1ULL << Op->Width) - 1)); - and(rax, rcx); - } - - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - } - break; - } - case IR::OP_FINDLSB: { - auto Op = IROp->C(); - tzcnt(rcx, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - xor(rax, rax); - cmp(qword [TEMP_STACK + Op->Header.Args[0].ID() * 16], 1); - sbb(rax, rax); - or(rax, rcx); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - - break; - } - case IR::OP_FINDMSB: { - auto Op = IROp->C(); - mov(rax, OpSize * 8); - lzcnt(rcx, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - sub(rax, rcx); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - - case IR::OP_LSHR: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - and(rcx, Mask); - shrx(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16], rcx); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_LSHL: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - and(rcx, Mask); - shlx(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16], rcx); - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_ASHR: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - and(rcx, Mask); - switch (OpSize) { - case 1: - movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - sar(al, cl); - break; - case 2: - movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - sar(ax, cl); - break; - case 4: - movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - sar(eax, cl); - break; - case 8: - mov(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - sar(rax, cl); - break; - default: LogMan::Msg::A("Unknown ASHR Size: %d\n", OpSize); break; - }; - - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - - case IR::OP_ROL: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - and(rcx, Mask); - switch (OpSize) { - case 1: { - movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - rol(al, cl); - break; - } - case 2: { - movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - rol(ax, cl); - break; - } - case 4: { - mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - rol(eax, cl); - break; - } - case 8: { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - rol(rax, cl); - break; - } - } - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_ROR: { - auto Op = IROp->C(); - uint8_t Mask = OpSize * 8 - 1; - - mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - and(rcx, Mask); - switch (OpSize) { - case 1: { - movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - ror(al, cl); - break; - } - case 2: { - movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - ror(ax, cl); - break; - } - case 4: { - mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - ror(eax, cl); - break; - } - case 8: { - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - ror(rax, cl); - break; - } - } - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - - case IR::OP_SELECT: { - auto Op = IROp->C(); - - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - cmp(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - - switch (Op->Cond) { - case FEXCore::IR::COND_EQ: - mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); - cmove(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - break; - case FEXCore::IR::COND_NEQ: - mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); - cmovne(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - break; - case FEXCore::IR::COND_GE: - mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); - cmovge(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - break; - case FEXCore::IR::COND_LT: - mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - cmovae(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); - break; - case FEXCore::IR::COND_GT: - mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); - cmovg(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - break; - case FEXCore::IR::COND_LE: - mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); - cmovle(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); - break; - case FEXCore::IR::COND_CS: - case FEXCore::IR::COND_CC: - case FEXCore::IR::COND_MI: - case FEXCore::IR::COND_PL: - case FEXCore::IR::COND_VS: - case FEXCore::IR::COND_VC: - case FEXCore::IR::COND_HI: - case FEXCore::IR::COND_LS: - default: - LogMan::Msg::A("Unsupported compare type"); - break; - } - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); - break; - } - - case IR::OP_CAS: { - auto Op = IROp->C(); - // Args[0]: Expected - // Args[1]: Desired - // Args[2]: Pointer - // DataSrc = *Src1 - // if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc - // This will write to memory! Careful! - // Third operand must be a calculated guest memory address - //OrderedNode *CASResult = _CAS(Src3, Src2, Src1); - uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); - - mov(rcx, Memory); - add(rcx, qword [TEMP_STACK + (Op->Header.Args[2].ID() * 16)]); - - mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); - mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); - - // RCX now contains pointer - // RAX contains our expected value - // RDX contains our desired - - lock(); - - switch (OpSize) { - case 1: { - cmpxchg(byte [rcx], dl); - movzx(rax, al); - break; - } - case 2: { - cmpxchg(word [rcx], dx); - movzx(rax, ax); - break; - } - case 4: { - cmpxchg(dword [rcx], edx); - break; - } - case 8: { - cmpxchg(qword [rcx], rdx); - break; - } - default: LogMan::Msg::A("Unsupported: %d", OpSize); - } - - // RAX now contains the result - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - case IR::OP_VCMPEQ: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - - LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); - - switch (Op->ElementSize) { - case 1: { - pcmpeqb(xmm0, xmm1); - break; - } - case 2: { - pcmpeqw(xmm0, xmm1); - break; - } - case 4: { - pcmpeqd(xmm0, xmm1); - break; - } - case 8: { - pcmpeqq(xmm0, xmm1); - break; - } - - default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); - } - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - break; - } - case IR::OP_VCMPGT: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - - LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); - - switch (Op->ElementSize) { - case 1: { - pcmpgtb(xmm0, xmm1); - break; - } - case 2: { - pcmpgtw(xmm0, xmm1); - break; - } - case 4: { - pcmpgtd(xmm0, xmm1); - break; - } - case 8: { - pcmpgtq(xmm0, xmm1); - break; - } - - default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); - } - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - break; - } - - case IR::OP_VXOR: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - pxor(xmm0, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - break; - } - case IR::OP_VOR: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - por(xmm0, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - break; - } - - case IR::OP_VINSELEMENT: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - - // Dst_d[Op->DestIdx] = Src2_d[Op->SrcIdx]; - - // pextrq reg64/mem64, xmm, imm - // pinsrq xmm, reg64/mem64, imm8 - switch (Op->ElementSize) { - case 1: { - pextrb(al, xmm1, Op->SrcIdx); - pinsrb(xmm0, al, Op->DestIdx); - break; - } - case 2: { - pextrw(ax, xmm1, Op->SrcIdx); - pinsrw(xmm0, ax, Op->DestIdx); - break; - } - case 4: { - pextrd(eax, xmm1, Op->SrcIdx); - pinsrd(xmm0, eax, Op->DestIdx); - break; - } - case 8: { - pextrq(rax, xmm1, Op->SrcIdx); - pinsrq(xmm0, rax, Op->DestIdx); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - break; - } - case IR::OP_VADD: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - switch (Op->ElementSize) { - case 1: { - paddb(xmm0, xmm1); - break; - } - case 2: { - paddw(xmm0, xmm1); - break; - } - case 4: { - paddd(xmm0, xmm1); - break; - } - case 8: { - paddq(xmm0, xmm1); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - - break; - } - - case IR::OP_VUSHLS: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - - switch (Op->ElementSize) { - case 2: { - psllw(xmm0, xmm1); - break; - } - case 4: { - pslld(xmm0, xmm1); - break; - } - case 8: { - psllq(xmm0, xmm1); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - - break; - } - case IR::OP_VZIP: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - switch (Op->ElementSize) { - case 1: { - punpcklbw(xmm0, xmm1); - break; - } - case 2: { - punpcklwd(xmm0, xmm1); - break; - } - case 4: { - punpckldq(xmm0, xmm1); - break; - } - case 8: { - punpcklqdq(xmm0, xmm1); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - - break; - } - case IR::OP_VZIP2: { - auto Op = IROp->C(); - movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); - movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); - switch (Op->ElementSize) { - case 1: { - punpckhbw(xmm0, xmm1); - break; - } - case 2: { - punpckhwd(xmm0, xmm1); - break; - } - case 4: { - punpckhdq(xmm0, xmm1); - break; - } - case 8: { - punpckhqdq(xmm0, xmm1); - break; - } - default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; - } - movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); - - break; - } - - case IR::OP_CYCLECOUNTER: { -#ifdef DEBUG_CYCLES - mov (rax, 0); -#else - rdtsc(); - shl(rdx, 32); - or(rax, rdx); -#endif - mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); - break; - } - default: break; - } + if (SpillSlots) { + sub(rsp, SpillSlots * 16); } - ++Begin; + auto HeaderIterator = CurrentIR->begin(); + IR::OrderedNodeWrapper *HeaderNodeWrapper = HeaderIterator(); + IR::OrderedNode *HeaderNode = HeaderNodeWrapper->GetNode(ListBegin); + auto HeaderOp = HeaderNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == IR::OP_IRHEADER, "First op wasn't IRHeader"); + + IR::OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); + while (1) { + using namespace FEXCore::IR; + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block"); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR->at(BlockIROp->Begin); + auto CodeLast = CurrentIR->at(BlockIROp->Last); + + while (1) { + OrderedNodeWrapper *WrapperOp = CodeBegin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); + FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + uint8_t OpSize = IROp->Size; + uint32_t Node = WrapperOp->ID(); + + if (HasRA) { + #ifdef DEBUG_RA + if (IROp->Op != IR::OP_BEGINBLOCK && + IROp->Op != IR::OP_CONDJUMP && + IROp->Op != IR::OP_JUMP) { + std::stringstream Inst; + auto Name = FEXCore::IR::GetName(IROp->Op); + + if (IROp->HasDest) { + uint32_t PhysReg = RAPass->GetNodeRegister(Node); + if (PhysReg >= XMMBase) + Inst << "\tXMM" << GetPhys(Node) << " = " << Name << " "; + else + Inst << "\tReg" << GetPhys(Node) << " = " << Name << " "; + } + else { + Inst << "\t" << Name << " "; + } + + for (uint8_t i = 0; i < IROp->NumArgs; ++i) { + uint32_t ArgNode = IROp->Args[i].ID(); + uint32_t PhysReg = RAPass->GetNodeRegister(ArgNode); + if (PhysReg >= XMMBase) + Inst << "XMM" << GetPhys(ArgNode) << (i + 1 == IROp->NumArgs ? "" : ", "); + else + Inst << "Reg" << GetPhys(ArgNode) << (i + 1 == IROp->NumArgs ? "" : ", "); + } + + LogMan::Msg::D("%s", Inst.str().c_str()); + } + #endif + } + + switch (IROp->Op) { + case IR::OP_BEGINBLOCK: { + auto IsTarget = JumpTargets.find(WrapperOp->ID()); + if (IsTarget == JumpTargets.end()) { + JumpTargets.try_emplace(WrapperOp->ID()); + } + else { + L(IsTarget->second); + } + break; + } + case IR::OP_ENDBLOCK: { + auto Op = IROp->C(); + if (Op->RIPIncrement) { + add(qword [STATE + offsetof(FEXCore::Core::CPUState, rip)], Op->RIPIncrement); + } + break; + } + case IR::OP_EXITFUNCTION: + case IR::OP_ENDFUNCTION: { + if (SpillSlots) { + add(rsp, SpillSlots * 16); + } + + if (HasRA) { + pop(r15); + pop(r14); + pop(r13); + pop(r12); + pop(rbp); + pop(rbx); + } + + ret(); + break; + } + case IR::OP_BREAK: { + auto Op = IROp->C(); + switch (Op->Reason) { + case 4: // HLT + ud2(); + break; + default: LogMan::Msg::A("Unknown Break reason: %d", Op->Reason); + } + break; + } + case IR::OP_JUMP: { + auto Op = IROp->C(); + + Label *TargetLabel; + auto IsTarget = JumpTargets.find(Op->Header.Args[0].ID()); + if (IsTarget == JumpTargets.end()) { + TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[0].ID()).first->second; + } + else { + TargetLabel = &IsTarget->second; + } + + jmp(*TargetLabel, T_NEAR); + break; + } + default: break; + } + + if (HasRA) { + switch (IROp->Op) { + case IR::OP_CONDJUMP: { + auto Op = IROp->C(); + + Label *TargetLabel; + auto IsTarget = JumpTargets.find(Op->Header.Args[1].ID()); + if (IsTarget == JumpTargets.end()) { + TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[1].ID()).first->second; + } + else { + TargetLabel = &IsTarget->second; + } + + cmp(GetSrc(Op->Header.Args[0].ID()), 0); + jne(*TargetLabel, T_NEAR); + break; + } + case IR::OP_LOADCONTEXT: { + auto Op = IROp->C(); + switch (Op->Size) { + case 1: { + mov(GetDst(Node), byte [STATE + Op->Offset]); + } + break; + case 2: { + mov(GetDst(Node), word [STATE + Op->Offset]); + } + break; + case 4: { + mov(GetDst(Node), dword [STATE + Op->Offset]); + } + break; + case 8: { + mov(GetDst(Node), qword [STATE + Op->Offset]); + } + break; + case 16: { + if (Op->Offset % 16 == 0) + movaps(GetDst(Node), xword [STATE + Op->Offset]); + else + movups(GetDst(Node), xword [STATE + Op->Offset]); + } + break; + default: LogMan::Msg::A("Unhandled LoadContext size: %d", Op->Size); + } + break; + } + case IR::OP_STORECONTEXT: { + auto Op = IROp->C(); + + switch (Op->Size) { + case 1: { + mov(byte [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); + } + break; + + case 2: { + mov(word [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 4: { + mov(dword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 8: { + mov(qword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 16: { + if (Op->Offset % 16 == 0) + movaps(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); + else + movups(xword [STATE + Op->Offset], GetSrc(Op->Header.Args[0].ID())); + } + break; + default: LogMan::Msg::A("Unhandled StoreContext size: %d", Op->Size); + } + break; + } + case IR::OP_FILLREGISTER: { + auto Op = IROp->C(); + uint32_t SlotOffset = Op->Slot * 16; + switch (OpSize) { + case 1: { + mov(GetDst(Node), byte [rsp + SlotOffset]); + } + break; + case 2: { + mov(GetDst(Node), word [rsp + SlotOffset]); + } + break; + case 4: { + mov(GetDst(Node), dword [rsp + SlotOffset]); + } + break; + case 8: { + mov(GetDst(Node), qword [rsp + SlotOffset]); + } + break; + case 16: { + movaps(GetDst(Node), xword [rsp + SlotOffset]); + } + break; + default: LogMan::Msg::A("Unhandled FillRegister size: %d", OpSize); + } + break; + } + case IR::OP_SPILLREGISTER: { + auto Op = IROp->C(); + uint32_t SlotOffset = Op->Slot * 16; + switch (OpSize) { + case 1: { + mov(byte [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 2: { + mov(word [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 4: { + mov(dword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 8: { + mov(qword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID())); + } + break; + case 16: { + movaps(xword [rsp + SlotOffset], GetSrc(Op->Header.Args[0].ID())); + } + break; + default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize); + } + break; + } + case IR::OP_ADD: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + mov(rax, GetSrc(Op->Header.Args[1].ID())); + add(rax, GetSrc(Op->Header.Args[0].ID())); + mov(Dst, rax); + break; + } + case IR::OP_SUB: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + mov(rax, GetSrc(Op->Header.Args[0].ID())); + sub(rax, GetSrc(Op->Header.Args[1].ID())); + mov(Dst, rax); + break; + } + case IR::OP_XOR: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + mov(rax, GetSrc(Op->Header.Args[1].ID())); + xor(rax, GetSrc(Op->Header.Args[0].ID())); + mov(Dst, rax); + break; + } + case IR::OP_AND: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + mov(rax, GetSrc(Op->Header.Args[1].ID())); + and(rax, GetSrc(Op->Header.Args[0].ID())); + mov(Dst, rax); + break; + } + case IR::OP_OR: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + mov(rax, GetSrc(Op->Header.Args[1].ID())); + or (rax, GetSrc(Op->Header.Args[0].ID())); + mov(Dst, rax); + break; + } + case IR::OP_MOV: { + auto Op = IROp->C(); + mov (GetDst(Node), GetSrc(Op->Header.Args[0].ID())); + break; + } + case IR::OP_CONSTANT: { + auto Op = IROp->C(); + mov(GetDst(Node), Op->Constant); + break; + } + case IR::OP_POPCOUNT: { + auto Op = IROp->C(); + auto Dst64 = GetDst(Node); + + switch (OpSize) { + case 1: + movzx(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); + popcnt(Dst64, Dst64); + break; + case 2: { + movzx(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); + popcnt(Dst64, Dst64); + break; + } + case 4: + popcnt(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); + break; + case 8: + popcnt(GetDst(Node), GetSrc(Op->Header.Args[0].ID())); + break; + } + break; + } + case IR::OP_ZEXT: { + auto Op = IROp->C(); + LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); + + uint32_t PhysReg = RAPass->GetNodeRegister(Op->Header.Args[0].ID()); + if (PhysReg >= XMMBase) { + // XMM -> GPR transfer with free truncation + switch (Op->SrcSize) { + case 8: + pextrb(al, GetSrc(Op->Header.Args[0].ID()), 0); + break; + case 16: + pextrw(ax, GetSrc(Op->Header.Args[0].ID()), 0); + break; + case 32: + pextrd(eax, GetSrc(Op->Header.Args[0].ID()), 0); + break; + case 64: + pextrw(rax, GetSrc(Op->Header.Args[0].ID()), 0); + break; + default: LogMan::Msg::A("Unhandled Zext size: %d", Op->SrcSize); break; + } + auto Dst = GetDst(Node); + mov(Dst, rax); + } + else { + if (Op->SrcSize == 64) { + vmovq(xmm15, Reg64(GetSrc(Op->Header.Args[0].ID()).getIdx())); + movapd(GetDst(Node), xmm15); + } + else { + auto Dst = GetDst(Node); + mov(rax, uint64_t((1ULL << Op->SrcSize) - 1)); + and(rax, GetSrc(Op->Header.Args[0].ID())); + mov(Dst, rax); + } + } + break; + } + case IR::OP_SEXT: { + auto Op = IROp->C(); + LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); + auto Dst = GetDst(Node); + + switch (Op->SrcSize / 8) { + case 1: + movsx(Dst, GetSrc(Op->Header.Args[0].ID())); + break; + case 2: + movsx(Dst, GetSrc(Op->Header.Args[0].ID())); + break; + case 4: + movsxd(Reg64(Dst.getIdx()), GetSrc(Op->Header.Args[0].ID())); + break; + case 8: + mov(Dst, GetSrc(Op->Header.Args[0].ID())); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", Op->SrcSize / 8); + } + break; + } + case IR::OP_BFE: { + auto Op = IROp->C(); + LogMan::Throw::A(OpSize <= 16, "OpSize is too large for BFE: %d", OpSize); + if (OpSize == 16) { + LogMan::Throw::A(!(Op->lsb < 64 && (Op->lsb + Op->Width > 64)), "Trying to BFE an XMM across the 64bit split: Beginning at %d, ending at %d", Op->lsb, Op->lsb + Op->Width); + movups(xmm15, GetSrc(Op->Header.Args[0].ID())); + uint8_t Offset = Op->lsb; + if (Offset < 64) { + pextrq(rax, xmm15, 0); + } + else { + pextrq(rax, xmm15, 1); + Offset -= 64; + } + + if (Offset) { + shr(rax, Offset); + } + + if (Op->Width != 64) { + mov(rcx, uint64_t((1ULL << Op->Width) - 1)); + and(rax, rcx); + } + + mov (GetDst(Node), rax); + } + else { + auto Dst = GetDst(Node); + mov(rax, GetSrc(Op->Header.Args[0].ID())); + + if (Op->lsb != 0) + shr(rax, Op->lsb); + + if (Op->Width != 64) { + mov(rcx, uint64_t((1ULL << Op->Width) - 1)); + and(rax, rcx); + } + mov(Dst, rax); + } + break; + } + case IR::OP_LSHR: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + auto Dst = GetDst(Node); + mov (rcx, GetSrc(Op->Header.Args[1].ID())); + and(rcx, Mask); + + shrx(Reg32e(Dst.getIdx(), 64), GetSrc(Op->Header.Args[0].ID()), rcx); + break; + } + case IR::OP_LSHL: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + auto Dst = GetDst(Node); + mov (rcx, GetSrc(Op->Header.Args[1].ID())); + and(rcx, Mask); + + shlx(Reg32e(Dst.getIdx(), 64), GetSrc(Op->Header.Args[0].ID()), rcx); + break; + } + case IR::OP_ASHR: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov (rcx, GetSrc(Op->Header.Args[1].ID())); + and(rcx, Mask); + switch (OpSize) { + case 1: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + sar(al, cl); + movsx(GetDst(Node), al); + break; + case 2: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + sar(ax, cl); + movsx(GetDst(Node), ax); + break; + case 4: + sarx(Reg32e(GetDst(Node).getIdx(), 32), GetSrc(Op->Header.Args[0].ID()), ecx); + break; + case 8: + sarx(Reg32e(GetDst(Node).getIdx(), 64), GetSrc(Op->Header.Args[0].ID()), rcx); + break; + default: LogMan::Msg::A("Unknown ASHR Size: %d\n", OpSize); break; + }; + break; + } + case IR::OP_ROL: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov (rcx, GetSrc(Op->Header.Args[1].ID())); + and(rcx, Mask); + switch (OpSize) { + case 1: { + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + rol(al, cl); + break; + } + case 2: { + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + rol(ax, cl); + break; + } + case 4: { + mov(eax, GetSrc(Op->Header.Args[0].ID())); + rol(eax, cl); + break; + } + case 8: { + mov(rax, GetSrc(Op->Header.Args[0].ID())); + rol(rax, cl); + break; + } + } + mov(GetDst(Node), rax); + break; + } + case IR::OP_ROR: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov (rcx, GetSrc(Op->Header.Args[1].ID())); + and(rcx, Mask); + switch (OpSize) { + case 1: { + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + ror(al, cl); + break; + } + case 2: { + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + ror(ax, cl); + break; + } + case 4: { + mov(eax, GetSrc(Op->Header.Args[0].ID())); + ror(eax, cl); + break; + } + case 8: { + mov(rax, GetSrc(Op->Header.Args[0].ID())); + ror(rax, cl); + break; + } + } + mov(GetDst(Node), rax); + break; + } + case IR::OP_MUL: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + + switch (OpSize) { + case 1: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + movsx(rcx, GetSrc(Op->Header.Args[1].ID())); + imul(cl); + movsx(Dst, al); + break; + case 2: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + movsx(rcx, GetSrc(Op->Header.Args[1].ID())); + imul(cx); + movsx(Dst, ax); + break; + case 4: + movsxd(rax, GetSrc(Op->Header.Args[0].ID())); + imul(eax, GetSrc(Op->Header.Args[1].ID())); + movsx(Dst, eax); + break; + case 8: + mov(rax, GetSrc(Op->Header.Args[0].ID())); + imul(rax, GetSrc(Op->Header.Args[1].ID())); + mov(Dst, rax); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_MULH: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + movsx(rcx, GetSrc(Op->Header.Args[1].ID())); + imul(cl); + movsx(rax, ax); + mov(GetDst(Node), rax); + break; + case 2: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + movsx(rcx, GetSrc(Op->Header.Args[1].ID())); + imul(cx); + movsx(rax, dx); + mov(GetDst(Node), rax); + break; + case 4: + movsx(rax, GetSrc(Op->Header.Args[0].ID())); + imul(GetSrc(Op->Header.Args[1].ID())); + movsxd(rax, edx); + mov(GetDst(Node), rdx); + break; + case 8: + mov(rax, GetSrc(Op->Header.Args[0].ID())); + imul(GetSrc(Op->Header.Args[1].ID())); + mov(GetDst(Node), rdx); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_UMUL: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + movzx(rcx, GetSrc(Op->Header.Args[1].ID())); + mul(cl); + movzx(rax, al); + mov(GetDst(Node), rax); + break; + case 2: + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + movzx(rcx, GetSrc(Op->Header.Args[1].ID())); + mul(cx); + movzx(rax, ax); + mov(GetDst(Node), rax); + break; + case 4: + mov(rax, GetSrc(Op->Header.Args[0].ID())); + mul(GetSrc(Op->Header.Args[1].ID())); + mov(GetDst(Node), rax); + break; + case 8: + mov(rax, GetSrc(Op->Header.Args[0].ID())); + mul(GetSrc(Op->Header.Args[1].ID())); + mov(GetDst(Node), rax); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_UMULH: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + movzx(rcx, GetSrc(Op->Header.Args[1].ID())); + mul(cl); + movzx(rax, ax); + mov(GetDst(Node), rax); + break; + case 2: + movzx(rax, GetSrc(Op->Header.Args[0].ID())); + movzx(rcx, GetSrc(Op->Header.Args[1].ID())); + mul(cx); + movzx(rax, dx); + mov(GetDst(Node), rax); + break; + case 4: + mov(rax, GetSrc(Op->Header.Args[0].ID())); + mul(GetSrc(Op->Header.Args[1].ID())); + mov(GetDst(Node), rdx); + break; + case 8: + mov(rax, GetSrc(Op->Header.Args[0].ID())); + mul(GetSrc(Op->Header.Args[1].ID())); + mov(GetDst(Node), rdx); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_LDIV: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov(eax, GetSrc(Op->Header.Args[0].ID())); + mov(edx, GetSrc(Op->Header.Args[1].ID())); + mov(ecx, GetSrc(Op->Header.Args[2].ID())); + idiv(ecx); + mov(GetDst(Node), rax); + break; + } + case 8: { + mov(rax, GetSrc(Op->Header.Args[0].ID())); + mov(rdx, GetSrc(Op->Header.Args[1].ID())); + mov(rcx, GetSrc(Op->Header.Args[2].ID())); + idiv(rcx); + mov(GetDst(Node), rax); + break; + } + default: LogMan::Msg::A("Unknown LDIV Size: %d", Size); break; + } + break; + } + case IR::OP_LREM: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov(eax, GetSrc(Op->Header.Args[0].ID())); + mov(edx, GetSrc(Op->Header.Args[1].ID())); + mov(ecx, GetSrc(Op->Header.Args[2].ID())); + idiv(ecx); + mov(GetDst(Node), rdx); + break; + } + + case 8: { + mov(rax, GetSrc(Op->Header.Args[0].ID())); + mov(rdx, GetSrc(Op->Header.Args[1].ID())); + mov(rcx, GetSrc(Op->Header.Args[2].ID())); + idiv(rcx); + mov(GetDst(Node), rdx); + break; + } + default: LogMan::Msg::A("Unknown LREM Size: %d", Size); break; + } + break; + } + case IR::OP_LUDIV: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov (eax, GetSrc(Op->Header.Args[0].ID())); + mov (edx, GetSrc(Op->Header.Args[1].ID())); + mov (ecx, GetSrc(Op->Header.Args[2].ID())); + div(ecx); + mov(GetDst(Node), rax); + break; + } + case 8: { + mov (rax, GetSrc(Op->Header.Args[0].ID())); + mov (rdx, GetSrc(Op->Header.Args[1].ID())); + mov (rcx, GetSrc(Op->Header.Args[2].ID())); + div(rcx); + mov(GetDst(Node), rax); + break; + } + default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; + } + break; + } + case IR::OP_LUREM: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov (eax, GetSrc(Op->Header.Args[0].ID())); + mov (edx, GetSrc(Op->Header.Args[1].ID())); + mov (ecx, GetSrc(Op->Header.Args[2].ID())); + div(ecx); + mov(GetDst(Node), rdx); + break; + } + + case 8: { + mov (rax, GetSrc(Op->Header.Args[0].ID())); + mov (rdx, GetSrc(Op->Header.Args[1].ID())); + mov (rcx, GetSrc(Op->Header.Args[2].ID())); + div(rcx); + mov(GetDst(Node), rdx); + break; + } + default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; + } + break; + } + case IR::OP_LOADFLAG: { + auto Op = IROp->C(); + + auto Dst = GetDst(Node); + movzx(Dst, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]); + and(Dst, 1); + break; + } + case IR::OP_STOREFLAG: { + auto Op = IROp->C(); + + mov (rax, GetSrc(Op->Header.Args[0].ID())); + and(rax, 1); + mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al); + break; + } + case IR::OP_SELECT: { + auto Op = IROp->C(); + auto Dst = GetDst(Node); + + mov(rax, GetSrc(Op->Header.Args[0].ID())); + cmp(rax, GetSrc(Op->Header.Args[1].ID())); + + switch (Op->Cond) { + case FEXCore::IR::COND_EQ: + mov(rax, GetSrc(Op->Header.Args[3].ID())); + cmove(rax, GetSrc(Op->Header.Args[2].ID())); + break; + case FEXCore::IR::COND_NEQ: + mov(rax, GetSrc(Op->Header.Args[3].ID())); + cmovne(rax, GetSrc(Op->Header.Args[2].ID())); + break; + case FEXCore::IR::COND_GE: + mov(rax, GetSrc(Op->Header.Args[3].ID())); + cmovge(rax, GetSrc(Op->Header.Args[2].ID())); + break; + case FEXCore::IR::COND_LT: + mov(rax, GetSrc(Op->Header.Args[2].ID())); + cmovae(rax, GetSrc(Op->Header.Args[3].ID())); + break; + case FEXCore::IR::COND_GT: + mov(rax, GetSrc(Op->Header.Args[3].ID())); + cmovg(rax, GetSrc(Op->Header.Args[2].ID())); + break; + case FEXCore::IR::COND_LE: + mov(rax, GetSrc(Op->Header.Args[3].ID())); + cmovle(rax, GetSrc(Op->Header.Args[2].ID())); + break; + case FEXCore::IR::COND_CS: + case FEXCore::IR::COND_CC: + case FEXCore::IR::COND_MI: + case FEXCore::IR::COND_PL: + case FEXCore::IR::COND_VS: + case FEXCore::IR::COND_VC: + case FEXCore::IR::COND_HI: + case FEXCore::IR::COND_LS: + default: + LogMan::Msg::A("Unsupported compare type"); + break; + } + mov (Dst, rax); + break; + } + case IR::OP_LOADMEM: { + auto Op = IROp->C(); + uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); + + auto Dst = GetDst(Node); + mov(rax, Memory); + add(rax, GetSrc(Op->Header.Args[0].ID())); + switch (Op->Size) { + case 1: { + movzx (Dst, byte [rax]); + } + break; + case 2: { + movzx (Dst, word [rax]); + } + break; + case 4: { + mov(Dst, dword [rax]); + } + break; + case 8: { + mov(Dst, qword [rax]); + } + break; + case 16: { + movups(GetDst(Node), xword [rax]); + if (MemoryDebug) { + movq(rcx, GetDst(Node)); + } + } + break; + default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size); + } + break; + } + case IR::OP_STOREMEM: { + auto Op = IROp->C(); + uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); + + mov(rax, Memory); + add(rax, GetSrc(Op->Header.Args[0].ID())); + switch (Op->Size) { + case 1: + mov(byte [rax], GetSrc(Op->Header.Args[1].ID())); + break; + case 2: + mov(word [rax], GetSrc(Op->Header.Args[1].ID())); + break; + case 4: + mov(dword [rax], GetSrc(Op->Header.Args[1].ID())); + break; + case 8: + mov(qword [rax], GetSrc(Op->Header.Args[1].ID())); + break; + case 16: + movups(xword [rax], GetSrc(Op->Header.Args[1].ID())); + break; + default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size); + } + break; + } + case IR::OP_SYSCALL: { + auto Op = IROp->C(); + // XXX: This is very terrible, but I don't care for right now + + push(rdi); + + for (auto &Reg : RA64) + push(Reg); + + // Syscall ABI for x86-64 + // this: rdi + // Thread: rsi + // ArgPointer: rdx (Stack) + // + // Result: RAX + + // These are pushed in reverse order because stacks + for (uint32_t i = 7; i > 0; --i) + push(GetSrc(Op->Header.Args[i - 1].ID())); + + mov(rsi, rdi); // Move thread in to rsi + mov(rdi, reinterpret_cast(&CTX->SyscallHandler)); + mov(rdx, rsp); + + using PtrType = uint64_t (FEXCore::SyscallHandler::*)(FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args); + union { + PtrType ptr; + uint64_t Raw; + } PtrCast; + PtrCast.ptr = &FEXCore::SyscallHandler::HandleSyscall; + mov(rax, PtrCast.Raw); + call(rax); + + // Reload arguments just in case they are sill live after the fact + for (uint32_t i = 0; i < 7; ++i) + pop(GetSrc(Op->Header.Args[i].ID())); + + for (uint32_t i = RA64.size(); i > 0; --i) + pop(RA64[i - 1]); + + pop(rdi); + + mov (GetDst(Node), rax); + break; + } + case IR::OP_CPUID: { + auto Op = IROp->C(); + using ClassPtrType = FEXCore::CPUIDEmu::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t Function); + union { + ClassPtrType ClassPtr; + uint64_t Raw; + } Ptr; + Ptr.ClassPtr = &CPUIDEmu::RunFunction; + + for (auto &Reg : RA64) + push(Reg); + + // CPUID ABI + // this: rdi + // Function: rsi + // + // Result: RAX, RDX. 4xi32 + push(rdi); + mov (rsi, GetSrc(Op->Header.Args[0].ID())); + mov (rdi, reinterpret_cast(&CTX->CPUID)); + + sub(rsp, 8); // Align + + mov(rax, Ptr.Raw); + call(rax); + + add(rsp, 8); // Align + + pop(rdi); + + for (uint32_t i = RA64.size(); i > 0; --i) + pop(RA64[i - 1]); + + auto Dst = GetDst(Node); + pinsrq(Dst, rax, 0); + pinsrd(Dst, rdx, 1); + break; + } + case IR::OP_EXTRACTELEMENT: { + auto Op = IROp->C(); + + uint32_t PhysReg = RAPass->GetNodeRegister(Op->Header.Args[0].ID()); + if (PhysReg >= XMMBase) { + switch (OpSize) { + case 1: + pextrb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); + break; + case 2: + pextrw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); + break; + case 4: + pextrd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); + break; + case 8: + pextrq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), Op->Idx); + break; + default: LogMan::Msg::A("Unhandled ExtractElementSize: %d", OpSize); + } + } + else { + LogMan::Msg::A("Can't handle extract from GPR yet"); + } + break; + } + case IR::OP_VINSELEMENT: { + auto Op = IROp->C(); + movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); + + // Dst_d[Op->DestIdx] = Src2_d[Op->SrcIdx]; + + // pextrq reg64/mem64, xmm, imm + // pinsrq xmm, reg64/mem64, imm8 + switch (Op->ElementSize) { + case 1: { + pextrb(al, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); + pinsrb(xmm15, al, Op->DestIdx); + break; + } + case 2: { + pextrw(ax, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); + pinsrw(xmm15, ax, Op->DestIdx); + break; + } + case 4: { + pextrd(eax, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); + pinsrd(xmm15, eax, Op->DestIdx); + break; + } + case 8: { + pextrq(rax, GetSrc(Op->Header.Args[1].ID()), Op->SrcIdx); + pinsrq(xmm15, rax, Op->DestIdx); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + + movapd(GetDst(Node), xmm15); + break; + } + case IR::OP_VADD: { + auto Op = IROp->C(); + switch (Op->ElementSize) { + case 1: { + vpaddb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 2: { + vpaddw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 4: { + vpaddd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 8: { + vpaddq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + break; + } + case IR::OP_VSUB: { + auto Op = IROp->C(); + switch (Op->ElementSize) { + case 1: { + vpsubb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 2: { + vpsubw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 4: { + vpsubd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 8: { + vpsubq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + break; + } + case IR::OP_VXOR: { + auto Op = IROp->C(); + vpxor(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case IR::OP_VOR: { + auto Op = IROp->C(); + vpor(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case IR::OP_VCMPEQ: { + auto Op = IROp->C(); + LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); + + switch (Op->ElementSize) { + case 1: + vpcmpeqb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + case 2: + vpcmpeqw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + case 4: + vpcmpeqd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + case 8: + vpcmpeqq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); + } + break; + } + case IR::OP_VCMPGT: { + auto Op = IROp->C(); + LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); + + switch (Op->ElementSize) { + case 1: + vpcmpgtb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + case 2: + vpcmpgtw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + case 4: + vpcmpgtd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + case 8: + vpcmpgtq(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); + } + break; + } + case IR::OP_VZIP: { + auto Op = IROp->C(); + movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); + + switch (Op->ElementSize) { + case 1: { + punpcklbw(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + case 2: { + punpcklwd(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + case 4: { + punpckldq(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + case 8: { + punpcklqdq(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movapd(GetDst(Node), xmm15); + break; + } + case IR::OP_VZIP2: { + auto Op = IROp->C(); + movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); + + switch (Op->ElementSize) { + case 1: { + punpckhbw(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + case 2: { + punpckhwd(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + case 4: { + punpckhdq(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + case 8: { + punpckhqdq(xmm15, GetSrc(Op->Header.Args[1].ID())); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movapd(GetDst(Node), xmm15); + break; + } + case IR::OP_VUSHLS: { + auto Op = IROp->C(); + movapd(xmm15, GetSrc(Op->Header.Args[0].ID())); + vmovq(xmm14, Reg64(GetSrc(Op->Header.Args[1].ID()).getIdx())); + + switch (Op->ElementSize) { + case 2: { + psllw(xmm15, xmm14); + break; + } + case 4: { + pslld(xmm15, xmm14); + break; + } + case 8: { + psllq(xmm15, xmm14); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movapd(GetDst(Node), xmm15); + + break; + } + case IR::OP_VEXTR: { + auto Op = IROp->C(); + vpalignr(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID()), Op->Index); + break; + } + case IR::OP_VUMIN: { + auto Op = IROp->C(); + switch (Op->ElementSize) { + case 1: { + vpminub(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 2: { + vpminuw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 4: { + vpminud(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + break; + } + case IR::OP_VSMIN: { + auto Op = IROp->C(); + switch (Op->ElementSize) { + case 1: { + vpminsb(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 2: { + vpminsw(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + case 4: { + vpminsd(GetDst(Node), GetSrc(Op->Header.Args[0].ID()), GetSrc(Op->Header.Args[1].ID())); + break; + } + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + break; + } + case IR::OP_CAS: { + auto Op = IROp->C(); + // Args[0]: Desired + // Args[1]: Expected + // Args[2]: Pointer + // DataSrc = *Src1 + // if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc + // This will write to memory! Careful! + // Third operand must be a calculated guest memory address + //OrderedNode *CASResult = _CAS(Src3, Src2, Src1); + uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); + + mov(rcx, Memory); + add(rcx, GetSrc(Op->Header.Args[2].ID())); + mov(rdx, GetSrc(Op->Header.Args[1].ID())); + mov(rax, GetSrc(Op->Header.Args[0].ID())); + + // RCX now contains pointer + // RAX contains our expected value + // RDX contains our desired + + lock(); + + switch (OpSize) { + case 1: { + cmpxchg(byte [rcx], dl); + movzx(rax, al); + break; + } + case 2: { + cmpxchg(word [rcx], dx); + movzx(rax, ax); + break; + } + case 4: { + cmpxchg(dword [rcx], edx); + break; + } + case 8: { + cmpxchg(qword [rcx], rdx); + break; + } + default: LogMan::Msg::A("Unsupported: %d", OpSize); + } + + // RAX now contains the result + mov (GetDst(Node), rax); + break; + } + case IR::OP_CYCLECOUNTER: { + #ifdef DEBUG_CYCLES + mov (GetDst(Node), 0); + #else + rdtsc(); + shl(rdx, 32); + or(rax, rdx); + mov (GetDst(Node), rax); + #endif + break; + } + case IR::OP_FINDLSB: { + auto Op = IROp->C(); + tzcnt(rcx, GetSrc(Op->Header.Args[0].ID())); + xor(rax, rax); + cmp(GetSrc(Op->Header.Args[0].ID()), 1); + sbb(rax, rax); + or(rax, rcx); + mov (GetDst(Node), rax); + break; + } + case IR::OP_FINDMSB: { + auto Op = IROp->C(); + mov(rax, OpSize * 8); + lzcnt(rcx, GetSrc(Op->Header.Args[0].ID())); + sub(rax, rcx); + mov (GetDst(Node), rax); + break; + } + case IR::OP_CODEBLOCK: + case IR::OP_IRHEADER: + case IR::OP_BEGINBLOCK: + case IR::OP_ENDBLOCK: + case IR::OP_EXITFUNCTION: + case IR::OP_ENDFUNCTION: + case IR::OP_BREAK: + case IR::OP_JUMP: + break; + default: + LogMan::Msg::A("Unknown IR Op: %d(%s)", IROp->Op, FEXCore::IR::GetName(IROp->Op).data()); + break; + } + } + else { + switch (IROp->Op) { + case IR::OP_LOADCONTEXT: { + auto Op = IROp->C(); + #define LOAD_CTX(x, y) \ + case x: { \ + movzx(rax, y [STATE + Op->Offset]); \ + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); \ + } \ + break + switch (Op->Size) { + LOAD_CTX(1, byte); + LOAD_CTX(2, word); + case 4: { + mov(eax, dword [STATE + Op->Offset]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + case 8: { + mov(rax, qword [STATE + Op->Offset]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + case 16: { + if (Op->Offset % 16 == 0) { + movaps(xmm0, xword [STATE + Op->Offset]); + movaps(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + } + else { + movups(xmm0, xword [STATE + Op->Offset]); + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + } + } + break; + default: LogMan::Msg::A("Unhandled LoadContext size: %d", Op->Size); + } + #undef LOAD_CTX + break; + } + case IR::OP_STORECONTEXT: { + auto Op = IROp->C(); + + switch (Op->Size) { + case 1: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(byte [STATE + Op->Offset], al); + } + break; + + case 2: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(word [STATE + Op->Offset], ax); + } + break; + case 4: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(dword [STATE + Op->Offset], eax); + } + break; + case 8: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(qword [STATE + Op->Offset], rax); + } + break; + case 16: { + if (Op->Offset % 16 == 0) { + movaps(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movaps(xword [STATE + Op->Offset], xmm0); + } + else { + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xword [STATE + Op->Offset], xmm0); + } + } + break; + default: LogMan::Msg::A("Unhandled StoreContext size: %d", Op->Size); + } + + break; + } + case IR::OP_FILLREGISTER: { + auto Op = IROp->C(); + uint32_t SlotOffset = Op->Slot * 16; + switch (OpSize) { + case 1: { + movzx(rax, byte [rsp + SlotOffset]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + case 2: { + movzx(rax, word [rsp + SlotOffset]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + case 4: { + mov(rax, dword [rsp + SlotOffset]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + case 8: { + mov(rax, qword [rsp + SlotOffset]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + case 16: { + movaps(xmm0, xword [rsp + SlotOffset]); + movaps(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + } + break; + default: LogMan::Msg::A("Unhandled FillRegister size: %d", OpSize); + } + break; + } + case IR::OP_SPILLREGISTER: { + auto Op = IROp->C(); + uint32_t SlotOffset = Op->Slot * 16; + switch (OpSize) { + case 1: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(byte [rsp + SlotOffset], al); + } + break; + case 2: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(word [rsp + SlotOffset], ax); + } + break; + case 4: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(dword [rsp + SlotOffset], eax); + } + break; + case 8: { + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(qword [rsp + SlotOffset], rax); + } + break; + case 16: { + movaps(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movaps(xword [rsp + SlotOffset], xmm0); + } + break; + default: LogMan::Msg::A("Unhandled SpillRegister size: %d", OpSize); + } + break; + } + case IR::OP_SYSCALL: { + auto Op = IROp->C(); + + push(rdi); + push(r11); + + // Syscall ABI for x86-64 + // this: rdi + // Thread: rsi + // ArgPointer: rdx (Stack) + // + // Result: RAX + + mov(rsi, rdi); // Move thread in to rsi + mov(rdi, reinterpret_cast(&CTX->SyscallHandler)); + + // These are pushed in reverse order because stacks + push(qword [TEMP_STACK + (Op->Header.Args[6].ID() * 16)]); + push(qword [TEMP_STACK + (Op->Header.Args[5].ID() * 16)]); + push(qword [TEMP_STACK + (Op->Header.Args[4].ID() * 16)]); + push(qword [TEMP_STACK + (Op->Header.Args[3].ID() * 16)]); + push(qword [TEMP_STACK + (Op->Header.Args[2].ID() * 16)]); + push(qword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + push(qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov (rdx, rsp); + + using PtrType = uint64_t (FEXCore::SyscallHandler::*)(FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args); + union { + PtrType ptr; + uint64_t Raw; + } PtrCast; + PtrCast.ptr = &FEXCore::SyscallHandler::HandleSyscall; + mov(rax, PtrCast.Raw); + call(rax); + add(rsp, 7 * 8); + + pop(r11); + pop(rdi); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_CPUID: { + auto Op = IROp->C(); + using ClassPtrType = FEXCore::CPUIDEmu::FunctionResults (FEXCore::CPUIDEmu::*)(uint32_t Function); + union { + ClassPtrType ClassPtr; + uint64_t Raw; + } Ptr; + Ptr.ClassPtr = &CPUIDEmu::RunFunction; + + // CPUID ABI + // this: rdi + // Function: rsi + // + // Result: RAX, RDX. 4xi32 + push(rdi); + push(r11); + mov (rsi, qword [TEMP_STACK + (Op->Header.Args[0].ID() *16)]); + mov (rdi, reinterpret_cast(&CTX->CPUID)); + + push(rax); // align + + mov(rax, Ptr.Raw); + call(rax); + + pop(r11); // align + + pop(r11); + pop(rdi); + + mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 0], eax); + shr(rax, 32); + mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 4], eax); + + mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 8], edx); + shr(rdx, 32); + mov(dword [TEMP_STACK + (WrapperOp->ID() * 16) + 12], edx); + break; + } + case IR::OP_EXTRACTELEMENT: { + auto Op = IROp->C(); + + uint32_t Offset = Op->Header.Args[0].ID() * 16 + OpSize * Op->Idx; + switch (OpSize) { + case 1: + movzx(rax, byte [TEMP_STACK + Offset]); + break; + case 2: + movzx(rax, word [TEMP_STACK + Offset]); + break; + case 4: + mov(eax, dword [TEMP_STACK + Offset]); + break; + case 8: + mov(rax, qword [TEMP_STACK + Offset]); + break; + default: LogMan::Msg::A("Unhandled ExtractElementSize: %d", OpSize); + } + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_LOADFLAG: { + auto Op = IROp->C(); + + movzx(rax, byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)]); + and(rax, 1); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_STOREFLAG: { + auto Op = IROp->C(); + + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + and(rax, 1); + mov(byte [STATE + (offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag)], al); + break; + } + case IR::OP_CONDJUMP: { + auto Op = IROp->C(); + + Label *TargetLabel; + auto IsTarget = JumpTargets.find(Op->Header.Args[1].ID()); + if (IsTarget == JumpTargets.end()) { + TargetLabel = &JumpTargets.try_emplace(Op->Header.Args[1].ID()).first->second; + } + else { + TargetLabel = &IsTarget->second; + } + + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + cmp(rax, 0); + jne(*TargetLabel, T_NEAR); + break; + } + case IR::OP_LOADMEM: { + auto Op = IROp->C(); + uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); + + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(rcx, Memory); + add(rax, rcx); + switch (Op->Size) { + case 1: { + movzx (rcx, byte [rax]); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); + break; + } + case 2: { + movzx (rcx, word [rax]); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); + break; + } + case 4: { + mov(ecx, dword [rax]); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); + break; + } + case 8: { + mov(rcx, qword [rax]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); + break; + } + case 16: { + movups(xmm0, xword [rax]); + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + if (MemoryDebug) { + movq(rcx, xmm0); + } + break; + } + default: LogMan::Msg::A("Unhandled LoadMem size: %d", Op->Size); + } + + if (MemoryDebug) { + push(rdi); + push(r11); + sub(rsp, 8); + + // Load the address in to Arg1 + mov(rdi, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + // Move the loaded value to Arg2 + mov(rsi, rcx); + mov (rdx, Op->Size); + + mov(rax, reinterpret_cast(LoadMem)); + call(rax); + + add(rsp, 8); + + pop(r11); + pop(rdi); + } + + break; + } + case IR::OP_STOREMEM: { + auto Op = IROp->C(); + uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); + + mov(rax, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + mov(rcx, Memory); + add(rax, rcx); + switch (Op->Size) { + case 1: { + mov(cl, byte [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + mov(byte [rax], cl); + if (MemoryDebug) { + movzx(rcx, cl); + } + break; + } + case 2: { + mov(cx, word [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + mov(word [rax], cx); + + if (MemoryDebug) { + movzx(rcx, cx); + } + break; + } + case 4: { + mov(ecx, dword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + mov(dword [rax], ecx); + break; + } + case 8: { + mov(rcx, qword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + mov(qword [rax], rcx); + break; + } + case 16: { + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + movups(xword [rax], xmm0); + if (MemoryDebug) { + movq(rcx, xmm0); + } + break; + } + default: LogMan::Msg::A("Unhandled StoreMem size: %d", Op->Size); + } + + if (MemoryDebug) { + push(rdi); + push(r11); + sub(rsp, 8); + + // Load the address in to Arg1 + mov(rdi, qword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + // Load the value from RAX in to Arg2 + mov(rsi, rcx); + + mov (rdx, Op->Size); + + mov(rax, reinterpret_cast(StoreMem)); + call(rax); + + add(rsp, 8); + + pop(r11); + pop(rdi); + } + break; + } + case IR::OP_MOV: { + auto Op = IROp->C(); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_CONSTANT: { + auto Op = IROp->C(); + if (Op->Constant >> 31) { + mov(rax, Op->Constant); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + else { + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], Op->Constant); + } + break; + } + case IR::OP_POPCOUNT: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movzx(al, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + popcnt(eax, eax); + break; + case 2: + popcnt(ax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movzx(rax, ax); + break; + case 4: + popcnt(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + break; + case 8: + popcnt(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + break; + } + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_ADD: { + auto Op = IROp->C(); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + add(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_SUB: { + auto Op = IROp->C(); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + sub(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_XOR: { + auto Op = IROp->C(); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + xor(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_AND: { + auto Op = IROp->C(); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + and(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_OR: { + auto Op = IROp->C(); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + or(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_MUL: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movsx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + imul(cl); + movsx(rax, al); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 2: + movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movsx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + imul(cx); + movsx(rax, ax); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 4: + movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + imul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + movsxd(rax, eax); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 8: + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + imul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_UMUL: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movzx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mul(cl); + movzx(rax, al); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 2: + movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movzx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mul(cx); + movzx(rax, ax); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 4: + mov(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 8: + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_MULH: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movsx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + imul(cl); + movsx(rax, ax); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 2: + movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movsx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + imul(cx); + movsx(rax, dx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 4: + movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + imul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + movsxd(rax, edx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 8: + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + imul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_UMULH: { + auto Op = IROp->C(); + switch (OpSize) { + case 1: + movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movzx(rcx, byte [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mul(cl); + movzx(rax, ax); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 2: + movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movzx(rcx, word [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mul(cx); + movzx(rax, dx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 4: + mov(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mul(dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + case 8: + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mul(qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", OpSize); + } + break; + } + case IR::OP_LDIV: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + idiv(ecx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case 8: { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + idiv(rcx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + default: LogMan::Msg::A("Unknown LDIV Size: %d", Size); break; + } + break; + } + case IR::OP_LREM: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + idiv(ecx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + } + case 8: { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + idiv(rcx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + } + default: LogMan::Msg::A("Unknown LREM Size: %d", Size); break; + } + break; + } + case IR::OP_LUDIV: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + div(ecx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case 8: { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + div(rcx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; + } + break; + } + case IR::OP_LUREM: { + auto Op = IROp->C(); + // Each source is OpSize in size + // So you can have up to a 128bit divide from x86-64 + auto Size = OpSize; + switch (Size) { + case 4: { + mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(edx, dword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(ecx, dword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + div(ecx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + } + case 8: { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + div(rcx); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rdx); + break; + } + default: LogMan::Msg::A("Unknown LUDIV Size: %d", Size); break; + } + break; + } + case IR::OP_ZEXT: { + auto Op = IROp->C(); + LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); + + if (Op->SrcSize == 64) { + movd(xmm0, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + movups(qword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + } + else { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(rcx, uint64_t((1ULL << Op->SrcSize) - 1)); + and(rax, rcx); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + } + case IR::OP_SEXT: { + auto Op = IROp->C(); + LogMan::Throw::A(Op->SrcSize <= 64, "Can't support Zext of size: %ld", Op->SrcSize); + switch (Op->SrcSize / 8) { + case 1: + movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 2: + movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 4: + movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + case 8: + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + default: LogMan::Msg::A("Unknown Sext size: %d", Op->SrcSize / 8); + } + break; + } + case IR::OP_BFI: { + auto Op = IROp->C(); + LogMan::Throw::A(OpSize <= 8, "OpSize is too large for BFI: %d", OpSize); + + uint64_t SourceMask = (1ULL << Op->Width) - 1; + + uint64_t DestMask = ~(SourceMask << Op->lsb); + + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + + if (Op->Width != 64) { + mov(rdx, SourceMask); + and(rcx, rdx); + } + + mov(rdx, DestMask); + and(rax, rdx); + shl(rdx, Op->lsb); + or(rax, rdx); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_BFE: { + auto Op = IROp->C(); + LogMan::Throw::A(OpSize <= 16, "OpSize is too large for BFE: %d", OpSize); + // %ssa64 i128 = Bfe %ssa48 i128, 0x1, 0x7 + if (OpSize == 16) { + LogMan::Throw::A(!(Op->lsb < 64 && (Op->lsb + Op->Width > 64)), "Trying to BFE an XMM across the 64bit split: Beginning at %d, ending at %d", Op->lsb, Op->lsb + Op->Width); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + uint8_t Offset = Op->lsb; + if (Offset < 64) { + pextrq(rax, xmm0, 0); + } + else { + pextrq(rax, xmm0, 1); + Offset -= 64; + } + + if (Offset) { + shr(rax, Offset); + } + + if (Op->Width != 64) { + mov(rcx, uint64_t((1ULL << Op->Width) - 1)); + and(rax, rcx); + } + + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + else { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + if (Op->lsb != 0) + shr(rax, Op->lsb); + + if (Op->Width != 64) { + mov(rcx, uint64_t((1ULL << Op->Width) - 1)); + and(rax, rcx); + } + + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + } + break; + } + case IR::OP_FINDLSB: { + auto Op = IROp->C(); + tzcnt(rcx, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + xor(rax, rax); + cmp(qword [TEMP_STACK + Op->Header.Args[0].ID() * 16], 1); + sbb(rax, rax); + or(rax, rcx); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_FINDMSB: { + auto Op = IROp->C(); + mov(rax, OpSize * 8); + lzcnt(rcx, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + sub(rax, rcx); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_LSHR: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + and(rcx, Mask); + shrx(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16], rcx); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_LSHL: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + and(rcx, Mask); + shlx(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16], rcx); + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_ASHR: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + and(rcx, Mask); + switch (OpSize) { + case 1: + movsx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + sar(al, cl); + break; + case 2: + movsx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + sar(ax, cl); + break; + case 4: + movsxd(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + sar(eax, cl); + break; + case 8: + mov(rax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + sar(rax, cl); + break; + default: LogMan::Msg::A("Unknown ASHR Size: %d\n", OpSize); break; + }; + + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_ROL: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + and(rcx, Mask); + switch (OpSize) { + case 1: { + movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + rol(al, cl); + break; + } + case 2: { + movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + rol(ax, cl); + break; + } + case 4: { + mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + rol(eax, cl); + break; + } + case 8: { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + rol(rax, cl); + break; + } + } + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_ROR: { + auto Op = IROp->C(); + uint8_t Mask = OpSize * 8 - 1; + + mov(rcx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + and(rcx, Mask); + switch (OpSize) { + case 1: { + movzx(rax, byte [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + ror(al, cl); + break; + } + case 2: { + movzx(rax, word [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + ror(ax, cl); + break; + } + case 4: { + mov(eax, dword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + ror(eax, cl); + break; + } + case 8: { + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + ror(rax, cl); + break; + } + } + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_SELECT: { + auto Op = IROp->C(); + + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + cmp(rax, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + + switch (Op->Cond) { + case FEXCore::IR::COND_EQ: + mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); + cmove(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + break; + case FEXCore::IR::COND_NEQ: + mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); + cmovne(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + break; + case FEXCore::IR::COND_GE: + mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); + cmovge(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + break; + case FEXCore::IR::COND_LT: + mov(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + cmovae(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); + break; + case FEXCore::IR::COND_GT: + mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); + cmovg(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + break; + case FEXCore::IR::COND_LE: + mov(rcx, qword [TEMP_STACK + Op->Header.Args[3].ID() * 16]); + cmovle(rcx, qword [TEMP_STACK + Op->Header.Args[2].ID() * 16]); + break; + case FEXCore::IR::COND_CS: + case FEXCore::IR::COND_CC: + case FEXCore::IR::COND_MI: + case FEXCore::IR::COND_PL: + case FEXCore::IR::COND_VS: + case FEXCore::IR::COND_VC: + case FEXCore::IR::COND_HI: + case FEXCore::IR::COND_LS: + default: + LogMan::Msg::A("Unsupported compare type"); + break; + } + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rcx); + break; + } + case IR::OP_CAS: { + auto Op = IROp->C(); + // Args[0]: Expected + // Args[1]: Desired + // Args[2]: Pointer + // DataSrc = *Src1 + // if (DataSrc == Src3) { *Src1 == Src2; } Src2 = DataSrc + // This will write to memory! Careful! + // Third operand must be a calculated guest memory address + //OrderedNode *CASResult = _CAS(Src3, Src2, Src1); + uint64_t Memory = CTX->MemoryMapper.GetBaseOffset(0); + + mov(rcx, Memory); + add(rcx, qword [TEMP_STACK + (Op->Header.Args[2].ID() * 16)]); + + mov(rdx, qword [TEMP_STACK + Op->Header.Args[1].ID() * 16]); + mov(rax, qword [TEMP_STACK + Op->Header.Args[0].ID() * 16]); + + // RCX now contains pointer + // RAX contains our expected value + // RDX contains our desired + + lock(); + + switch (OpSize) { + case 1: { + cmpxchg(byte [rcx], dl); + movzx(rax, al); + break; + } + case 2: { + cmpxchg(word [rcx], dx); + movzx(rax, ax); + break; + } + case 4: { + cmpxchg(dword [rcx], edx); + break; + } + case 8: { + cmpxchg(qword [rcx], rdx); + break; + } + default: LogMan::Msg::A("Unsupported: %d", OpSize); + } + + // RAX now contains the result + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_VCMPEQ: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + + LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); + + switch (Op->ElementSize) { + case 1: + pcmpeqb(xmm0, xmm1); + break; + case 2: + pcmpeqw(xmm0, xmm1); + break; + case 4: + pcmpeqd(xmm0, xmm1); + break; + case 8: + pcmpeqq(xmm0, xmm1); + break; + default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VCMPGT: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + + LogMan::Throw::A(Op->RegisterSize == 16, "Can't handle register size of: %d", Op->RegisterSize); + + switch (Op->ElementSize) { + case 1: + pcmpgtb(xmm0, xmm1); + case 2: + pcmpgtw(xmm0, xmm1); + case 4: + pcmpgtd(xmm0, xmm1); + case 8: + pcmpgtq(xmm0, xmm1); + default: LogMan::Msg::A("Unsupported elementSize: %d", Op->ElementSize); + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VXOR: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + pxor(xmm0, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VOR: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + por(xmm0, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VINSELEMENT: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + + // Dst_d[Op->DestIdx] = Src2_d[Op->SrcIdx]; + + // pextrq reg64/mem64, xmm, imm + // pinsrq xmm, reg64/mem64, imm8 + switch (Op->ElementSize) { + case 1: + pextrb(al, xmm1, Op->SrcIdx); + pinsrb(xmm0, al, Op->DestIdx); + break; + case 2: + pextrw(ax, xmm1, Op->SrcIdx); + pinsrw(xmm0, ax, Op->DestIdx); + break; + case 4: + pextrd(eax, xmm1, Op->SrcIdx); + pinsrd(xmm0, eax, Op->DestIdx); + break; + case 8: + pextrq(rax, xmm1, Op->SrcIdx); + pinsrq(xmm0, rax, Op->DestIdx); + break; + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VADD: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + switch (Op->ElementSize) { + case 1: + paddb(xmm0, xmm1); + break; + case 2: + paddw(xmm0, xmm1); + break; + case 4: + paddd(xmm0, xmm1); + break; + case 8: + paddq(xmm0, xmm1); + break; + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VSUB: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + switch (Op->ElementSize) { + case 1: + psubb(xmm0, xmm1); + break; + case 2: + psubw(xmm0, xmm1); + break; + case 4: + psubd(xmm0, xmm1); + break; + case 8: + psubq(xmm0, xmm1); + break; + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VUSHLS: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + + switch (Op->ElementSize) { + case 2: + psllw(xmm0, xmm1); + break; + case 4: + pslld(xmm0, xmm1); + break; + case 8: + psllq(xmm0, xmm1); + break; + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VZIP: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + switch (Op->ElementSize) { + case 1: + punpcklbw(xmm0, xmm1); + break; + case 2: + punpcklwd(xmm0, xmm1); + break; + case 4: + punpckldq(xmm0, xmm1); + break; + case 8: + punpcklqdq(xmm0, xmm1); + break; + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VZIP2: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + switch (Op->ElementSize) { + case 1: + punpckhbw(xmm0, xmm1); + break; + case 2: + punpckhwd(xmm0, xmm1); + break; + case 4: + punpckhdq(xmm0, xmm1); + break; + case 8: + punpckhqdq(xmm0, xmm1); + break; + default: LogMan::Msg::A("Unknown Element Size: %d", Op->ElementSize); break; + } + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_VEXTR: { + auto Op = IROp->C(); + movups(xmm0, xword [TEMP_STACK + (Op->Header.Args[0].ID() * 16)]); + movups(xmm1, xword [TEMP_STACK + (Op->Header.Args[1].ID() * 16)]); + palignr(xmm0, xmm1, Op->Index); + movups(xword [TEMP_STACK + (WrapperOp->ID() * 16)], xmm0); + break; + } + case IR::OP_CYCLECOUNTER: { + #ifdef DEBUG_CYCLES + mov (rax, 0); + #else + rdtsc(); + shl(rdx, 32); + or(rax, rdx); + #endif + mov (qword [TEMP_STACK + (WrapperOp->ID() * 16)], rax); + break; + } + case IR::OP_CODEBLOCK: + case IR::OP_IRHEADER: + case IR::OP_BEGINBLOCK: + case IR::OP_ENDBLOCK: + case IR::OP_EXITFUNCTION: + case IR::OP_ENDFUNCTION: + case IR::OP_BREAK: + case IR::OP_JUMP: + break; + default: + LogMan::Msg::A("Unknown IR Op: %d(%s)", IROp->Op, FEXCore::IR::GetName(IROp->Op).data()); + break; + } + } + + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; + } + + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + } } ready(); -// LogMan::Msg::D("Ptr: %p,+%ld", Entry, getCurr() - (uintptr_t)Entry); -// static int a = 0; -// if (a++ > 100) -// __builtin_trap(); + return Entry; } +void JITCore::CreateCustomDispatch() { +// Temp registers +// rax, rcx, rdx, rsi, r8, r9, +// r10, r11 +// +// Callee Saved +// rbx, rbp, r12, r13, r14, r15 +// +// 1St Argument: rdi +// XMM: +// All temp +// r11 assigned to temp state + void *Entry = getCurr(); + + // while (!Thread->State.RunningEvents.ShouldStop.load()) { + // Ptr = FindBlock(RIP) + // if (!Ptr) + // Ptr = CTX->CompileBlock(RIP); + // + // if (Ptr) + // Ptr(); + // else + // { + // Ptr = FallbackCore->CompileBlock() + // if (Ptr) + // Ptr() + // else { + // ShouldStop = true; + // } + // } + // } + // Bunch of exit state stuff + ready(); + // CustomDispatchGenerated = true; +} + FEXCore::CPU::CPUBackend *CreateJITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) { return new JITCore(ctx); } diff --git a/Source/Interface/Core/LLVMJIT/LLVMCore.cpp b/Source/Interface/Core/LLVMJIT/LLVMCore.cpp index fe42562e9..976531622 100644 --- a/Source/Interface/Core/LLVMJIT/LLVMCore.cpp +++ b/Source/Interface/Core/LLVMJIT/LLVMCore.cpp @@ -21,7 +21,7 @@ #include #include -#define DESTMAP_AS_MAP 0 +#define DESTMAP_AS_MAP 1 #if DESTMAP_AS_MAP using DestMapType = std::unordered_map; #else @@ -185,18 +185,18 @@ private: llvm::Value *CastVectorToType(llvm::Value *Arg, bool Integer, uint8_t RegisterSize, uint8_t ElementSize); llvm::Value *CastToOpaqueStructure(llvm::Value *Arg, llvm::Type *DstType); - void SetDest(IR::NodeWrapper Op, llvm::Value *Val); - llvm::Value *GetSrc(IR::NodeWrapper Src); + void SetDest(IR::OrderedNodeWrapper Op, llvm::Value *Val); + llvm::Value *GetSrc(IR::OrderedNodeWrapper Src); DestMapType DestMap; FEXCore::IR::IRListView const *CurrentIR; - std::unordered_map JumpTargets; - std::unordered_map ForwardJumpTargets; + std::unordered_map JumpTargets; + std::unordered_map ForwardJumpTargets; // Target Machines const std::string arch = "x86-64"; - const std::string cpu = "znver2"; + const std::string cpu = "skylake"; const llvm::Triple TargetTriple{"x86_64", "unknown", "linux", "gnu"}; const llvm::SmallVector Attrs; llvm::TargetMachine *LLVMTarget; @@ -754,16 +754,16 @@ llvm::Value *LLVMJITCore::CastToOpaqueStructure(llvm::Value *Arg, llvm::Type *Ds return JITState.IRBuilder->CreateZExtOrTrunc(Arg, DstType->getPointerElementType()); } -void LLVMJITCore::SetDest(IR::NodeWrapper Op, llvm::Value *Val) { - DestMap[Op.NodeOffset] = Val; +void LLVMJITCore::SetDest(IR::OrderedNodeWrapper Op, llvm::Value *Val) { + DestMap[Op.ID()] = Val; } -llvm::Value *LLVMJITCore::GetSrc(IR::NodeWrapper Src) { +llvm::Value *LLVMJITCore::GetSrc(IR::OrderedNodeWrapper Src) { #if DESTMAP_AS_MAP - LogMan::Throw::A(DestMap.find(Src.NodeOffset) != DestMap.end(), "Op had Src but wasn't added to the dest map"); + LogMan::Throw::A(DestMap.find(Src.ID()) != DestMap.end(), "Op had Src but wasn't added to the dest map"); #endif - auto DstPtr = DestMap[Src.NodeOffset]; + auto DstPtr = DestMap[Src.ID()]; LogMan::Throw::A(DstPtr != nullptr, "Destmap had slot but wasn't allocated memory"); return DstPtr; } @@ -771,11 +771,11 @@ llvm::Value *LLVMJITCore::GetSrc(IR::NodeWrapper Src) { void LLVMJITCore::HandleIR(FEXCore::IR::IRListView const *IR, IR::NodeWrapperIterator *Node) { using namespace llvm; - uintptr_t ListBegin = CurrentIR->GetListData(); - uintptr_t DataBegin = CurrentIR->GetData(); + uintptr_t ListBegin = CurrentIR->GetListData(); + uintptr_t DataBegin = CurrentIR->GetData(); - IR::NodeWrapper *WrapperOp = (*Node)(); - IR::OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + IR::OrderedNodeWrapper *WrapperOp = (*Node)(); + IR::OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); uint8_t OpSize = IROp->Size; @@ -1812,7 +1812,6 @@ void LLVMJITCore::HandleIR(FEXCore::IR::IRListView const *IR, IR::NodeWrap LogMan::Msg::A("Unknown IR Op: %d(%s)", IROp->Op, FEXCore::IR::GetName(IROp->Op).data()); break; } - ++*Node; } void* FEXCore::CPU::LLVMJITCore::CompileCode(FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData) { @@ -1855,35 +1854,66 @@ void* FEXCore::CPU::LLVMJITCore::CompileCode(FEXCore::IR::IRListView const FunctionModule); Func->setCallingConv(CallingConv::C); - { - auto Entry = BasicBlock::Create(*Con, "Entry", Func); - JITCurrentState.Blocks.emplace_back(Entry); - JITState.IRBuilder->SetInsertPoint(Entry); - JITCurrentState.CurrentBlock = Entry; + { + auto Entry = BasicBlock::Create(*Con, "Entry", Func); + JITCurrentState.Blocks.emplace_back(Entry); + JITState.IRBuilder->SetInsertPoint(Entry); + JITCurrentState.CurrentBlock = Entry; - CreateGlobalVariables(Engine, FunctionModule); + CreateGlobalVariables(Engine, FunctionModule); - auto Builder = JITState.IRBuilder; + auto Builder = JITState.IRBuilder; - // Let's create the exit block quick - JITCurrentState.ExitBlock = BasicBlock::Create(*Con, "ExitBlock", Func); - JITCurrentState.Blocks.emplace_back(JITCurrentState.ExitBlock); + // Let's create the exit block quick + JITCurrentState.ExitBlock = BasicBlock::Create(*Con, "ExitBlock", Func); + JITCurrentState.Blocks.emplace_back(JITCurrentState.ExitBlock); - JITState.IRBuilder->SetInsertPoint(JITCurrentState.ExitBlock); - Builder->CreateRetVoid(); + JITState.IRBuilder->SetInsertPoint(JITCurrentState.ExitBlock); + Builder->CreateRetVoid(); - JITState.IRBuilder->SetInsertPoint(Entry); - JITCurrentState.CurrentBlock = Entry; - JITCurrentState.Blocks.emplace_back(JITCurrentState.CurrentBlock); - JITCurrentState.CurrentBlockHasTerm = false; + JITState.IRBuilder->SetInsertPoint(Entry); + JITCurrentState.CurrentBlock = Entry; + JITCurrentState.Blocks.emplace_back(JITCurrentState.CurrentBlock); + JITCurrentState.CurrentBlockHasTerm = false; - IR::NodeWrapperIterator Begin = CurrentIR->begin(); - IR::NodeWrapperIterator End = CurrentIR->end(); + uintptr_t ListBegin = CurrentIR->GetListData(); + uintptr_t DataBegin = CurrentIR->GetData(); - while (Begin != End) { - HandleIR(CurrentIR, &Begin); - } - } + auto HeaderIterator = CurrentIR->begin(); + IR::OrderedNodeWrapper *HeaderNodeWrapper = HeaderIterator(); + IR::OrderedNode *HeaderNode = HeaderNodeWrapper->GetNode(ListBegin); + auto HeaderOp = HeaderNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == IR::OP_IRHEADER, "First op wasn't IRHeader"); + + IR::OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); + + while (1) { + using namespace FEXCore::IR; + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block"); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR->at(BlockIROp->Begin); + auto CodeLast = CurrentIR->at(BlockIROp->Last); + + while (1) { + HandleIR(CurrentIR, &CodeBegin); + + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; + + } + + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + } + } + } for (auto &Block : JITCurrentState.Blocks) { // If the block is empty then let is just jump to the exit block @@ -1904,8 +1934,7 @@ void* FEXCore::CPU::LLVMJITCore::CompileCode(FEXCore::IR::IRListView const raw_ostream &Out = outs(); - //if (CTX->Config.LLVM_PrinterPass) - if (ThreadState->State.State.rip == 0x4021b0) + // if (CTX->Config.LLVM_PrinterPass) { FPM.addPass(PrintModulePass(Out)); } diff --git a/Source/Interface/Core/OpcodeDispatcher.cpp b/Source/Interface/Core/OpcodeDispatcher.cpp index 78258300d..e7acac8aa 100644 --- a/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/Source/Interface/Core/OpcodeDispatcher.cpp @@ -94,6 +94,7 @@ void OpDispatchBuilder::RETOp(OpcodeArgs) { // Store the new RIP _StoreContext(8, offsetof(FEXCore::Core::CPUState, rip), NewRIP); _EndFunction(); + CreateNewEndBlock(0); Information.HadUnconditionalExit = true; } @@ -279,11 +280,7 @@ void OpDispatchBuilder::CALLOp(OpcodeArgs) { _StoreContext(8, offsetof(FEXCore::Core::CPUState, rip), NewRIP); _ExitFunction(); // If we get here then leave the function now - // Fracking RIPSetter check ending the block causes issues - // Split the block and leave early to work around the bug - _EndBlock(0); - // Make sure to start a new block after ending this one - _BeginBlock(); + CreateNewEndBlock(0); Information.HadUnconditionalExit = true; } @@ -306,13 +303,8 @@ void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) { _StoreContext(8, offsetof(FEXCore::Core::CPUState, rip), JMPPCOffset); _ExitFunction(); // If we get here then leave the function now - // Fracking RIPSetter check ending the block causes issues - // Split the block and leave early to work around the bug - _EndBlock(0); - // Make sure to start a new block after ending this one - _BeginBlock(); + CreateNewEndBlock(0); Information.HadUnconditionalExit = true; - } void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) { @@ -514,9 +506,6 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) { #endif // Fallback { - // XXX: Test - GetPackedRFLAG(false); - auto CondJump = _CondJump(SrcCond); auto RIPOffset = LoadSource(Op, Op->Src1, Op->Flags); @@ -528,10 +517,10 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) { _StoreContext(8, offsetof(FEXCore::Core::CPUState, rip), NewRIP); _ExitFunction(); - _EndBlock(0); + CreateNewEndBlock(0); // Make sure to start a new block after ending this one - auto JumpTarget = _BeginBlock(); + auto JumpTarget = CreateNewBeginBlock(); // This very explicitly avoids the isDest path for Ops. We want the actual destination here SetJumpTarget(CondJump, JumpTarget); } @@ -580,10 +569,7 @@ void OpDispatchBuilder::JUMPOp(OpcodeArgs) { _StoreContext(8, offsetof(FEXCore::Core::CPUState, rip), NewRIP); _ExitFunction(); - _EndBlock(0); - - // Make sure to start a new block after ending this one - _BeginBlock(); + CreateNewEndBlock(0); Information.HadUnconditionalExit = true; } } @@ -598,12 +584,8 @@ void OpDispatchBuilder::JUMPAbsoluteOp(OpcodeArgs) { _StoreContext(8, offsetof(FEXCore::Core::CPUState, rip), RIPOffset); _ExitFunction(); - _EndBlock(0); - - // Make sure to start a new block after ending this one - _BeginBlock(); + CreateNewEndBlock(0); Information.HadUnconditionalExit = true; - } void OpDispatchBuilder::SETccOp(OpcodeArgs) { @@ -1159,19 +1141,10 @@ void OpDispatchBuilder::SHLOp(OpcodeArgs) { GenerateFlags_Shift(Op, _Bfe(Size, 0, ALUOp), _Bfe(Size, 0, Dest), _Bfe(Size, 0, Src)); } +template void OpDispatchBuilder::SHROp(OpcodeArgs) { - bool SHR1Bit = false; -#define OPD(group, prefix, Reg) (((group - FEXCore::X86Tables::TYPE_GROUP_1) << 6) | (prefix) << 3 | (Reg)) - switch (Op->OP) { - case OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 5): - case OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 5): - SHR1Bit = true; - break; - } -#undef OPD - OrderedNode *Src; - OrderedNode *Dest = LoadSource(Op, Op->Dest, Op->Flags); + auto Dest = LoadSource(Op, Op->Dest, Op->Flags); if (SHR1Bit) { Src = _Constant(1); @@ -1496,9 +1469,7 @@ void OpDispatchBuilder::RDTSCOp(OpcodeArgs) { } void OpDispatchBuilder::INCOp(OpcodeArgs) { - if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX) { - LogMan::Msg::A("Can't handle REP on this\n"); - } + LogMan::Throw::A(!(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX), "Can't handle REP on this\n"); OrderedNode *Dest = LoadSource(Op, Op->Dest, Op->Flags); auto OneConst = _Constant(1); @@ -1511,9 +1482,7 @@ void OpDispatchBuilder::INCOp(OpcodeArgs) { } void OpDispatchBuilder::DECOp(OpcodeArgs) { - if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX) { - LogMan::Msg::A("Can't handle REP on this\n"); - } + LogMan::Throw::A(!(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX), "Can't handle REP on this\n"); OrderedNode *Dest = LoadSource(Op, Op->Dest, Op->Flags); auto OneConst = _Constant(1); @@ -1526,9 +1495,8 @@ void OpDispatchBuilder::DECOp(OpcodeArgs) { } void OpDispatchBuilder::STOSOp(OpcodeArgs) { - if (!(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX)) { - LogMan::Msg::A("Can't handle REP not existing on STOS\n"); - } + LogMan::Throw::A(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX, "Can't handle REP not existing on STOS\n"); + auto Size = GetSrcSize(Op); auto ZeroConst = _Constant(0); @@ -1545,10 +1513,10 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) { OrderedNode *Src = LoadSource(Op, Op->Src1, Op->Flags); auto JumpStart = _Jump(); - _EndBlock(0); + CreateNewEndBlock(0); // Make sure to start a new block after ending this one - auto LoopStart = _BeginBlock(); + auto LoopStart = CreateNewBeginBlock(); SetJumpTarget(JumpStart, LoopStart); OrderedNode *Counter = _LoadContext(8, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RCX])); @@ -1576,23 +1544,18 @@ void OpDispatchBuilder::STOSOp(OpcodeArgs) { // Jump back to the start, we have more work to do _Jump(LoopStart); - _EndBlock(0); + CreateNewEndBlock(0); // Make sure to start a new block after ending this one - auto LoopEnd = _BeginBlock(); + auto LoopEnd = CreateNewBeginBlock(); SetJumpTarget(CondJump, LoopEnd); } void OpDispatchBuilder::MOVSOp(OpcodeArgs) { - if (Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX) { - LogMan::Msg::A("Can't handle REP\n"); - } - + LogMan::Throw::A(!(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX), "Can't handle REP on this\n"); _Break(0, 0); } void OpDispatchBuilder::CMPSOp(OpcodeArgs) { - if (!(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX)) { - LogMan::Msg::A("Can't only handle REP\n"); - } + LogMan::Throw::A(Op->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_REP_PREFIX, "Can't only handle REP\n"); auto Size = GetSrcSize(Op); @@ -1608,9 +1571,9 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) { SizeConst, NegSizeConst); auto JumpStart = _Jump(); - _EndBlock(0); + CreateNewEndBlock(0); // Make sure to start a new block after ending this one - auto LoopStart = _BeginBlock(); + auto LoopStart = CreateNewBeginBlock(); SetJumpTarget(JumpStart, LoopStart); OrderedNode *Counter = _LoadContext(8, offsetof(FEXCore::Core::CPUState, gregs[FEXCore::X86State::REG_RCX])); @@ -1645,9 +1608,9 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) { // Jump back to the start, we have more work to do _Jump(LoopStart); - _EndBlock(0); + CreateNewEndBlock(0); // Make sure to start a new block after ending this one - auto LoopEnd = _BeginBlock(); + auto LoopEnd = CreateNewBeginBlock(); SetJumpTarget(CondJump, LoopEnd); } @@ -2178,12 +2141,43 @@ void OpDispatchBuilder::CMPXCHGOp(OpcodeArgs) { } } -void OpDispatchBuilder::BeginBlock() { - _BeginBlock(); +OpDispatchBuilder::IRPair OpDispatchBuilder::CreateNewBeginBlock() { + auto CodeNode = CreateCodeNode(); + auto BeginBlock = _BeginBlock(); + SetCodeNodeBegin(CodeNode, BeginBlock); + CurrentCodeBlock = CodeNode; + return BeginBlock; } -void OpDispatchBuilder::EndBlock(uint64_t RIPIncrement) { - _EndBlock(RIPIncrement); +OpDispatchBuilder::IRPair OpDispatchBuilder::CreateNewEndBlock(uint64_t RIPIncrement) { + auto EndBlock = _EndBlock(RIPIncrement); + SetCodeNodeLast(CurrentCodeBlock, EndBlock); + return EndBlock; +} + +void OpDispatchBuilder::BeginFunction(uint64_t RIP) { + _IRHeader(RIP, InvalidNode->Wrapped(ListData.Begin()), 0); + CreateNewBeginBlock(); +} + +void OpDispatchBuilder::Finalize() { + // Node 0 is invalid node + OrderedNode *RealNode = reinterpret_cast(GetNode(1)); + FEXCore::IR::IROp_Header *IROp = RealNode->Op(Data.Begin()); + LogMan::Throw::A(IROp->Op == OP_IRHEADER, "First op in function must be our header"); + FEXCore::IR::IROp_IRHeader *Op = IROp->CW(); + Op->BlockCount = CodeBlocks.size(); + + OrderedNode *PrevCodeBlock{}; + for (auto &CodeBlock : CodeBlocks) { + if (PrevCodeBlock) { + LinkCodeBlocks(PrevCodeBlock, CodeBlock); + } + PrevCodeBlock = CodeBlock; + } + + Op->Blocks = CodeBlocks[0]->Wrapped(ListData.Begin()); + CodeBlocks.clear(); } void OpDispatchBuilder::ExitFunction() { @@ -2437,7 +2431,7 @@ void OpDispatchBuilder::StoreResult(FEXCore::X86Tables::DecodedOp Op, OrderedNod void OpDispatchBuilder::TestFunction() { printf("Doing Test Function\n"); - _BeginBlock(); + CreateNewBeginBlock(); auto Load1 = _LoadContext(8, 0); auto Load2 = _LoadContext(8, 0); //auto Res = Load1 Load2; @@ -2461,12 +2455,14 @@ OpDispatchBuilder::OpDispatchBuilder() void OpDispatchBuilder::ResetWorkingList() { Data.Reset(); ListData.Reset(); + CodeBlocks.clear(); CurrentWriteCursor = nullptr; // This is necessary since we do "null" pointer checks InvalidNode = reinterpret_cast(ListData.Allocate(sizeof(OrderedNode))); DecodeFailure = false; Information.HadUnconditionalExit = false; ShouldDump = false; + CurrentCodeBlock = nullptr; } template @@ -2529,9 +2525,6 @@ OrderedNode *OpDispatchBuilder::GetPackedRFLAG(bool Lower8) { } void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - auto Size = GetSrcSize(Op) * 8; // AF { @@ -2542,9 +2535,9 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde // SF { - auto ThirtyOneConst = _Constant(Size - 1); + auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); - auto LshrOp = _Lshr(Res, ThirtyOneConst); + auto LshrOp = _Lshr(Res, SignBitConst); SetRFLAG(LshrOp); } @@ -2552,7 +2545,7 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde { auto PopCountOp = _Popcount(_And(Res, _Constant(0xFF))); - auto XorOp = _Xor(PopCountOp, OneConst); + auto XorOp = _Xor(PopCountOp, _Constant(1)); SetRFLAG(XorOp); } @@ -2561,7 +2554,7 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde auto Dst8 = _Bfe(Size, 0, Res); auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Dst8, ZeroConst, OneConst, ZeroConst); + Dst8, _Constant(0), _Constant(1), _Constant(0)); SetRFLAG(SelectOp); } @@ -2571,9 +2564,9 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde auto Dst8 = _Bfe(Size, 0, Res); auto Src8 = _Bfe(Size, 0, Src2); - auto SelectOpLT = _Select(FEXCore::IR::COND_LT, Dst8, Src8, OneConst, ZeroConst); - auto SelectOpLE = _Select(FEXCore::IR::COND_LE, Dst8, Src8, OneConst, ZeroConst); - auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, OneConst, SelectOpLE, SelectOpLT); + auto SelectOpLT = _Select(FEXCore::IR::COND_LT, Dst8, Src8, _Constant(1), _Constant(0)); + auto SelectOpLE = _Select(FEXCore::IR::COND_LE, Dst8, Src8, _Constant(1), _Constant(0)); + auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, _Constant(1), SelectOpLE, SelectOpLT); SetRFLAG(SelectCF); } @@ -2605,9 +2598,6 @@ void OpDispatchBuilder::GenerateFlags_ADC(FEXCore::X86Tables::DecodedOp Op, Orde } void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2, OrderedNode *CF) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - // AF { OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); @@ -2617,9 +2607,9 @@ void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, Orde // SF { - auto ThirtyOneConst = _Constant(GetSrcSize(Op) * 8 - 1); + auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); - auto LshrOp = _Lshr(Res, ThirtyOneConst); + auto LshrOp = _Lshr(Res, SignBitConst); SetRFLAG(LshrOp); } @@ -2627,14 +2617,14 @@ void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, Orde { auto PopCountOp = _Popcount(_And(Res, _Constant(0xFF))); - auto XorOp = _Xor(PopCountOp, OneConst); + auto XorOp = _Xor(PopCountOp, _Constant(1)); SetRFLAG(XorOp); } // ZF { auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, ZeroConst, OneConst, ZeroConst); + Res, _Constant(0), _Constant(1), _Constant(0)); SetRFLAG(SelectOp); } @@ -2644,9 +2634,9 @@ void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, Orde auto Dst8 = _Bfe(GetSrcSize(Op) * 8, 0, Res); auto Src8_1 = _Bfe(GetSrcSize(Op) * 8, 0, Src1); - auto SelectOpLT = _Select(FEXCore::IR::COND_GT, Dst8, Src8_1, OneConst, ZeroConst); - auto SelectOpLE = _Select(FEXCore::IR::COND_GE, Dst8, Src8_1, OneConst, ZeroConst); - auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, OneConst, SelectOpLE, SelectOpLT); + auto SelectOpLT = _Select(FEXCore::IR::COND_GT, Dst8, Src8_1, _Constant(1), _Constant(0)); + auto SelectOpLE = _Select(FEXCore::IR::COND_GE, Dst8, Src8_1, _Constant(1), _Constant(0)); + auto SelectCF = _Select(FEXCore::IR::COND_EQ, CF, _Constant(1), SelectOpLE, SelectOpLT); SetRFLAG(SelectCF); } @@ -2677,8 +2667,6 @@ void OpDispatchBuilder::GenerateFlags_SBB(FEXCore::X86Tables::DecodedOp Op, Orde } void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); // AF { OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); @@ -2698,12 +2686,14 @@ void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, Orde { auto EightBitMask = _Constant(0xFF); auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, OneConst); + auto XorOp = _Xor(PopCountOp, _Constant(1)); SetRFLAG(XorOp); } // ZF { + auto ZeroConst = _Constant(0); + auto OneConst = _Constant(1); auto Bfe8 = _Bfe(GetSrcSize(Op) * 8, 0, Res); auto SelectOp = _Select(FEXCore::IR::COND_EQ, Bfe8, ZeroConst, OneConst, ZeroConst); @@ -2712,6 +2702,9 @@ void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, Orde // CF { + auto ZeroConst = _Constant(0); + auto OneConst = _Constant(1); + auto SelectOp = _Select(FEXCore::IR::COND_LT, Src1, Src2, OneConst, ZeroConst); @@ -2743,9 +2736,6 @@ void OpDispatchBuilder::GenerateFlags_SUB(FEXCore::X86Tables::DecodedOp Op, Orde } void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - // AF { OrderedNode *AFRes = _Xor(_Xor(Src1, Src2), Res); @@ -2765,14 +2755,14 @@ void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, Orde { auto EightBitMask = _Constant(0xFF); auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, OneConst); + auto XorOp = _Xor(PopCountOp, _Constant(1)); SetRFLAG(XorOp); } // ZF { auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, ZeroConst, OneConst, ZeroConst); + Res, _Constant(0), _Constant(1), _Constant(0)); SetRFLAG(SelectOp); } // CF @@ -2780,7 +2770,7 @@ void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, Orde auto Dst8 = _Bfe(GetSrcSize(Op) * 8, 0, Res); auto Src8 = _Bfe(GetSrcSize(Op) * 8, 0, Src2); - auto SelectOp = _Select(FEXCore::IR::COND_LT, Dst8, Src8, OneConst, ZeroConst); + auto SelectOp = _Select(FEXCore::IR::COND_LT, Dst8, Src8, _Constant(1), _Constant(0)); SetRFLAG(SelectOp); } @@ -2813,17 +2803,15 @@ void OpDispatchBuilder::GenerateFlags_ADD(FEXCore::X86Tables::DecodedOp Op, Orde } void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *High) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); auto SignBitConst = _Constant(GetSrcSize(Op) * 8 - 1); // PF/AF/ZF/SF // Undefined { - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); } // CF/OF @@ -2833,7 +2821,7 @@ void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, Orde auto SignBit = _Ashr(Res, SignBitConst); - auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, SignBit, ZeroConst, OneConst); + auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, SignBit, _Constant(0), _Constant(1)); SetRFLAG(SelectOp); SetRFLAG(SelectOp); @@ -2841,16 +2829,13 @@ void OpDispatchBuilder::GenerateFlags_MUL(FEXCore::X86Tables::DecodedOp Op, Orde } void OpDispatchBuilder::GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, OrderedNode *High) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - // AF/SF/PF/ZF // Undefined { - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); } // CF/OF @@ -2858,7 +2843,7 @@ void OpDispatchBuilder::GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, Ord // CF and OF are set if the result of the operation can't be fit in to the destination register // The result register will be all zero if it can't fit due to how multiplication behaves - auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, ZeroConst, ZeroConst, OneConst); + auto SelectOp = _Select(FEXCore::IR::COND_EQ, High, _Constant(0), _Constant(0), _Constant(1)); SetRFLAG(SelectOp); SetRFLAG(SelectOp); @@ -2866,13 +2851,11 @@ void OpDispatchBuilder::GenerateFlags_UMUL(FEXCore::X86Tables::DecodedOp Op, Ord } void OpDispatchBuilder::GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); // AF { // Undefined // Set to zero anyway - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); } // SF @@ -2887,35 +2870,32 @@ void OpDispatchBuilder::GenerateFlags_Logical(FEXCore::X86Tables::DecodedOp Op, { auto EightBitMask = _Constant(0xFF); auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, OneConst); + auto XorOp = _Xor(PopCountOp, _Constant(1)); SetRFLAG(XorOp); } // ZF { auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, ZeroConst, OneConst, ZeroConst); + Res, _Constant(0), _Constant(1), _Constant(0)); SetRFLAG(SelectOp); } // CF/OF { - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); } } void OpDispatchBuilder::GenerateFlags_Shift(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - auto ZeroConst = _Constant(0); - auto OneConst = _Constant(1); - - auto CmpResult = _Select(FEXCore::IR::COND_EQ, Src2, ZeroConst, OneConst, ZeroConst); + auto CmpResult = _Select(FEXCore::IR::COND_EQ, Src2, _Constant(0), _Constant(1), _Constant(0)); auto CondJump = _CondJump(CmpResult); // AF { // Undefined // Set to zero anyway - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); } // SF @@ -2930,36 +2910,35 @@ void OpDispatchBuilder::GenerateFlags_Shift(FEXCore::X86Tables::DecodedOp Op, Or { auto EightBitMask = _Constant(0xFF); auto PopCountOp = _Popcount(_And(Res, EightBitMask)); - auto XorOp = _Xor(PopCountOp, OneConst); + auto XorOp = _Xor(PopCountOp, _Constant(1)); SetRFLAG(XorOp); } // ZF { auto SelectOp = _Select(FEXCore::IR::COND_EQ, - Res, ZeroConst, OneConst, ZeroConst); + Res, _Constant(0), _Constant(1), _Constant(0)); SetRFLAG(SelectOp); } // CF/OF { - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); } - _EndBlock(0); - auto NewBlock = _BeginBlock(); + CreateNewEndBlock(0); + + auto NewBlock = CreateNewBeginBlock(); SetJumpTarget(CondJump, NewBlock); } void OpDispatchBuilder::GenerateFlags_Rotate(FEXCore::X86Tables::DecodedOp Op, OrderedNode *Res, OrderedNode *Src1, OrderedNode *Src2) { - auto ZeroConst = _Constant(0); - // CF/OF // XXX: These are wrong { - SetRFLAG(ZeroConst); - SetRFLAG(ZeroConst); + SetRFLAG(_Constant(0)); + SetRFLAG(_Constant(0)); } } @@ -3091,10 +3070,10 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) { // If condition doesn't hold then keep going auto CondJump = _CondJump(_Xor(Flag, _Constant(1))); _Break(Reason, Literal); - _EndBlock(0); + CreateNewEndBlock(0); // Make sure to start a new block after ending this one - auto JumpTarget = _BeginBlock(); + auto JumpTarget = CreateNewBeginBlock(); SetJumpTarget(CondJump, JumpTarget); } else { @@ -3240,8 +3219,40 @@ void OpDispatchBuilder::FXRStoreOp(OpcodeArgs) { } } +void OpDispatchBuilder::PAlignrOp(OpcodeArgs) { + OrderedNode *Src1 = LoadSource(Op, Op->Dest, Op->Flags); + OrderedNode *Src2 = LoadSource(Op, Op->Src1, Op->Flags); + + uint8_t Index = Op->Src2.TypeLiteral.Literal; + OrderedNode *Res = _VExtr(GetDstSize(Op), 1, Src1, Src2, Index); + StoreResult(Op, Res); +} + #undef OpcodeArgs + +void OpDispatchBuilder::ReplaceAllUsesWithInclusive(OrderedNode *Node, OrderedNode *NewNode, IR::NodeWrapperIterator After, IR::NodeWrapperIterator End) { + uintptr_t ListBegin = ListData.Begin(); + uintptr_t DataBegin = Data.Begin(); + + while (After != End) { + OrderedNodeWrapper *WrapperOp = After(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); + FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + + for (uint8_t i = 0; i < IROp->NumArgs; ++i) { + if (IROp->Args[i].ID() == Node->Wrapped(ListBegin).ID()) { + LogMan::Msg::D("\tAt %%ssa%d: Replacing ID %%ssa%d with %%ssa%d", WrapperOp->ID(), IROp->Args[i].ID(), NewNode->Wrapped(ListBegin).ID()); + Node->RemoveUse(); + NewNode->AddUse(); + IROp->Args[i].NodeOffset = NewNode->Wrapped(ListBegin).NodeOffset; + } + } + + ++After; + } +} + void InstallOpcodeHandlers() { const std::vector> BaseOpTable = { // Instructions @@ -3279,6 +3290,7 @@ void InstallOpcodeHandlers() { {0x9F, 1, &OpDispatchBuilder::LAHFOp}, {0xA0, 4, &OpDispatchBuilder::MOVOffsetOp}, {0xA4, 2, &OpDispatchBuilder::MOVSOp}, + // XXX: Causes issues with ld.so {0xA6, 2, &OpDispatchBuilder::CMPSOp}, {0xA8, 2, &OpDispatchBuilder::TESTOp}, {0xAA, 2, &OpDispatchBuilder::STOSOp}, @@ -3396,37 +3408,37 @@ void InstallOpcodeHandlers() { {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 0), 1, &OpDispatchBuilder::ROLOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 1), 1, &OpDispatchBuilder::ROROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 4), 1, &OpDispatchBuilder::SHLOp}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 5), 1, &OpDispatchBuilder::SHROp}, + {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 5), 1, &OpDispatchBuilder::SHROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC0), 7), 1, &OpDispatchBuilder::ASHROp}, // SAR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 0), 1, &OpDispatchBuilder::ROLOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 1), 1, &OpDispatchBuilder::ROROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 4), 1, &OpDispatchBuilder::SHLOp}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 5), 1, &OpDispatchBuilder::SHROp}, + {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 5), 1, &OpDispatchBuilder::SHROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xC1), 7), 1, &OpDispatchBuilder::ASHROp}, // SAR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 0), 1, &OpDispatchBuilder::ROLOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 1), 1, &OpDispatchBuilder::ROROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 4), 1, &OpDispatchBuilder::SHLOp}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 5), 1, &OpDispatchBuilder::SHROp}, // 1Bit SHR + {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 5), 1, &OpDispatchBuilder::SHROp}, // 1Bit SHR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD0), 7), 1, &OpDispatchBuilder::ASHROp}, // SAR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 0), 1, &OpDispatchBuilder::ROLOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 1), 1, &OpDispatchBuilder::ROROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 4), 1, &OpDispatchBuilder::SHLOp}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 5), 1, &OpDispatchBuilder::SHROp}, // 1Bit SHR + {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 5), 1, &OpDispatchBuilder::SHROp}, // 1Bit SHR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD1), 7), 1, &OpDispatchBuilder::ASHROp}, // SAR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 0), 1, &OpDispatchBuilder::ROLOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 1), 1, &OpDispatchBuilder::ROROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 4), 1, &OpDispatchBuilder::SHLOp}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL + {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD2), 7), 1, &OpDispatchBuilder::ASHROp}, // SAR {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 0), 1, &OpDispatchBuilder::ROLOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 1), 1, &OpDispatchBuilder::ROROp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 4), 1, &OpDispatchBuilder::SHLOp}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL + {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 5), 1, &OpDispatchBuilder::SHROp}, // SHR by CL {OPD(FEXCore::X86Tables::TYPE_GROUP_2, OpToIndex(0xD3), 7), 1, &OpDispatchBuilder::ASHROp}, // SAR // GROUP 3 @@ -3459,7 +3471,6 @@ void InstallOpcodeHandlers() { {OPD(FEXCore::X86Tables::TYPE_GROUP_5, OpToIndex(0xFF), 6), 1, &OpDispatchBuilder::PUSHOp}, // GROUP 11 - // XXX: LLVM hangs when commented out? {OPD(FEXCore::X86Tables::TYPE_GROUP_11, OpToIndex(0xC6), 0), 1, &OpDispatchBuilder::MOVOp}, {OPD(FEXCore::X86Tables::TYPE_GROUP_11, OpToIndex(0xC7), 0), 1, &OpDispatchBuilder::MOVOp}, #undef OPD @@ -3467,9 +3478,7 @@ void InstallOpcodeHandlers() { const std::vector> RepModOpTable = { {0x19, 7, &OpDispatchBuilder::NOPOp}, - {0x6F, 1, &OpDispatchBuilder::MOVUPSOp}, - // XXX: Causes LLVM to crash if commented out? {0x7E, 1, &OpDispatchBuilder::MOVQOp}, {0x7F, 1, &OpDispatchBuilder::MOVUPSOp}, }; @@ -3545,7 +3554,8 @@ constexpr uint16_t PF_F2 = 3; {OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 2), 1, &OpDispatchBuilder::PSRLD<4>}, {OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 6), 1, &OpDispatchBuilder::PSLL<8, true>}, {OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 3), 1, &OpDispatchBuilder::PSRLDQ}, - {OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 7), 1, &OpDispatchBuilder::PSLL<16, true>}, + // XXX: Causes issues with ld.so + // {OPD(FEXCore::X86Tables::TYPE_GROUP_14, PF_66, 7), 1, &OpDispatchBuilder::PSLL<16, true>}, // GROUP 15 {OPD(FEXCore::X86Tables::TYPE_GROUP_15, PF_NONE, 0), 1, &OpDispatchBuilder::FXSaveOp}, @@ -3564,6 +3574,18 @@ constexpr uint16_t PF_F2 = 3; const std::vector> X87OpTable = { }; +#define OPD(REX, prefix, opcode) ((REX << 9) | (prefix << 8) | opcode) +#define PF_3A_NONE 0 +#define PF_3A_66 1 + const std::vector> H0F3ATable = { + {OPD(0, PF_3A_66, 0x0F), 1, &OpDispatchBuilder::PAlignrOp}, + }; +#undef PF_3A_NONE +#undef PF_3A_66 + +#undef OPD + + uint64_t NumInsts{}; auto InstallToTable = [&NumInsts](auto& FinalTable, auto& LocalTable) { for (auto Op : LocalTable) { @@ -3600,6 +3622,8 @@ constexpr uint16_t PF_F2 = 3; InstallToTable(FEXCore::X86Tables::X87Ops, X87OpTable); + InstallToTable(FEXCore::X86Tables::H0F3ATableOps, H0F3ATable); + // Useful for debugging // CheckTable(FEXCore::X86Tables::BaseOps); printf("We installed %ld instructions to the tables\n", NumInsts); diff --git a/Source/Interface/Core/OpcodeDispatcher.h b/Source/Interface/Core/OpcodeDispatcher.h index 715b56661..fc50495ac 100644 --- a/Source/Interface/Core/OpcodeDispatcher.h +++ b/Source/Interface/Core/OpcodeDispatcher.h @@ -28,9 +28,9 @@ public: void ResetWorkingList(); bool HadDecodeFailure() { return DecodeFailure; } - void BeginBlock(); - void EndBlock(uint64_t RIPIncrement); + void BeginFunction(uint64_t RIP); void ExitFunction(); + void Finalize(); // Dispatch builder functions #define OpcodeArgs [[maybe_unused]] FEXCore::X86Tables::DecodedOp Op @@ -70,6 +70,7 @@ public: void CMOVOp(OpcodeArgs); void CPUIDOp(OpcodeArgs); void SHLOp(OpcodeArgs); + template void SHROp(OpcodeArgs); void ASHROp(OpcodeArgs); void ROROp(OpcodeArgs); @@ -128,6 +129,8 @@ public: void FXSaveOp(OpcodeArgs); void FXRStoreOp(OpcodeArgs); + void PAlignrOp(OpcodeArgs); + #undef OpcodeArgs /** @@ -215,7 +218,9 @@ public: IRPair _VUShr(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1) { return _VUShr(ssa0, ssa1, RegisterSize, ElementSize); } - + IRPair _VExtr(uint8_t RegisterSize, uint8_t ElementSize, OrderedNode *ssa0, OrderedNode *ssa1, uint8_t Index) { + return _VExtr(ssa0, ssa1, RegisterSize, ElementSize, Index); + } IRPair _Jump() { return _Jump(InvalidNode); } @@ -232,8 +237,8 @@ public: /** @} */ - bool IsValueConstant(NodeWrapper ssa, uint64_t *Constant) { - OrderedNode *RealNode = reinterpret_cast(ssa.GetPtr(ListData.Begin())); + bool IsValueConstant(OrderedNodeWrapper ssa, uint64_t *Constant) { + OrderedNode *RealNode = ssa.GetNode(ListData.Begin()); FEXCore::IR::IROp_Header *IROp = RealNode->Op(Data.Begin()); if (IROp->Op == OP_CONSTANT) { auto Op = IROp->C(); @@ -259,6 +264,8 @@ public: return Node; } + void ReplaceAllUsesWithInclusive(OrderedNode *Node, OrderedNode *NewNode, IR::NodeWrapperIterator After, IR::NodeWrapperIterator End); + void Unlink(OrderedNode *Node) { Node->Unlink(ListData.Begin()); } @@ -271,8 +278,55 @@ public: LogMan::Throw::A(rhs.ListData.BackingSize() <= ListData.BackingSize(), "Trying to take ownership of data that is too large"); Data.CopyData(rhs.Data); ListData.CopyData(rhs.ListData); + InvalidNode = rhs.InvalidNode; + CurrentWriteCursor = rhs.CurrentWriteCursor; + CodeBlocks = rhs.CodeBlocks; } + void SetWriteCursor(OrderedNode *Node) { + CurrentWriteCursor = Node; + } + + OrderedNode *GetWriteCursor() { + return CurrentWriteCursor; + } + + /** + * @brief This creates an orphaned code node + * The IROp backing is in the correct list but the OrderedNode lives outside of the list + * + * XXX: This is because we don't want code blocks to interleave with current instruction IR ops currently + * We can change this behaviour once we remove the old BeginBlock/EndBlock types + * + * @return OrderedNode + */ + OrderedNode *CreateCodeNode() { + OrderedNode *CodeNode = _CodeBlock(InvalidNode, InvalidNode, InvalidNode); + CodeBlocks.emplace_back(CodeNode); + return CodeNode; + } + + void SetCodeNodeBegin(OrderedNode *CodeNode, OrderedNode *Begin) { + FEXCore::IR::IROp_CodeBlock *IROp = CodeNode->Op(Data.Begin())->CW(); + LogMan::Throw::A(IROp->Header.Op == IROps::OP_CODEBLOCK, "Invalid"); + IROp->Begin = Begin->Wrapped(ListData.Begin()); + } + + void SetCodeNodeLast(OrderedNode *CodeNode, OrderedNode *Last) { + FEXCore::IR::IROp_CodeBlock *IROp = CodeNode->Op(Data.Begin())->CW(); + LogMan::Throw::A(IROp->Header.Op == IROps::OP_CODEBLOCK, "Invalid"); + IROp->Last = Last->Wrapped(ListData.Begin()); + } + + void LinkCodeBlocks(OrderedNode *CodeNode, OrderedNode *Next) { + FEXCore::IR::IROp_CodeBlock *IROp = CodeNode->Op(Data.Begin())->CW(); + LogMan::Throw::A(IROp->Header.Op == IROps::OP_CODEBLOCK, "Invalid"); + IROp->Next = Next->Wrapped(ListData.Begin()); + } + + IRPair CreateNewBeginBlock(); + IRPair CreateNewEndBlock(uint64_t RIPIncrement); + private: void TestFunction(); bool DecodeFailure{false}; @@ -314,12 +368,17 @@ private: return Node; } - void SetWriteCursor(OrderedNode *Node) { - CurrentWriteCursor = Node; + OrderedNode *GetNode(uint32_t SSANode) { + uintptr_t ListBegin = ListData.Begin(); + OrderedNode *Node = reinterpret_cast(ListBegin + SSANode * sizeof(OrderedNode)); + return Node; } - OrderedNode *GetWriteCursor() { - return CurrentWriteCursor; + OrderedNode *EmplaceOrphanedNode(OrderedNode *OldNode) { + size_t Size = sizeof(OrderedNode); + OrderedNode *Ptr = reinterpret_cast(ListData.Allocate(Size)); + memcpy(Ptr, OldNode, Size); + return Ptr; } OrderedNode *CurrentWriteCursor = nullptr; @@ -329,6 +388,8 @@ private: IntrusiveAllocator ListData; OrderedNode *InvalidNode; + OrderedNode *CurrentCodeBlock{}; + std::vector CodeBlocks; }; void InstallOpcodeHandlers(); diff --git a/Source/Interface/Core/RegisterAllocation.cpp b/Source/Interface/Core/RegisterAllocation.cpp deleted file mode 100644 index 8563640b4..000000000 --- a/Source/Interface/Core/RegisterAllocation.cpp +++ /dev/null @@ -1,230 +0,0 @@ -#include "Common/BitSet.h" -#include "Interface/Core/RegisterAllocation.h" - -#include - -#include - -constexpr uint32_t INVALID_REG = ~0U; -constexpr uint32_t INVALID_CLASS = ~0U; - - -namespace FEXCore::RA { - - struct Register { - }; - - struct RegisterClass { - uint32_t RegisterBase; - uint32_t NumberOfRegisters{0}; - BitSet Registers; - }; - - struct RegisterNode { - uint32_t RegisterClass; - uint32_t Register; - uint32_t InterferenceCount; - uint32_t InterferenceListSize; - uint32_t *InterferenceList; - BitSet Interference; - }; - - static_assert(std::is_pod::value, "We want this to be POD"); - - struct RegisterSet { - Register *Registers; - RegisterClass *RegisterClasses; - uint32_t RegisterCount; - uint32_t ClassCount; - }; - - struct SpillStackUnit { - uint32_t Node; - uint32_t Class; - }; - - struct RegisterGraph { - RegisterSet *Set; - RegisterNode *Nodes; - uint32_t NodeCount; - uint32_t MaxNodeCount; - std::vector SpillStack; - }; - - RegisterSet *AllocateRegisterSet(uint32_t RegisterCount, uint32_t ClassCount) { - RegisterSet *Set = new RegisterSet; - - Set->RegisterCount = RegisterCount; - Set->ClassCount = ClassCount; - - Set->Registers = static_cast(calloc(RegisterCount, sizeof(Register))); - Set->RegisterClasses = static_cast(calloc(ClassCount, sizeof(RegisterClass))); - - for (uint32_t i = 0; i < ClassCount; ++i) { - Set->RegisterClasses[i].Registers.Allocate(RegisterCount); - } - - return Set; - } - - void FreeRegisterSet(RegisterSet *Set) { - for (uint32_t i = 0; i < Set->ClassCount; ++i) { - Set->RegisterClasses[i].Registers.Free(); - } - free(Set->RegisterClasses); - free(Set->Registers); - delete Set; - } - - void AddRegisters(RegisterSet *Set, uint32_t Class, uint32_t RegistersBase, uint32_t RegisterCount) { - for (uint32_t i = 0; i < RegisterCount; ++i) { - Set->RegisterClasses[Class].Registers.Set(RegistersBase + i); - } - Set->RegisterClasses[Class].RegisterBase = RegistersBase; - Set->RegisterClasses[Class].NumberOfRegisters += RegisterCount; - } - - RegisterGraph *AllocateRegisterGraph(RegisterSet *Set, uint32_t NodeCount) { - RegisterGraph *Graph = new RegisterGraph; - Graph->Set = Set; - Graph->NodeCount = NodeCount; - Graph->MaxNodeCount = NodeCount; - Graph->Nodes = static_cast(calloc(NodeCount, sizeof(RegisterNode))); - - // Initialize nodes - for (uint32_t i = 0; i < NodeCount; ++i) { - Graph->Nodes[i].Register = INVALID_REG; - Graph->Nodes[i].RegisterClass = INVALID_CLASS; - Graph->Nodes[i].InterferenceListSize = 32; - Graph->Nodes[i].InterferenceList = reinterpret_cast(calloc(Graph->Nodes[i].InterferenceListSize, sizeof(uint32_t))); - Graph->Nodes[i].InterferenceCount = 0; - Graph->Nodes[i].Interference.Allocate(NodeCount); - Graph->Nodes[i].Interference.Clear(NodeCount); - } - - return Graph; - } - - void ResetRegisterGraph(RegisterGraph *Graph, uint32_t NodeCount) { - if (NodeCount > Graph->MaxNodeCount) { - uint32_t OldNodeCount = Graph->MaxNodeCount; - Graph->NodeCount = NodeCount; - Graph->MaxNodeCount = NodeCount; - Graph->Nodes = static_cast(realloc(Graph->Nodes, NodeCount * sizeof(RegisterNode))); - - // Initialize nodes - for (uint32_t i = 0; i < OldNodeCount; ++i) { - Graph->Nodes[i].Register = INVALID_REG; - Graph->Nodes[i].RegisterClass = INVALID_CLASS; - Graph->Nodes[i].InterferenceCount = 0; - Graph->Nodes[i].Interference.Realloc(NodeCount); - Graph->Nodes[i].Interference.Clear(NodeCount); - } - - for (uint32_t i = OldNodeCount; i < NodeCount; ++i) { - Graph->Nodes[i].Register = INVALID_REG; - Graph->Nodes[i].RegisterClass = INVALID_CLASS; - Graph->Nodes[i].InterferenceListSize = 32; - Graph->Nodes[i].InterferenceList = reinterpret_cast(calloc(Graph->Nodes[i].InterferenceListSize, sizeof(uint32_t))); - Graph->Nodes[i].InterferenceCount = 0; - Graph->Nodes[i].Interference.Allocate(NodeCount); - Graph->Nodes[i].Interference.Clear(NodeCount); - } - } - else { - // We are only handling a node count of this size right now - Graph->NodeCount = NodeCount; - - // Initialize nodes - for (uint32_t i = 0; i < NodeCount; ++i) { - Graph->Nodes[i].Register = INVALID_REG; - Graph->Nodes[i].RegisterClass = INVALID_CLASS; - Graph->Nodes[i].InterferenceCount = 0; - Graph->Nodes[i].Interference.Clear(NodeCount); - } - } - } - - void FreeRegisterGraph(RegisterGraph *Graph) { - for (uint32_t i = 0; i < Graph->MaxNodeCount; ++i) { - RegisterNode *Node = &Graph->Nodes[i]; - Node->InterferenceCount = 0; - Node->InterferenceListSize = 0; - free(Node->InterferenceList); - Node->Interference.Free(); - } - - free(Graph->Nodes); - Graph->NodeCount = 0; - Graph->MaxNodeCount = 0; - delete Graph; - } - - void SetNodeClass(RegisterGraph *Graph, uint32_t Node, uint32_t Class) { - Graph->Nodes[Node].RegisterClass = Class; - } - - void AddNodeInterference(RegisterGraph *Graph, uint32_t Node1, uint32_t Node2) { - auto AddInterference = [&Graph](uint32_t Node1, uint32_t Node2) { - RegisterNode *Node = &Graph->Nodes[Node1]; - Node->Interference.Set(Node2); - if (Node->InterferenceListSize <= Node->InterferenceCount) { - Node->InterferenceListSize *= 2; - Node->InterferenceList = reinterpret_cast(realloc(Node->InterferenceList, Node->InterferenceListSize * sizeof(uint32_t))); - } - Node->InterferenceList[Node->InterferenceCount] = Node2; - ++Node->InterferenceCount; - }; - - AddInterference(Node1, Node2); - AddInterference(Node2, Node1); - } - - uint32_t GetNodeRegister(RegisterGraph *Graph, uint32_t Node) { - return Graph->Nodes[Node].Register; - } - - static bool HasInterference(RegisterGraph *Graph, RegisterNode *Node, uint32_t Register) { - for (uint32_t i = 0; i < Node->InterferenceCount; ++i) { - RegisterNode *IntNode = &Graph->Nodes[Node->InterferenceList[i]]; - if (IntNode->Register == Register) { - return true; - } - } - - return false; - } - - bool AllocateRegisters(RegisterGraph *Graph) { - Graph->SpillStack.clear(); - for (uint32_t i = 0; i < Graph->NodeCount; ++i) { - RegisterNode *CurrentNode = &Graph->Nodes[i]; - if (CurrentNode->RegisterClass == INVALID_CLASS) - continue; - - uint32_t Reg = ~0U; - RegisterClass *RAClass = &Graph->Set->RegisterClasses[CurrentNode->RegisterClass]; - for (uint32_t ri = 0; ri < RAClass->NumberOfRegisters; ++ri) { - if (!HasInterference(Graph, CurrentNode, RAClass->RegisterBase + ri)) { - Reg = ri; - break; - } - } - - if (Reg == ~0U) { - Graph->SpillStack.emplace_back(SpillStackUnit{i, CurrentNode->RegisterClass}); - } - else { - CurrentNode->Register = RAClass->RegisterBase + Reg; - } - } - - if (!Graph->SpillStack.empty()) { - printf("Couldn't allocate %ld registers\n", Graph->SpillStack.size()); - return false; - } - return true; - } - -} - diff --git a/Source/Interface/Core/RegisterAllocation.h b/Source/Interface/Core/RegisterAllocation.h deleted file mode 100644 index d7c044d5e..000000000 --- a/Source/Interface/Core/RegisterAllocation.h +++ /dev/null @@ -1,30 +0,0 @@ -#pragma once -#include -#include - -namespace FEXCore::RA { -struct RegisterSet; -struct RegisterGraph; -using CrappyBitset = std::vector; - -RegisterSet *AllocateRegisterSet(uint32_t RegisterCount, uint32_t ClassCount); -void FreeRegisterSet(RegisterSet *Set); -void AddRegisters(RegisterSet *Set, uint32_t Class, uint32_t RegistersBase, uint32_t RegisterCount); - -/** - * @name Inference graph handling - * @{ */ - -RegisterGraph *AllocateRegisterGraph(RegisterSet *Set, uint32_t NodeCount); -void FreeRegisterGraph(RegisterGraph *Graph); -void ResetRegisterGraph(RegisterGraph *Graph, uint32_t NodeCount); -void SetNodeClass(RegisterGraph *Graph, uint32_t Node, uint32_t Class); -void AddNodeInterference(RegisterGraph *Graph, uint32_t Node1, uint32_t Node2); -uint32_t GetNodeRegister(RegisterGraph *Graph, uint32_t Node); - -bool AllocateRegisters(RegisterGraph *Graph); - -/** @} */ - -} - diff --git a/Source/Interface/HLE/Syscalls.cpp b/Source/Interface/HLE/Syscalls.cpp index 68d28a074..0b5e041ae 100644 --- a/Source/Interface/HLE/Syscalls.cpp +++ b/Source/Interface/HLE/Syscalls.cpp @@ -8,6 +8,7 @@ #include #include +#include constexpr uint64_t PAGE_SIZE = 4096; @@ -128,7 +129,7 @@ void SyscallHandler::DefaultProgramBreak(FEXCore::Core::InternalThreadState *Thr DefaultProgramBreakAddress = Addr; // Just allocate 1GB of data memory past the default program break location at this point - CTX->MapRegion(Thread, Addr, 0x1000'0000); + CTX->MapRegion(Thread, Addr, 0x1000'0000, true); } uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Thread, FEXCore::HLE::SyscallArguments *Args) { @@ -154,8 +155,11 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa } // Memory management case SYSCALL_BRK: { - LogMan::Msg::D("\tBRK: 0x%lx - 0x%lx", Args->Argument[1], DataSpace); if (Args->Argument[1] == 0) { // Just wants to get the location of the program break atm + if (DataSpace == 0) { + // XXX: We need to setup our default BRK space first + DefaultProgramBreak(Thread, 0xe000'0000); + } Result = DataSpace + DataSpaceSize; } else { @@ -175,12 +179,8 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa break; } case SYSCALL_MMAP: { - LogMan::Msg::D("\tMMAP( %p, 0x%lx, %d, 0x%x, %d, 0x%lx)", - Args->Argument[1], Args->Argument[2], - Args->Argument[3], Args->Argument[4], - Args->Argument[5], Args->Argument[6]); int Flags = Args->Argument[4]; - int GuestFD = Args->Argument[5]; + int GuestFD = static_cast(Args->Argument[5]); int HostFD = -1; if (GuestFD != -1) { @@ -193,13 +193,13 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa uint64_t Prot = Args->Argument[3]; #ifdef DEBUG_MMAP - FileSizeToUse = Size; - Prot = PROT_READ | PROT_WRITE | PROT_EXEC; +// FileSizeToUse = Size; #endif + Prot = PROT_READ | PROT_WRITE | PROT_EXEC; if (Flags & MAP_FIXED) { Base = Args->Argument[1]; - void *HostPtr = CTX->MemoryMapper.GetPointer(Base); + void *HostPtr = CTX->MemoryMapper.GetPointerSizeCheck(Base, FileSizeToUse); if (!HostPtr) { HostPtr = CTX->MapRegion(Thread, Base, Size, true); } @@ -232,7 +232,13 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa } else { // XXX: MMAP should map memory regions for all threads - void *HostPtr = CTX->MapRegion(Thread, Base, Size, true); + void *HostPtr = CTX->MemoryMapper.GetPointerSizeCheck(Base, FileSizeToUse); + if (!HostPtr) { + HostPtr = CTX->MapRegion(Thread, Base, Size, true); + } + else { + LogMan::Msg::D("\tMapping Fixed pointer in already mapped space: 0x%lx -> %p", Base, HostPtr); + } if (HostFD != -1) { #ifdef DEBUG_MMAP @@ -258,14 +264,12 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa break; } case SYSCALL_MPROTECT: { - LogMan::Msg::D("\tMPROTECT: 0x%x, 0x%lx, 0x%lx", Args->Argument[1], Args->Argument[2], Args->Argument[3]); void *HostPtr = CTX->MemoryMapper.GetPointer(Args->Argument[1]); - Result = mprotect(HostPtr, Args->Argument[2], Args->Argument[3]); +// Result = mprotect(HostPtr, Args->Argument[2], Args->Argument[3]); break; } case SYSCALL_ARCH_PRCTL: { - LogMan::Msg::D("\tPRTCL: 0x%x: 0x%lx", Args->Argument[1], Args->Argument[2]); switch (Args->Argument[1]) { case 0x1001: // ARCH_SET_GS Thread->State.State.gs = Args->Argument[2]; @@ -508,7 +512,21 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa Args->Argument[3], Args->Argument[4]); break; + case SYSCALL_IOCTL: + Result = FM.Ioctl( + Args->Argument[1], + Args->Argument[2], + CTX->MemoryMapper.GetPointer(Args->Argument[3])); + break; + case SYSCALL_TIME: { + time_t *ClockResult = CTX->MemoryMapper.GetPointer(Args->Argument[2]); + Result = time(ClockResult); + // XXX: Debug + // memset(ClockResult, 0, sizeof(time_t)); + // Result = 0; + } + break; case SYSCALL_CLOCK_GETTIME: { timespec *ClockResult = CTX->MemoryMapper.GetPointer(Args->Argument[2]); Result = clock_gettime(Args->Argument[1], ClockResult); @@ -533,25 +551,31 @@ uint64_t SyscallHandler::HandleSyscall(FEXCore::Core::InternalThreadState *Threa break; } case SYSCALL_PRLIMIT64: { - LogMan::Throw::A(Args->Argument[3] == 0, "Guest trying to set limit for %d", Args->Argument[2]); - struct rlimit { - uint64_t rlim_cur; - uint64_t rlim_max; - }; - switch (Args->Argument[2]) { - case 3: { // Stack limits - rlimit *old_limit = CTX->MemoryMapper.GetPointer(Args->Argument[3]); - // Default size - old_limit->rlim_cur = 8 * 1024; - old_limit->rlim_max = ~0ULL; - break; - } - default: LogMan::Msg::A("Unknown PRLimit: %d", Args->Argument[2]); - } - Result = 0; - + LogMan::Throw::A(Args->Argument[3] == 0, "Guest trying to set limit for %d", Args->Argument[2]); + struct rlimit { + uint64_t rlim_cur; + uint64_t rlim_max; + }; + switch (Args->Argument[2]) { + case 3: { // Stack limits + rlimit *old_limit = CTX->MemoryMapper.GetPointer(Args->Argument[3]); + // Default size + old_limit->rlim_cur = 8 * 1024; + old_limit->rlim_max = ~0ULL; + break; + } + default: LogMan::Msg::A("Unknown PRLimit: %d", Args->Argument[2]); + } + Result = 0; break; } + case SYSCALL_UMASK: + // Just say that the mask has always matched what was passed in + Result = Args->Argument[1]; + break; + case SYSCALL_CHDIR: + Result = chdir(CTX->MemoryMapper.GetPointer(Args->Argument[1])); + break; // Currently unhandled // Return fake result case SYSCALL_RT_SIGACTION: diff --git a/Source/Interface/IR/IR.cpp b/Source/Interface/IR/IR.cpp index 3966a5628..7663dd45a 100644 --- a/Source/Interface/IR/IR.cpp +++ b/Source/Interface/IR/IR.cpp @@ -8,12 +8,20 @@ namespace FEXCore::IR { static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, uint64_t Arg) { *out << "0x" << std::hex << Arg; } +static void PrintArg(std::stringstream *out, [[maybe_unused]] IRListView const* IR, RegisterClassType Arg) { + if (Arg == 0) + *out << "GPR"; + else if (Arg == 1) + *out << "FPR"; + else + *out << "Unknown Registerclass " << Arg; +} -static void PrintArg(std::stringstream *out, IRListView const* IR, NodeWrapper Arg) { +static void PrintArg(std::stringstream *out, IRListView const* IR, OrderedNodeWrapper Arg) { uintptr_t Data = IR->GetData(); uintptr_t ListBegin = IR->GetListData(); - OrderedNode *RealNode = reinterpret_cast(Arg.GetPtr(ListBegin)); + OrderedNode *RealNode = Arg.GetNode(ListBegin); auto IROp = RealNode->Op(Data); *out << "%ssa" << std::to_string(Arg.ID()) << " i" << std::dec << (IROp->Size * 8); @@ -23,44 +31,95 @@ static void PrintArg(std::stringstream *out, IRListView const* IR, NodeWr } void Dump(std::stringstream *out, IRListView const* IR) { - uintptr_t Data = IR->GetData(); uintptr_t ListBegin = IR->GetListData(); + uintptr_t DataBegin = IR->GetData(); auto Begin = IR->begin(); - auto End = IR->end(); - while (Begin != End) { - auto Op = Begin(); - OrderedNode *RealNode = reinterpret_cast(Op->GetPtr(ListBegin)); - auto IROp = RealNode->Op(Data); + auto Op = Begin(); - auto Name = FEXCore::IR::GetName(IROp->Op); + OrderedNode *RealNode = Op->GetNode(ListBegin); + auto HeaderOp = RealNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); - if (IROp->HasDest) { - *out << "%ssa" << std::to_string(Op->ID()) << " i" << std::dec << (IROp->Size * 8); - if (IROp->Elements > 1) { - *out << "v" << std::dec << IROp->Elements; + OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); + uint8_t CurrentIndent = 0; + auto AddIndent = [&out, &CurrentIndent]() { + for (uint8_t i = 0; i < CurrentIndent; ++i) { + *out << "\t"; + } + }; + + *out << "(%%ssa" << std::to_string(RealNode->Wrapped(ListBegin).ID()) << ") " << "IRHeader "; + *out << "0x" << std::hex << HeaderOp->Entry << ", "; + *out << "%%ssa" << HeaderOp->Blocks.ID() << ", "; + *out << std::dec << HeaderOp->BlockCount << std::endl; + + while (1) { + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = IR->at(BlockIROp->Begin); + auto CodeLast = IR->at(BlockIROp->Last); + *out << "(%%ssa" << std::to_string(BlockNode->Wrapped(ListBegin).ID()) << ") " << "CodeBlock "; + + *out << "%%ssa" << std::to_string(BlockIROp->Begin.ID()) << ", "; + *out << "%%ssa" << std::to_string(BlockIROp->Last.ID()) << ", "; + *out << "%%ssa" << std::to_string(BlockIROp->Next.ID()) << std::endl; + + while (1) { + OrderedNodeWrapper *CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); + + auto Name = FEXCore::IR::GetName(IROp->Op); + + AddIndent(); + if (IROp->HasDest) { + *out << "%ssa" << std::to_string(CodeOp->ID()) << " i" << std::dec << (IROp->Size * 8); + if (IROp->Elements > 1) { + *out << "v" << std::dec << IROp->Elements; + } + *out << " = "; } - *out << " = "; + else { + *out << "(%%ssa" << std::to_string(CodeOp->ID()) << ") "; + } + + *out << Name; + switch (IROp->Op) { + case IR::OP_BEGINBLOCK: + *out << " %ssa" << std::to_string(CodeOp->ID()); + ++CurrentIndent; + break; + case IR::OP_ENDBLOCK: + --CurrentIndent; + break; + default: break; + } + + #define IROP_ARGPRINTER_HELPER + #include "IRDefines.inc" + default: *out << ""; break; + } + + *out << "\n"; + printf("%s", out->str().c_str()); + *out = std::stringstream{}; + + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; } - *out << Name; - switch (IROp->Op) { - case IR::OP_BEGINBLOCK: - *out << " %ssa" << std::to_string(Op->ID()); + if (BlockIROp->Next.ID() == 0) { break; - default: break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); } - -#define IROP_ARGPRINTER_HELPER -#include "IRDefines.inc" - default: *out << ""; break; - } - - *out << "\n"; - - ++Begin; } - } } diff --git a/Source/Interface/IR/IR.json b/Source/Interface/IR/IR.json index 3e17763ab..bdb31cb25 100644 --- a/Source/Interface/IR/IR.json +++ b/Source/Interface/IR/IR.json @@ -19,6 +19,21 @@ "Ops": { "Dummy": { }, + "IRHeader": { + "Args": [ + "uint64_t", "Entry", + "OrderedNodeWrapper", "Blocks", + "uint32_t", "BlockCount" + ] + }, + "CodeBlock": { + "SSAArgs": "3", + "SSANames": [ + "Begin", + "Last", + "Next" + ] + }, "Constant": { "HasDest": true, "FixedDestSize": "8", @@ -79,6 +94,20 @@ "uint32_t", "Offset" ] }, + "SpillRegister": { + "SSAArgs": "1", + "Args": [ + "uint32_t", "Slot", + "RegisterClassType", "Class" + ] + }, + "FillRegister": { + "HasDest": true, + "Args": [ + "uint32_t", "Slot", + "RegisterClassType", "Class" + ] + }, "LoadFlag": { "HasDest": true, @@ -343,6 +372,11 @@ "SSAArgs": "3" }, + "Print": { + "DispatcherUnary": true, + "SSAArgs": "1" + }, + "CreateVector2": { "HasDest": true, "DestSize": "GetOpSize(ssa0) * 2", @@ -513,11 +547,15 @@ ] }, - "Print": { - "DispatcherUnary": true, - "SSAArgs": "1" + "VExtr": { + "HasDest": true, + "SSAArgs": "2", + "Args": [ + "uint8_t", "RegisterSize", + "uint8_t", "ElementSize", + "uint8_t", "Index" + ] }, - "Last": { "Last": true, "Args": [] diff --git a/Source/Interface/IR/PassManager.cpp b/Source/Interface/IR/PassManager.cpp index bf77e6f24..9b341b508 100644 --- a/Source/Interface/IR/PassManager.cpp +++ b/Source/Interface/IR/PassManager.cpp @@ -1,14 +1,19 @@ #include "Interface/IR/Passes.h" +#include "Interface/IR/Passes/RegisterAllocationPass.h" #include "Interface/IR/PassManager.h" namespace FEXCore::IR { + void PassManager::AddDefaultPasses() { Passes.emplace_back(std::unique_ptr(CreateConstProp())); - Passes.emplace_back(std::unique_ptr(CreateRedundantContextLoadElimination())); + // XXX: Causes corrupted output in test app + // Passes.emplace_back(std::unique_ptr(CreateRedundantContextLoadElimination())); Passes.emplace_back(std::unique_ptr(CreateRedundantFlagCalculationEliminination())); Passes.emplace_back(std::unique_ptr(CreateSyscallOptimization())); Passes.emplace_back(std::unique_ptr(CreatePassDeadContextStoreElimination())); + // If the IR is compacted post-RA then the node indexing gets messed up and the backend isn't able to find the register assigned to a node + // Compact before IR, don't worry about RA generating spills/fills Passes.emplace_back(std::unique_ptr(CreateIRCompaction())); } diff --git a/Source/Interface/IR/PassManager.h b/Source/Interface/IR/PassManager.h index 379e5cbb8..f07a06714 100644 --- a/Source/Interface/IR/PassManager.h +++ b/Source/Interface/IR/PassManager.h @@ -18,6 +18,9 @@ class PassManager final { public: void AddDefaultPasses(); void AddDefaultValidationPasses(); + void InsertPass(Pass *Pass) { + Passes.emplace_back(Pass); + } bool Run(OpDispatchBuilder *Disp); private: diff --git a/Source/Interface/IR/Passes/ConstProp.cpp b/Source/Interface/IR/Passes/ConstProp.cpp index 7793f48b7..979cd7cc4 100644 --- a/Source/Interface/IR/Passes/ConstProp.cpp +++ b/Source/Interface/IR/Passes/ConstProp.cpp @@ -14,33 +14,60 @@ bool ConstProp::Run(OpDispatchBuilder *Disp) { uintptr_t ListBegin = CurrentIR.GetListData(); uintptr_t DataBegin = CurrentIR.GetData(); - IR::NodeWrapperIterator Begin = CurrentIR.begin(); - IR::NodeWrapperIterator End = CurrentIR.end(); + auto Begin = CurrentIR.begin(); + auto Op = Begin(); - while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + OrderedNode *RealNode = Op->GetNode(ListBegin); + auto HeaderOp = RealNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); - switch (IROp->Op) { - case OP_ZEXT: { - auto Op = IROp->C(); - uint64_t Constant; - if (Disp->IsValueConstant(Op->Header.Args[0], &Constant)) { - uint64_t NewConstant = Constant & ((1ULL << Op->SrcSize) - 1); - auto ConstantVal = Disp->_Constant(NewConstant); - Disp->ReplaceAllUsesWith(RealNode, ConstantVal); - Changed = true; + OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); + + auto OriginalWriteCursor = Disp->GetWriteCursor(); + + while (1) { + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + while (1) { + auto CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); + switch (IROp->Op) { + case OP_ZEXT: { + auto Op = IROp->C(); + uint64_t Constant; + if (Disp->IsValueConstant(Op->Header.Args[0], &Constant)) { + uint64_t NewConstant = Constant & ((1ULL << Op->SrcSize) - 1); + Disp->SetWriteCursor(CodeNode); + auto ConstantVal = Disp->_Constant(NewConstant); + Disp->ReplaceAllUsesWith(CodeNode, ConstantVal); + Changed = true; + } + break; + } + default: break; } - break; - } - default: break; + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; } - ++Begin; + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + } } + Disp->SetWriteCursor(OriginalWriteCursor); + return Changed; } diff --git a/Source/Interface/IR/Passes/DeadContextStoreElimination.cpp b/Source/Interface/IR/Passes/DeadContextStoreElimination.cpp index f373a13d3..bb431e261 100644 --- a/Source/Interface/IR/Passes/DeadContextStoreElimination.cpp +++ b/Source/Interface/IR/Passes/DeadContextStoreElimination.cpp @@ -42,68 +42,88 @@ static bool IsGPR(uint32_t Offset, uint8_t *greg) { bool RCLE::Run(OpDispatchBuilder *Disp) { bool Changed = false; auto CurrentIR = Disp->ViewIR(); + std::array LastValidGPRStores{}; + auto OriginalWriteCursor = Disp->GetWriteCursor(); + uintptr_t ListBegin = CurrentIR.GetListData(); uintptr_t DataBegin = CurrentIR.GetData(); - IR::NodeWrapperIterator Begin = CurrentIR.begin(); - IR::NodeWrapperIterator End = CurrentIR.end(); + auto Begin = CurrentIR.begin(); + auto Op = Begin(); - std::array LastValidGPRStores{}; + OrderedNode *RealNode = Op->GetNode(ListBegin); + auto HeaderOp = RealNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); - while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); - if (IROp->Op == OP_BEGINBLOCK || - IROp->Op == OP_ENDBLOCK || - IROp->Op == OP_JUMP || - IROp->Op == OP_CONDJUMP || - IROp->Op == OP_EXITFUNCTION) { - // We don't track across block boundaries - LastValidGPRStores.fill(nullptr); - } + while (1) { + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); - if (IROp->Op == OP_STORECONTEXT) { - auto Op = IROp->CW(); - // Make sure we are within GREG state - uint8_t greg = ~0; - if (IsAlignedGPR(Op->Size, Op->Offset, &greg)) { - FEXCore::IR::IROp_Header *ArgOp = reinterpret_cast(Op->Header.Args[0].GetPtr(ListBegin))->Op(DataBegin); - // Ensure we aren't doing a mismatched store - // XXX: We should really catch this in IR validation - if (ArgOp->Size == 8) { - LastValidGPRStores[greg] = &Op->Header.Args[0]; - } - else { - LastValidGPRStores[greg] = nullptr; - } - } else if (IsGPR(Op->Offset, &greg)) { - // If we aren't overwriting the whole state then we don't want to track this value - LastValidGPRStores[greg] = nullptr; - } - } + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + while (1) { + auto CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); - if (IROp->Op == OP_LOADCONTEXT) { - auto Op = IROp->C(); - - // Make sure we are within GREG state - uint8_t greg = ~0; - if (IsAlignedGPR(Op->Size, Op->Offset, &greg)) { - if (LastValidGPRStores[greg] != nullptr) { - // If the last store matches this load value then we can replace the loaded value with the previous valid one - auto MovVal = Disp->_Mov(reinterpret_cast(LastValidGPRStores[greg]->GetPtr(ListBegin))); - Disp->ReplaceAllUsesWith(RealNode, MovVal); - Changed = true; - } - } else if (IsGPR(Op->Offset, &greg)) { + if (IROp->Op == OP_STORECONTEXT) { + auto Op = IROp->CW(); + // Make sure we are within GREG state + uint8_t greg = ~0; + if (IsAlignedGPR(Op->Size, Op->Offset, &greg)) { + LastValidGPRStores[greg] = Op->Header.Args[0].NodeOffset; + } else if (IsGPR(Op->Offset, &greg)) { // If we aren't overwriting the whole state then we don't want to track this value - LastValidGPRStores[greg] = nullptr; // 0 is invalid + LastValidGPRStores[greg] = 0; + } } + + if (IROp->Op == OP_LOADCONTEXT) { + auto Op = IROp->C(); + + // Make sure we are within GREG state + uint8_t greg = ~0; + if (IsAlignedGPR(Op->Size, Op->Offset, &greg)) { + if (LastValidGPRStores[greg] != 0) { + // If the last store matches this load value then we can replace the loaded value with the previous valid one + if (1) { + Disp->SetWriteCursor(CodeNode); + auto MovVal = Disp->_Mov(OrderedNodeWrapper::WrapOffset(LastValidGPRStores[greg]).GetNode(ListBegin)); + Disp->ReplaceAllUsesWith(CodeNode, MovVal); + } + else { + Disp->ReplaceAllUsesWith(CodeNode, OrderedNodeWrapper::WrapOffset(LastValidGPRStores[greg]).GetNode(ListBegin)); + } + Changed = true; + } + } else if (IsGPR(Op->Offset, &greg)) { + // If we aren't overwriting the whole state then we don't want to track this value + LastValidGPRStores[greg] = 0; // 0 is invalid + } + } + + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; } - ++Begin; + + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + } + + // We don't track across block boundaries + LastValidGPRStores.fill(0); } + Disp->SetWriteCursor(OriginalWriteCursor); + return Changed; } diff --git a/Source/Interface/IR/Passes/IRCompaction.cpp b/Source/Interface/IR/Passes/IRCompaction.cpp index e36a3ad62..81dd2d65e 100644 --- a/Source/Interface/IR/Passes/IRCompaction.cpp +++ b/Source/Interface/IR/Passes/IRCompaction.cpp @@ -12,7 +12,7 @@ public: private: OpDispatchBuilder LocalBuilder; - std::vector NodeLocationRemapper; + std::vector NodeLocationRemapper; }; IRCompaction::IRCompaction() { @@ -21,15 +21,16 @@ IRCompaction::IRCompaction() { bool IRCompaction::Run(OpDispatchBuilder *Disp) { auto CurrentIR = Disp->ViewIR(); - auto LocalIR = LocalBuilder.ViewIR(); - uint32_t NodeCount = LocalIR.GetListSize() / sizeof(OrderedNode); + uint32_t NodeCount = CurrentIR.GetSSACount(); - // Reset our local working list - LocalBuilder.ResetWorkingList(); if (NodeLocationRemapper.size() < NodeCount) { NodeLocationRemapper.resize(NodeCount); } - memset(&NodeLocationRemapper.at(0), 0xFF, NodeCount * sizeof(IR::NodeWrapper::NodeOffsetType)); + memset(&NodeLocationRemapper.at(0), 0xFF, NodeCount * sizeof(IR::OrderedNodeWrapper::NodeOffsetType)); + + // Reset our local working list + LocalBuilder.ResetWorkingList(); + auto LocalIR = LocalBuilder.ViewIR(); uintptr_t LocalListBegin = LocalIR.GetListData(); uintptr_t LocalDataBegin = LocalIR.GetData(); @@ -37,89 +38,201 @@ bool IRCompaction::Run(OpDispatchBuilder *Disp) { uintptr_t ListBegin = CurrentIR.GetListData(); uintptr_t DataBegin = CurrentIR.GetData(); - IR::NodeWrapperIterator Begin = CurrentIR.begin(); - IR::NodeWrapperIterator End = CurrentIR.end(); + auto HeaderIterator = CurrentIR.begin(); + OrderedNodeWrapper *HeaderNodeWrapper = HeaderIterator(); + OrderedNode *HeaderNode = HeaderNodeWrapper->GetNode(ListBegin); + auto HeaderOp = HeaderNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); - // This compaction pass is something that we need to ensure correct ordering and distances between IROps\ + // This compaction pass is something that we need to ensure correct ordering and distances between IROps // Later on we assume that an IROp's SSA value live range is its Node locations // // RA distance calculation is calculated purely on the Node locations - // So we just need to reorder those + // So we need to reorder those // // Additionally there may be some dead ops hanging out in the IR list that are orphaned. // These can also be dropped during this pass - while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + // First thing is first, we need to do some housekeeping + // Create the IRHeader op + // Create the codeblocks + // Then create all the ops inside the code blocks - if (IROp->HasDest && RealNode->GetUses() == 0) { - // Should this be in a dedicated DCE pass? - ++Begin; - continue; + auto LocalHeaderOp = LocalBuilder._IRHeader(HeaderOp->Entry, OrderedNodeWrapper::WrapOffset(0), HeaderOp->BlockCount); + NodeLocationRemapper[HeaderNode->Wrapped(ListBegin).ID()] = LocalHeaderOp.Node->Wrapped(LocalListBegin).ID(); + + struct CodeBlockData { + OrderedNode *OldNode; + OrderedNode *NewNode; + }; + std::vector GeneratedCodeBlocks{}; + + { + // Generate our codeblocks and link them together + OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); + OrderedNode* PrevCodeBlock{}; + while (1) { + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); + + auto LocalBlockIRNode = LocalBuilder.CreateCodeNode(); + NodeLocationRemapper[BlockNode->Wrapped(ListBegin).ID()] = LocalBlockIRNode->Wrapped(LocalListBegin).ID(); + GeneratedCodeBlocks.emplace_back(CodeBlockData{BlockNode, LocalBlockIRNode}); + + if (PrevCodeBlock) { + auto PrevLocalBlockIROp = PrevCodeBlock->Op(LocalDataBegin)->CW(); + PrevLocalBlockIROp->Next = LocalBlockIRNode->Wrapped(LocalListBegin); + } + + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + PrevCodeBlock = LocalBlockIRNode; + } } - size_t OpSize = FEXCore::IR::GetSize(IROp->Op); - // Allocate the ops locally for our local dispatch - auto LocalPair = LocalBuilder.AllocateRawOp(OpSize); - IR::NodeWrapper LocalNodeWrapper = LocalPair.Node->Wrapped(LocalListBegin); - - // Copy over the op - memcpy(LocalPair.first, IROp, OpSize); - - // Set our map remapper to map the new location - // Even nodes that don't have a destination need to be in this map - // Need to be able to remap branch targets any other bits - NodeLocationRemapper[WrapperOp->ID()] = LocalNodeWrapper.ID(); - ++Begin; + // Link the IRHeader to the first code block + LocalHeaderOp.first->Blocks = GeneratedCodeBlocks[0].NewNode->Wrapped(LocalListBegin); } - Begin = CurrentIR.begin(); - while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + { + // Copy all of our IR ops over to the new location + for (auto &Block : GeneratedCodeBlocks) { + auto BlockIROp = Block.OldNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); - if (IROp->HasDest && RealNode->GetUses() == 0) { - // Should this be in a dedicated DCE pass? - ++Begin; - continue; + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + + CodeBlockData FirstNode{}; + CodeBlockData LastNode{}; + uint32_t i {}; + while (1) { + auto CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); + LogMan::Throw::A(IROp->Op != OP_IRHEADER, "%%ssa%d ended up being IRHeader. Shouldn't have hit this", CodeOp->ID()); + + size_t OpSize = FEXCore::IR::GetSize(IROp->Op); + + // Allocate the ops locally for our local dispatch + auto LocalPair = LocalBuilder.AllocateRawOp(OpSize); + IR::OrderedNodeWrapper LocalNodeWrapper = LocalPair.Node->Wrapped(LocalListBegin); + + // Copy over the op + memcpy(LocalPair.first, IROp, OpSize); + LogMan::Throw::A(LocalPair.first->Op == IROp->Op, "What. How did this fail"); + + // Set our map remapper to map the new location + // Even nodes that don't have a destination need to be in this map + // Need to be able to remap branch targets any other bits + NodeLocationRemapper[CodeOp->ID()] = LocalNodeWrapper.ID(); + if (i == 0) { + FirstNode.OldNode = CodeNode; + FirstNode.NewNode = LocalPair.Node; + } + + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + LastNode.OldNode = CodeNode; + LastNode.NewNode = LocalPair.Node; + break; + } + ++CodeBegin; + ++i; + } + + // Set the code block's begin and end correctly + auto NewBlockIROp = Block.NewNode->Op(LocalDataBegin)->CW(); + NewBlockIROp->Begin = FirstNode.NewNode->Wrapped(LocalListBegin); + NewBlockIROp->Last = LastNode.NewNode->Wrapped(LocalListBegin); } - - NodeWrapper LocalNodeWrapper = NodeWrapper::WrapOffset(NodeLocationRemapper[WrapperOp->ID()] * sizeof(OrderedNode)); - OrderedNode *LocalNode = reinterpret_cast(LocalNodeWrapper.GetPtr(LocalListBegin)); - FEXCore::IR::IROp_Header *LocalIROp = LocalNode->Op(LocalDataBegin); - - // Now that we have the op copied over, we need to modify SSA values to point to the new correct locations - for (uint8_t i = 0; i < IROp->NumArgs; ++i) { - NodeWrapper OldArg = IROp->Args[i]; - LogMan::Throw::A(NodeLocationRemapper[OldArg.ID()] != ~0U, "Tried remapping unfound node"); - LocalIROp->Args[i].NodeOffset = NodeLocationRemapper[OldArg.ID()] * sizeof(OrderedNode); - } - ++Begin; } -// uintptr_t OldListSize = CurrentIR.GetListSize(); -// uintptr_t OldDataSize = CurrentIR.GetDataSize(); + { + // Fixup the arguments of all the IROps + for (auto &Block : GeneratedCodeBlocks) { + auto BlockIROp = Block.OldNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + while (1) { + auto CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); + + OrderedNodeWrapper LocalNodeWrapper = OrderedNodeWrapper::WrapOffset(NodeLocationRemapper[CodeOp->ID()] * sizeof(OrderedNode)); + OrderedNode *LocalNode = LocalNodeWrapper.GetNode(LocalListBegin); + FEXCore::IR::IROp_Header *LocalIROp = LocalNode->Op(LocalDataBegin); + + // Now that we have the op copied over, we need to modify SSA values to point to the new correct locations + for (uint8_t i = 0; i < IROp->NumArgs; ++i) { + uint32_t OldArg = IROp->Args[i].ID(); + LogMan::Throw::A(NodeLocationRemapper[OldArg] != ~0U, "Tried remapping unfound node %%ssa%d", OldArg); + LocalIROp->Args[i].NodeOffset = NodeLocationRemapper[OldArg] * sizeof(OrderedNode); + } + + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; + } + } + } + +// XXX: Example for iterating blocks +// OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); +// while (1) { +// auto BlockIROp = BlockNode->Op(DataBegin)->CW(); +// LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); // -// uintptr_t NewListSize = LocalIR.GetListSize(); -// uintptr_t NewDataSize = LocalIR.GetDataSize(); +// // We grab these nodes this way so we can iterate easily +// auto CodeBegin = CurrentIR.at(BlockIROp->Begin); +// auto CodeLast = CurrentIR.at(BlockIROp->Last); +// while (1) { +// auto CodeOp = CodeBegin(); +// OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); +// auto IROp = CodeNode->Op(DataBegin); // -// if (NewListSize < OldListSize || -// NewDataSize < OldDataSize) { -// if (NewListSize < OldListSize) { -// LogMan::Msg::D("Shaved %ld bytes off the list size", OldListSize - NewListSize); +// // CodeLast is inclusive. So we still need to dump the CodeLast op as well +// if (CodeBegin == CodeLast) { +// break; +// } +// ++CodeBegin; // } -// if (NewDataSize < OldDataSize) { -// LogMan::Msg::D("Shaved %ld bytes off the data size", OldDataSize - NewDataSize); +// +// if (BlockIROp->Next.ID() == 0) { +// break; +// } else { +// BlockNode = BlockIROp->Next.GetNode(ListBegin); // } // } -// if (NewListSize > OldListSize || -// NewDataSize > OldDataSize) { -// LogMan::Msg::A("Whoa. Compaction made the IR a different size when it shouldn't have. 0x%lx > 0x%lx or 0x%lx > 0x%lx",NewListSize, OldListSize, NewDataSize, OldDataSize); -// } + // uintptr_t OldListSize = CurrentIR.GetListSize(); + // uintptr_t OldDataSize = CurrentIR.GetDataSize(); + + // uintptr_t NewListSize = LocalIR.GetListSize(); + // uintptr_t NewDataSize = LocalIR.GetDataSize(); + + // if (NewListSize < OldListSize || + // NewDataSize < OldDataSize) { + // if (NewListSize < OldListSize) { + // LogMan::Msg::D("Shaved %ld bytes off the list size", OldListSize - NewListSize); + // } + // if (NewDataSize < OldDataSize) { + // LogMan::Msg::D("Shaved %ld bytes off the data size", OldDataSize - NewDataSize); + // } + // } + + // if (NewListSize > OldListSize || + // NewDataSize > OldDataSize) { + // LogMan::Msg::A("Whoa. Compaction made the IR a different size when it shouldn't have. 0x%lx > 0x%lx or 0x%lx > 0x%lx",NewListSize, OldListSize, NewDataSize, OldDataSize); + // } Disp->CopyData(LocalBuilder); diff --git a/Source/Interface/IR/Passes/IRValidation.cpp b/Source/Interface/IR/Passes/IRValidation.cpp index 7e2bc710b..a7b6b7d70 100644 --- a/Source/Interface/IR/Passes/IRValidation.cpp +++ b/Source/Interface/IR/Passes/IRValidation.cpp @@ -6,13 +6,13 @@ namespace FEXCore::IR::Validation { struct BlockInfo { - IR::NodeWrapper *Begin; - IR::NodeWrapper *End; + IR::OrderedNodeWrapper *Begin; + IR::OrderedNodeWrapper *End; bool HasExit; - std::vector Predecessors; - std::vector Successors; + std::vector Predecessors; + std::vector Successors; }; class IRValidation final : public FEXCore::IR::Pass { @@ -20,7 +20,7 @@ public: bool Run(OpDispatchBuilder *Disp) override; private: - std::unordered_map OffsetToBlockMap; + std::unordered_map OffsetToBlockMap; }; bool IRValidation::Run(OpDispatchBuilder *Disp) { @@ -37,8 +37,8 @@ bool IRValidation::Run(OpDispatchBuilder *Disp) { std::ostringstream Errors; while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); uint8_t OpSize = IROp->Size; @@ -56,7 +56,7 @@ bool IRValidation::Run(OpDispatchBuilder *Disp) { } for (uint8_t i = 0; i < IROp->NumArgs; ++i) { - NodeWrapper Arg = IROp->Args[i]; + OrderedNodeWrapper Arg = IROp->Args[i]; if (Arg.ID() == 0) { HadError |= true; Errors << "Op" << WrapperOp->ID() <<": Arg[" << i << "] has invalid target of %ssa0" << std::endl; @@ -70,7 +70,7 @@ bool IRValidation::Run(OpDispatchBuilder *Disp) { Errors << "BasicBlock " << WrapperOp->ID() << ": Begin in middle of block" << std::endl; } - auto Block = OffsetToBlockMap.try_emplace(WrapperOp->ID(), BlockInfo{}).first; + auto Block = OffsetToBlockMap.try_emplace(WrapperOp->ID()).first; CurrentBlock = &Block->second; CurrentBlock->Begin = WrapperOp; InBlock = true; @@ -109,14 +109,14 @@ bool IRValidation::Run(OpDispatchBuilder *Disp) { CurrentBlock->Successors.emplace_back(IterLocation()); } - OrderedNode *TargetNode = reinterpret_cast(IterLocation()->GetPtr(ListBegin)); + OrderedNode *TargetNode = IterLocation()->GetNode(ListBegin); FEXCore::IR::IROp_Header *TargetOp = TargetNode->Op(DataBegin); HadError |= TargetOp->Op != OP_BEGINBLOCK; if (TargetOp->Op != OP_BEGINBLOCK) { Errors << "CondJump " << WrapperOp->ID() << ": CondJump to Op that isn't the begining of a block" << std::endl; } else { - auto Block = OffsetToBlockMap.try_emplace(IterLocation()->NodeOffset, BlockInfo{}).first; + auto Block = OffsetToBlockMap.try_emplace(IterLocation()->NodeOffset).first; Block->second.Predecessors.emplace_back(CurrentBlock->Begin); } @@ -130,14 +130,14 @@ bool IRValidation::Run(OpDispatchBuilder *Disp) { CurrentBlock->Successors.emplace_back(IterLocation()); } - OrderedNode *TargetNode = reinterpret_cast(IterLocation()->GetPtr(ListBegin)); + OrderedNode *TargetNode = IterLocation()->GetNode(ListBegin); FEXCore::IR::IROp_Header *TargetOp = TargetNode->Op(DataBegin); HadError |= TargetOp->Op != OP_BEGINBLOCK; if (TargetOp->Op != OP_BEGINBLOCK) { Errors << "Jump " << WrapperOp->ID() << ": Jump to Op that isn't the begining of a block" << std::endl; } else { - auto Block = OffsetToBlockMap.try_emplace(IterLocation()->NodeOffset, BlockInfo{}).first; + auto Block = OffsetToBlockMap.try_emplace(IterLocation()->NodeOffset).first; Block->second.Predecessors.emplace_back(CurrentBlock->Begin); } break; diff --git a/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index 51df22214..22dff4862 100644 --- a/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -9,51 +9,70 @@ public: }; bool RedundantFlagCalculationEliminination::Run(OpDispatchBuilder *Disp) { + std::array LastValidFlagStores{}; + bool Changed = false; auto CurrentIR = Disp->ViewIR(); uintptr_t ListBegin = CurrentIR.GetListData(); uintptr_t DataBegin = CurrentIR.GetData(); - IR::NodeWrapperIterator Begin = CurrentIR.begin(); - IR::NodeWrapperIterator End = CurrentIR.end(); + auto Begin = CurrentIR.begin(); + auto Op = Begin(); - std::array LastValidFlagStores{}; + OrderedNode *RealNode = Op->GetNode(ListBegin); + auto HeaderOp = RealNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); - while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); - if (IROp->Op == OP_BEGINBLOCK || - IROp->Op == OP_ENDBLOCK || - IROp->Op == OP_JUMP || - IROp->Op == OP_CONDJUMP || - IROp->Op == OP_EXITFUNCTION) { - // We don't track across block boundaries - LastValidFlagStores.fill(nullptr); - } + while (1) { + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); - if (IROp->Op == OP_STOREFLAG) { - auto Op = IROp->CW(); + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + while (1) { + auto CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); - // If we have had a valid flag store previously and it hasn't been touched until this new store - // Then just delete the old one and let DCE to take care of the rest - if (LastValidFlagStores[Op->Flag] != nullptr) { - Disp->Unlink(LastValidFlagStores[Op->Flag]); - Changed = true; + if (IROp->Op == OP_STOREFLAG) { + auto Op = IROp->CW(); + + // If we have had a valid flag store previously and it hasn't been touched until this new store + // Then just delete the old one and let DCE to take care of the rest + if (LastValidFlagStores[Op->Flag] != nullptr) { + Disp->Unlink(LastValidFlagStores[Op->Flag]); + Changed = true; + } + + // Set this node as the last one valid for this flag + LastValidFlagStores[Op->Flag] = RealNode; + } + else if (IROp->Op == OP_LOADFLAG) { + auto Op = IROp->CW(); + + // If we loaded a flag then we can't track past this + LastValidFlagStores[Op->Flag] = nullptr; } - // Set this node as the last one valid for this flag - LastValidFlagStores[Op->Flag] = RealNode; - } - else if (IROp->Op == OP_LOADFLAG) { - auto Op = IROp->CW(); - // If we loaded a flag then we can't track past this - LastValidFlagStores[Op->Flag] = nullptr; + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; } - ++Begin; + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + } + + // We don't track across block boundaries + LastValidFlagStores.fill(nullptr); } return Changed; diff --git a/Source/Interface/IR/Passes/RegisterAllocationPass.cpp b/Source/Interface/IR/Passes/RegisterAllocationPass.cpp index 84f391d0a..8d8bdd53d 100644 --- a/Source/Interface/IR/Passes/RegisterAllocationPass.cpp +++ b/Source/Interface/IR/Passes/RegisterAllocationPass.cpp @@ -1,15 +1,13 @@ #include "Common/BitSet.h" -#include "Common/Profiler.h" #include "Interface/IR/Passes/RegisterAllocationPass.h" #include "Interface/Core/OpcodeDispatcher.h" #include -PROFILER_DEFINE(RA, "Passes", "RA", 0); - namespace FEXCore::IR { constexpr uint32_t INVALID_REG = ~0U; constexpr uint32_t INVALID_CLASS = ~0U; + constexpr uint32_t DEFAULT_INTERFERENCE_LIST_SIZE = 128; struct Register { }; @@ -17,7 +15,7 @@ namespace FEXCore::IR { struct RegisterClass { uint32_t RegisterBase; uint32_t NumberOfRegisters{0}; - BitSet Registers; + BitSet Registers; }; struct RegisterAllocationPass::RegisterNode { @@ -26,7 +24,7 @@ namespace FEXCore::IR { uint32_t InterferenceCount; uint32_t InterferenceListSize; uint32_t *InterferenceList; - BitSet Interference; + BitSet Interference; }; static_assert(std::is_pod::value, "We want this to be POD"); @@ -96,7 +94,7 @@ namespace FEXCore::IR { for (uint32_t i = 0; i < NodeCount; ++i) { Graph->Nodes[i].Register = INVALID_REG; Graph->Nodes[i].RegisterClass = INVALID_CLASS; - Graph->Nodes[i].InterferenceListSize = 32; + Graph->Nodes[i].InterferenceListSize = DEFAULT_INTERFERENCE_LIST_SIZE; Graph->Nodes[i].InterferenceList = reinterpret_cast(calloc(Graph->Nodes[i].InterferenceListSize, sizeof(uint32_t))); Graph->Nodes[i].InterferenceCount = 0; Graph->Nodes[i].Interference.Allocate(NodeCount); @@ -125,7 +123,7 @@ namespace FEXCore::IR { for (uint32_t i = OldNodeCount; i < NodeCount; ++i) { Graph->Nodes[i].Register = INVALID_REG; Graph->Nodes[i].RegisterClass = INVALID_CLASS; - Graph->Nodes[i].InterferenceListSize = 32; + Graph->Nodes[i].InterferenceListSize = DEFAULT_INTERFERENCE_LIST_SIZE; Graph->Nodes[i].InterferenceList = reinterpret_cast(calloc(Graph->Nodes[i].InterferenceListSize, sizeof(uint32_t))); Graph->Nodes[i].InterferenceCount = 0; Graph->Nodes[i].Interference.Allocate(NodeCount); @@ -165,22 +163,6 @@ namespace FEXCore::IR { Graph->Nodes[Node].RegisterClass = Class; } - void RegisterAllocationPass::AddNodeInterference(uint32_t Node1, uint32_t Node2) { - auto AddInterference = [&](uint32_t Node1, uint32_t Node2) { - RegisterNode *Node = &Graph->Nodes[Node1]; - Node->Interference.Set(Node2); - if (Node->InterferenceListSize <= Node->InterferenceCount) { - Node->InterferenceListSize *= 2; - Node->InterferenceList = reinterpret_cast(realloc(Node->InterferenceList, Node->InterferenceListSize * sizeof(uint32_t))); - } - Node->InterferenceList[Node->InterferenceCount] = Node2; - ++Node->InterferenceCount; - }; - - AddInterference(Node1, Node2); - AddInterference(Node2, Node1); - } - uint32_t RegisterAllocationPass::GetNodeRegister(uint32_t Node) { return Graph->Nodes[Node].Register; } @@ -210,8 +192,8 @@ namespace FEXCore::IR { while (Begin != End) { using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); if (IROp->HasDest) { @@ -266,7 +248,8 @@ namespace FEXCore::IR { } } - void RegisterAllocationPass::CalculateLiveRange(IRListView *CurrentIR, uint32_t Nodes) { + void RegisterAllocationPass::CalculateLiveRange(IRListView *CurrentIR) { + size_t Nodes = CurrentIR->GetSSACount(); if (Nodes > LiveRanges.size()) { LiveRanges.resize(Nodes); } @@ -281,14 +264,14 @@ namespace FEXCore::IR { constexpr uint32_t DEFAULT_REMAT_COST = 1000; while (Begin != End) { using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); - uint32_t Node = WrapperOp->ID(); // If the destination hasn't yet been set then set it now - if (IROp->HasDest && LiveRanges[Node].Begin == ~0U) { + if (IROp->HasDest) { + LogMan::Throw::A(LiveRanges[Node].Begin == ~0U, "Node begin already defined?"); LiveRanges[Node].Begin = Node; // Default to ending right where it starts LiveRanges[Node].End = Node; @@ -312,13 +295,27 @@ namespace FEXCore::IR { ++Begin; } + } + + void RegisterAllocationPass::CalculateNodeInterference(uint32_t NodeCount) { + auto AddInterference = [&](uint32_t Node1, uint32_t Node2) { + RegisterNode *Node = &Graph->Nodes[Node1]; + Node->Interference.Set(Node2); + if (Node->InterferenceListSize <= Node->InterferenceCount) { + Node->InterferenceListSize *= 2; + Node->InterferenceList = reinterpret_cast(realloc(Node->InterferenceList, Node->InterferenceListSize * sizeof(uint32_t))); + } + Node->InterferenceList[Node->InterferenceCount] = Node2; + ++Node->InterferenceCount; + }; // Now that we have all the live ranges calculated we need to add them to our interference graph - for (uint32_t i = 0; i < Nodes; ++i) { - for (uint32_t j = i + 1; j < Nodes; ++j) { + for (uint32_t i = 0; i < NodeCount; ++i) { + for (uint32_t j = i + 1; j < NodeCount; ++j) { if (!(LiveRanges[i].Begin >= LiveRanges[j].End || LiveRanges[j].Begin >= LiveRanges[i].End)) { - AddNodeInterference(i, j); + AddInterference(i, j); + AddInterference(j, i); } } } @@ -342,10 +339,10 @@ namespace FEXCore::IR { if (Reg == ~0U) { auto RegisterNode = GetRegisterNode(i); - LogMan::Msg::E("\t%%ssa%d with no-RA has live range [%d, %d): Remat cost: %d", i, LiveRanges[i].Begin, LiveRanges[i].End, LiveRanges[i].RematCost); + // LogMan::Msg::E("\t%%ssa%d with no-RA has live range [%d, %d): Remat cost: %d", i, LiveRanges[i].Begin, LiveRanges[i].End, LiveRanges[i].RematCost); for (uint32_t j = 0; j < RegisterNode->InterferenceCount; ++j) { uint32_t InterferenceNode = RegisterNode->InterferenceList[j]; - LogMan::Msg::E("\t\tInterferes with %%ssa%d: live range[%d, %d): Remat cost: %d", InterferenceNode, LiveRanges[InterferenceNode].Begin, LiveRanges[InterferenceNode].End, LiveRanges[InterferenceNode].RematCost); + // LogMan::Msg::E("\t\tInterferes with %%ssa%d: live range[%d, %d): Remat cost: %d", InterferenceNode, LiveRanges[InterferenceNode].Begin, LiveRanges[InterferenceNode].End, LiveRanges[InterferenceNode].RematCost); } Graph->SpillStack.emplace_back(SpillStackUnit{i, CurrentNode->RegisterClass}); } @@ -375,8 +372,8 @@ namespace FEXCore::IR { while (Begin != End) { using namespace FEXCore::IR; - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); LogMan::Msg::D("\t\t%%ssa%d Do we have %%ssa%d", WrapperOp->ID(), Node->Wrapped(ListBegin).ID()); @@ -406,8 +403,8 @@ namespace FEXCore::IR { auto LastCursor = Disp->GetWriteCursor(); while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); + OrderedNodeWrapper *WrapperOp = Begin(); + OrderedNode *RealNode = WrapperOp->GetNode(ListBegin); FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); auto Iter = IsInSpillStack(&Graph->SpillStack, WrapperOp->ID()); if (Iter != Graph->SpillStack.end()) { @@ -424,8 +421,8 @@ namespace FEXCore::IR { // We want to end the live range of this value here and continue it on first use auto ConstantRegisterNode = GetRegisterNode(InterferenceNode); auto *ConstantLiveRange = &LiveRanges[InterferenceNode]; - NodeWrapper ConstantOp = NodeWrapper::WrapOffset(InterferenceNode * sizeof(OrderedNode)); - OrderedNode *ConstantNode = reinterpret_cast(ConstantOp.GetPtr(ListBegin)); + OrderedNodeWrapper ConstantOp = OrderedNodeWrapper::WrapOffset(InterferenceNode * sizeof(OrderedNode)); + OrderedNode *ConstantNode = ConstantOp.GetNode(ListBegin); FEXCore::IR::IROp_Constant const *ConstantIROp = ConstantNode->Op(DataBegin)->C(); LogMan::Throw::A(ConstantIROp->Header.Op == OP_CONSTANT, "This needs to be const"); @@ -435,8 +432,8 @@ namespace FEXCore::IR { auto FirstUseLocation = FindFirstUse(Disp, ConstantNode, NextIter, End); if (FirstUseLocation != End) { // LogMan::Throw::A(FirstUseLocation != End, "Failure to find op use"); - NodeWrapper *FirstUseOp = FirstUseLocation(); - OrderedNode *FirstUseOrderedNode = reinterpret_cast(FirstUseOp->GetPtr(ListBegin)); + OrderedNodeWrapper *FirstUseOp = FirstUseLocation(); + OrderedNode *FirstUseOrderedNode = FirstUseOp->GetNode(ListBegin); Disp->SetWriteCursor(FirstUseOrderedNode); auto FilledConstant = Disp->_Constant(ConstantIROp->Constant); Disp->ReplaceAllUsesWithInclusive(ConstantNode, FilledConstant, FirstUseLocation, End); @@ -507,13 +504,13 @@ namespace FEXCore::IR { auto *InterferenceLiveRange = &LiveRanges[InterferenceNode]; // If the interference's live range is past this op's live range then we can dump it if (1) { - NodeWrapper InterferenceOp = NodeWrapper::WrapOffset(InterferenceNode * sizeof(OrderedNode)); - OrderedNode *InterferenceOrderedNode = reinterpret_cast(InterferenceOp.GetPtr(ListBegin)); + OrderedNodeWrapper InterferenceOp = OrderedNodeWrapper::WrapOffset(InterferenceNode * sizeof(OrderedNode)); + OrderedNode *InterferenceOrderedNode = InterferenceOp.GetNode(ListBegin); FEXCore::IR::IROp_Header *InterferenceIROp = InterferenceOrderedNode->Op(DataBegin); auto PrevIter = Begin; --PrevIter; - Disp->SetWriteCursor(reinterpret_cast(PrevIter()->GetPtr(ListBegin))); + Disp->SetWriteCursor(PrevIter()->GetNode(ListBegin)); auto SpillOp = Disp->_SpillRegister(InterferenceOrderedNode, SpillSlotCount, {InterferenceRegisterNode->RegisterClass}); SpillOp.first->Header.Size = InterferenceIROp->Size; SpillOp.first->Header.Elements = InterferenceIROp->Elements; @@ -525,8 +522,8 @@ namespace FEXCore::IR { auto FirstUseLocation = FindFirstUse(Disp, InterferenceOrderedNode, NextIter, End); if (FirstUseLocation != End) { // LogMan::Throw::A(FirstUseLocation != End, "Failure to find op use"); - NodeWrapper *FirstUseOp = FirstUseLocation(); - OrderedNode *FirstUseOrderedNode = reinterpret_cast(FirstUseOp->GetPtr(ListBegin)); + OrderedNodeWrapper *FirstUseOp = FirstUseLocation(); + OrderedNode *FirstUseOrderedNode = FirstUseOp->GetNode(ListBegin); Disp->SetWriteCursor(FirstUseOrderedNode); auto FilledInterference = Disp->_FillRegister(SpillSlotCount, {InterferenceRegisterNode->RegisterClass}); @@ -558,32 +555,39 @@ namespace FEXCore::IR { } bool RegisterAllocationPass::Run(OpDispatchBuilder *Disp) { - PROFILER_SCOPE(RA); - PROFILER_COUNTER_ADD("RA/counter", 1); bool Changed = false; constexpr uint32_t RATries = 1; SpillSlotCount = 0; HasSpills = false; + HadFullRA = false; for (uint32_t i = 0; i < RATries; ++i) { auto CurrentIR = Disp->ViewIR(); uintptr_t ListSize = CurrentIR.GetListSize(); - uint32_t SSACount = ListSize / sizeof(IR::OrderedNode); + uint32_t SSACount = CurrentIR.GetSSACount(); ResetRegisterGraph(SSACount); FindNodeClasses(&CurrentIR); - CalculateLiveRange(&CurrentIR, SSACount); + CalculateLiveRange(&CurrentIR); + CalculateNodeInterference(SSACount); AllocateRegisters(); if (!Graph->SpillStack.empty()) { - Disp->ShouldDump = true; - Changed = true; - //ClearSpillList(Disp); + if (Config_SupportsSpills) { + Disp->ShouldDump = true; + Changed = true; + ClearSpillList(Disp); + } + else { + HadFullRA = false; + SpillSlotCount = 0; + } return Changed; } else { // We managed to RA, leave now + HadFullRA = true; Disp->ShouldDump = false; return Changed; } diff --git a/Source/Interface/IR/Passes/RegisterAllocationPass.h b/Source/Interface/IR/Passes/RegisterAllocationPass.h index e444679a7..ab6bdbcb8 100644 --- a/Source/Interface/IR/Passes/RegisterAllocationPass.h +++ b/Source/Interface/IR/Passes/RegisterAllocationPass.h @@ -30,16 +30,18 @@ public: void FreeRegisterGraph(); void ResetRegisterGraph(uint32_t NodeCount); void SetNodeClass(uint32_t Node, uint32_t Class); - void AddNodeInterference(uint32_t Node1, uint32_t Node2); uint32_t GetNodeRegister(uint32_t Node); void AllocateRegisters(); /** @} */ + bool HasFullRA() const { return HadFullRA; } bool HadSpills() const { return HasSpills; } uint32_t SpillSlots() const { return SpillSlotCount; } + void SetSupportsSpills(bool Supports) { Config_SupportsSpills = Supports; } + private: RegisterGraph *Graph; void FindNodeClasses(IRListView *CurrentIR); @@ -52,7 +54,8 @@ private: std::vector LiveRanges; - void CalculateLiveRange(IRListView *CurrentIR, uint32_t Nodes); + void CalculateLiveRange(IRListView *CurrentIR); + void CalculateNodeInterference(uint32_t NodeCount); void ClearSpillList(OpDispatchBuilder *Disp); @@ -61,6 +64,9 @@ private: bool HasSpills {}; uint32_t SpillSlotCount {}; + bool HadFullRA {}; + + bool Config_SupportsSpills {true}; }; } diff --git a/Source/Interface/IR/Passes/SyscallOptimization.cpp b/Source/Interface/IR/Passes/SyscallOptimization.cpp index 98b79d74d..34e87fbaf 100644 --- a/Source/Interface/IR/Passes/SyscallOptimization.cpp +++ b/Source/Interface/IR/Passes/SyscallOptimization.cpp @@ -16,24 +16,51 @@ bool SyscallOptimization::Run(OpDispatchBuilder *Disp) { uintptr_t ListBegin = CurrentIR.GetListData(); uintptr_t DataBegin = CurrentIR.GetData(); - IR::NodeWrapperIterator Begin = CurrentIR.begin(); - IR::NodeWrapperIterator End = CurrentIR.end(); + auto Begin = CurrentIR.begin(); + auto Op = Begin(); - while (Begin != End) { - NodeWrapper *WrapperOp = Begin(); - OrderedNode *RealNode = reinterpret_cast(WrapperOp->GetPtr(ListBegin)); - FEXCore::IR::IROp_Header *IROp = RealNode->Op(DataBegin); + OrderedNode *RealNode = Op->GetNode(ListBegin); + auto HeaderOp = RealNode->Op(DataBegin)->CW(); + LogMan::Throw::A(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); + + OrderedNode *BlockNode = HeaderOp->Blocks.GetNode(ListBegin); + + while (1) { + auto BlockIROp = BlockNode->Op(DataBegin)->CW(); + LogMan::Throw::A(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); + + // We grab these nodes this way so we can iterate easily + auto CodeBegin = CurrentIR.at(BlockIROp->Begin); + auto CodeLast = CurrentIR.at(BlockIROp->Last); + while (1) { + auto CodeOp = CodeBegin(); + OrderedNode *CodeNode = CodeOp->GetNode(ListBegin); + auto IROp = CodeNode->Op(DataBegin); + + if (IROp->Op == FEXCore::IR::OP_SYSCALL) { + Disp->ShouldDump = true; + // Is the first argument a constant? + uint64_t Constant; + if (Disp->IsValueConstant(IROp->Args[0], &Constant)) { + LogMan::Msg::D("Whoa. Syscall argument is constant: %ld", Constant); + Changed = true; + } - if (IROp->Op == FEXCore::IR::OP_SYSCALL) { - // Is the first argument a constant? - uint64_t Constant; - if (Disp->IsValueConstant(IROp->Args[0], &Constant)) { - // LogMan::Msg::A("Whoa. Syscall argument is constant: %ld", Constant); - Changed = true; } + // CodeLast is inclusive. So we still need to dump the CodeLast op as well + if (CodeBegin == CodeLast) { + break; + } + ++CodeBegin; } - ++Begin; + + if (BlockIROp->Next.ID() == 0) { + break; + } else { + BlockNode = BlockIROp->Next.GetNode(ListBegin); + } + } return Changed; diff --git a/include/FEXCore/Core/CodeLoader.h b/include/FEXCore/Core/CodeLoader.h index c0d750035..a4f1a8865 100644 --- a/include/FEXCore/Core/CodeLoader.h +++ b/include/FEXCore/Core/CodeLoader.h @@ -34,6 +34,9 @@ public: */ virtual uint64_t DefaultRIP() const = 0; + virtual void GetInitLocations(std::vector *Locations) {} + virtual uint64_t InitializeThreadSlot(std::function Writer) const { return 0; }; + using MemoryLayout = std::tuple; /** * @brief Gets the default memory layout of the memory object being loaded diff --git a/include/FEXCore/IR/IR.h b/include/FEXCore/IR/IR.h index a699ae636..04dc97de9 100644 --- a/include/FEXCore/IR/IR.h +++ b/include/FEXCore/IR/IR.h @@ -6,8 +6,19 @@ namespace FEXCore::IR { +/** + * @brief The IROp_Header is an dynamically sized array + * At the end it contains a uint8_t for the number of arguments that Op has + * Then there is an unsized array of NodeWrapper arguments for the number of arguments this op has + * The op structures that are including the header must ensure that they pad themselves correctly to the number of arguments used + */ +struct IROp_Header; + +class OrderedNode; /** * @brief This is a very simple wrapper for our node pointers + * You probably don't want to use this directly + * Use OpNodeWrapper and OrderedNodeWrapper types below instead * * This is necessary to allow two things * - Reduce memory usage by having the pointer be an 32bit offset rather than the whole 64bit pointer @@ -19,46 +30,53 @@ namespace FEXCore::IR { * - We have to have the base offset live somewhere else * - Has to be POD and trivially copyable * - Makes every real node access turn in to a [Base + Offset] access + * - Can be confusing if you're mixing OpNodeWrapper and OrderedNodeWrapper usage */ -struct NodeWrapper final { +template +struct NodeWrapperBase final { // On x86-64 using a uint64_t type is more efficient since RIP addressing gives you [ + + ] // On AArch64 using uint32_t is just more memory efficient. 32bit or 64bit offset doesn't matter // We use uint32_t to be more memory efficient (Cuts our node list size in half) using NodeOffsetType = uint32_t; NodeOffsetType NodeOffset; - static NodeWrapper WrapOffset(NodeOffsetType Offset) { - NodeWrapper Wrapped; + static NodeWrapperBase WrapOffset(NodeOffsetType Offset) { + NodeWrapperBase Wrapped; Wrapped.NodeOffset = Offset; return Wrapped; } - static NodeWrapper WrapPtr(uintptr_t Base, uintptr_t Value) { - NodeWrapper Wrapped; + static NodeWrapperBase WrapPtr(uintptr_t Base, uintptr_t Value) { + NodeWrapperBase Wrapped; Wrapped.SetOffset(Base, Value); return Wrapped; } - static void *UnwrapNode(uintptr_t Base, NodeWrapper Node) { - return Node.GetPtr(Base); + static void *UnwrapNode(uintptr_t Base, NodeWrapperBase Node) { + return Node.GetNode(Base); } uint32_t ID() const; - explicit NodeWrapper() = default; - void *GetPtr(uintptr_t Base) { return reinterpret_cast(Base + NodeOffset); } - void const *GetPtr(uintptr_t Base) const { return reinterpret_cast(Base + NodeOffset); } + explicit NodeWrapperBase() = default; + + Type *GetNode(uintptr_t Base) { return reinterpret_cast(Base + NodeOffset); } + Type const *GetNode(uintptr_t Base) const { return reinterpret_cast(Base + NodeOffset); } + void SetOffset(uintptr_t Base, uintptr_t Value) { NodeOffset = Value - Base; } - bool operator==(NodeWrapper const &rhs) { return NodeOffset == rhs.NodeOffset; } + bool operator==(NodeWrapperBase const &rhs) { return NodeOffset == rhs.NodeOffset; } }; -static_assert(std::is_pod::value); -static_assert(sizeof(NodeWrapper) == sizeof(uint32_t)); +static_assert(std::is_pod>::value); +static_assert(sizeof(NodeWrapperBase) == sizeof(uint32_t)); + +using OpNodeWrapper = NodeWrapperBase; +using OrderedNodeWrapper = NodeWrapperBase; struct OrderedNodeHeader { - NodeWrapper Value; - NodeWrapper Next; - NodeWrapper Previous; + OpNodeWrapper Value; + OrderedNodeWrapper Next; + OrderedNodeWrapper Previous; }; static_assert(sizeof(OrderedNodeHeader) == sizeof(uint32_t) * 3); @@ -70,7 +88,7 @@ static_assert(sizeof(OrderedNodeHeader) == sizeof(uint32_t) * 3); */ class NodeWrapperIterator final { public: - using value_type = NodeWrapper; + using value_type = OrderedNodeWrapper; using size_type = std::size_t; using difference_type = std::ptrdiff_t; using reference = value_type&; @@ -83,9 +101,9 @@ public: using const_reverse_iterator = const_iterator; using iterator_category = std::bidirectional_iterator_tag; - using NodeType = NodeWrapper; - using NodePtr = NodeWrapper*; - using NodeRef = NodeWrapper&; + using NodeType = value_type; + using NodePtr = value_type*; + using NodeRef = value_type&; NodeWrapperIterator(uintptr_t Base) : BaseList {Base} {} explicit NodeWrapperIterator(uintptr_t Base, NodeType Ptr) : BaseList {Base}, Node {Ptr} {} @@ -99,13 +117,13 @@ public: } NodeWrapperIterator operator++() { - OrderedNodeHeader *RealNode = reinterpret_cast(Node.GetPtr(BaseList)); + OrderedNodeHeader *RealNode = reinterpret_cast(Node.GetNode(BaseList)); Node = RealNode->Next; return *this; } NodeWrapperIterator operator--() { - OrderedNodeHeader *RealNode = reinterpret_cast(Node.GetPtr(BaseList)); + OrderedNodeHeader *RealNode = reinterpret_cast(Node.GetNode(BaseList)); Node = RealNode->Previous; return *this; } @@ -123,14 +141,6 @@ private: NodeType Node{}; }; -/** - * @brief The IROp_Header is an dynamically sized array - * At the end it contains a uint8_t for the number of arguments that Op has - * Then there is an unsized array of NodeWrapper arguments for the number of arguments this op has - * The op structures that are including the header must ensure that they pad themselves correctly to the number of arguments used - */ -struct IROp_Header; - /** * @brief This is a node in our IR representation * Is a doubly linked list node that lives in a representation of a linearly allocated node list @@ -153,6 +163,8 @@ class OrderedNode final { OrderedNodeHeader Header; uint32_t NumUses; + using value_type = OrderedNodeWrapper; + OrderedNode() = default; /** @@ -163,7 +175,7 @@ class OrderedNode final { * * @return Pointer to the node being added */ - NodeWrapper append(uintptr_t Base, NodeWrapper Node) { + value_type append(uintptr_t Base, value_type Node) { // Set Next Node's Previous to incoming node SetPrevious(Base, Header.Next, Node); @@ -179,7 +191,7 @@ class OrderedNode final { } OrderedNode *append(uintptr_t Base, OrderedNode *Node) { - NodeWrapper WNode = Node->Wrapped(Base); + value_type WNode = Node->Wrapped(Base); // Set Next Node's Previous to incoming node SetPrevious(Base, Header.Next, WNode); @@ -201,7 +213,7 @@ class OrderedNode final { * * @return Pointer to the node being added */ - NodeWrapper prepend(uintptr_t Base, NodeWrapper Node) { + value_type prepend(uintptr_t Base, value_type Node) { // Set the previous node's next to the incoming node SetNext(Base, Header.Previous, Node); @@ -217,7 +229,7 @@ class OrderedNode final { } OrderedNode *prepend(uintptr_t Base, OrderedNode *Node) { - NodeWrapper WNode = Node->Wrapped(Base); + value_type WNode = Node->Wrapped(Base); // Set the previous node's next to the incoming node SetNext(Base, Header.Previous, WNode); @@ -241,10 +253,10 @@ class OrderedNode final { size_t size(uintptr_t Base) const { size_t Size = 1; // Walk the list forward until we hit a sentinal - NodeWrapper Current = Header.Next; + value_type Current = Header.Next; while (Current.NodeOffset != 0) { ++Size; - OrderedNode *RealNode = reinterpret_cast(Current.GetPtr(Base)); + OrderedNode *RealNode = Current.GetNode(Base); Current = RealNode->Header.Next; } return Size; @@ -258,41 +270,36 @@ class OrderedNode final { SetPrevious(Base, Header.Next, Header.Previous); } - IROp_Header const* Op(uintptr_t Base) const { return reinterpret_cast(Header.Value.GetPtr(Base)); } - IROp_Header *Op(uintptr_t Base) { return reinterpret_cast(Header.Value.GetPtr(Base)); } + IROp_Header const* Op(uintptr_t Base) const { return Header.Value.GetNode(Base); } + IROp_Header *Op(uintptr_t Base) { return Header.Value.GetNode(Base); } uint32_t GetUses() const { return NumUses; } void AddUse() { ++NumUses; } void RemoveUse() { --NumUses; } - using iterator = NodeWrapperIterator; - - iterator begin(uint64_t Base) noexcept { return iterator(Base, Wrapped(Base)); } - iterator end(uint64_t Base, uint64_t End) noexcept { return iterator(Base, WrappedOffset(End)); } - - NodeWrapper Wrapped(uintptr_t Base) { - NodeWrapper Tmp; + value_type Wrapped(uintptr_t Base) { + value_type Tmp; Tmp.SetOffset(Base, reinterpret_cast(this)); return Tmp; } private: - NodeWrapper WrappedOffset(uint32_t Offset) { - NodeWrapper Tmp; + value_type WrappedOffset(uint32_t Offset) { + value_type Tmp; Tmp.NodeOffset = Offset; return Tmp; } - static void SetPrevious(uintptr_t Base, NodeWrapper Node, NodeWrapper New) { + static void SetPrevious(uintptr_t Base, value_type Node, value_type New) { if (Node.NodeOffset == 0) return; - OrderedNode *RealNode = reinterpret_cast(Node.GetPtr(Base)); + OrderedNode *RealNode = Node.GetNode(Base); RealNode->Header.Previous = New; } - static void SetNext(uintptr_t Base, NodeWrapper Node, NodeWrapper New) { + static void SetNext(uintptr_t Base, value_type Node, value_type New) { if (Node.NodeOffset == 0) return; - OrderedNode *RealNode = reinterpret_cast(Node.GetPtr(Base)); + OrderedNode *RealNode = Node.GetNode(Base); RealNode->Header.Next = New; } @@ -304,26 +311,24 @@ static_assert(std::is_trivially_copyable::value); static_assert(offsetof(OrderedNode, Header) == 0); static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t))); +struct RegisterClassType final { + uint32_t Val; + operator uint32_t() { + return Val; + } +}; + #define IROP_ENUM #define IROP_STRUCTS #define IROP_SIZES #include "IRDefines.inc" -template -struct Wrapper final { - T *first; - OrderedNode *Node; ///< Actual offset of this IR in ths list - - operator Wrapper() const { return Wrapper {reinterpret_cast(first), Node}; } - operator OrderedNode *() { return Node; } - operator NodeWrapper () { return Node->Header.Value; } -}; - template class IRListView; void Dump(std::stringstream *out, IRListView const* IR); -inline uint32_t NodeWrapper::ID() const { return NodeOffset / sizeof(IR::OrderedNode); } +template +inline uint32_t NodeWrapperBase::ID() const { return NodeOffset / sizeof(IR::OrderedNode); } }; diff --git a/include/FEXCore/IR/IntrusiveIRList.h b/include/FEXCore/IR/IntrusiveIRList.h index f3fcd78e9..e1adc3a8e 100644 --- a/include/FEXCore/IR/IntrusiveIRList.h +++ b/include/FEXCore/IR/IntrusiveIRList.h @@ -22,7 +22,7 @@ class IntrusiveAllocator final { IntrusiveAllocator(IntrusiveAllocator &&) = delete; IntrusiveAllocator(size_t Size) : MemorySize {Size} { - Data = reinterpret_cast(calloc(Size, 1)); + Data = reinterpret_cast(malloc(Size)); } ~IntrusiveAllocator() { @@ -35,7 +35,7 @@ class IntrusiveAllocator final { } void *Allocate(size_t Size) { - assert(CheckSize(Size) && "Failure"); + LogMan::Throw::A(CheckSize(Size), "Ran out of space in IntrusiveAllocator during allocation"); size_t NewOffset = CurrentOffset + Size; uintptr_t NewPointer = Data + CurrentOffset; CurrentOffset = NewOffset; @@ -95,12 +95,13 @@ public: size_t GetDataSize() const { return DataSize; } size_t GetListSize() const { return ListSize; } + size_t GetSSACount() const { return ListSize / sizeof(OrderedNode); } using iterator = NodeWrapperIterator; iterator begin() const noexcept { - NodeWrapper Wrapped; + OrderedNodeWrapper Wrapped; Wrapped.NodeOffset = sizeof(OrderedNode); return iterator(reinterpret_cast(ListData), Wrapped); } @@ -112,11 +113,19 @@ public: */ iterator end() const noexcept { - NodeWrapper Wrapped; + OrderedNodeWrapper Wrapped; Wrapped.NodeOffset = 0; return iterator(reinterpret_cast(ListData), Wrapped); } + /** + * @brief Convert a OrderedNodeWrapper to an interator that we can iterate over + * @return Iterator for this op + */ + iterator at(OrderedNodeWrapper Node) const noexcept { + return iterator(reinterpret_cast(ListData), Node); + } + private: void *IRData; void *ListData;