From d41cb3b69d9ae829658ab5c3ca91ee7237f5a12f Mon Sep 17 00:00:00 2001 From: Billy Laws Date: Mon, 31 Mar 2025 23:48:36 +0100 Subject: [PATCH] FEXCore: Support multiple entrypoints into a multiblock If a multiblock contains a call instruction, we know at the point of compilation that the instruction after that call will likely be jumped to at some point. Avoid redundant recompilation by tracking such cases and including an entrypoint for that instruction in the multiblock aswell. --- FEXCore/Source/Interface/Context/Context.h | 2 +- FEXCore/Source/Interface/Core/CPUBackend.h | 11 +-- FEXCore/Source/Interface/Core/Core.cpp | 40 ++++----- FEXCore/Source/Interface/Core/JIT/JIT.cpp | 81 ++++++++++++------- FEXCore/Source/Interface/Core/JIT/JITClass.h | 2 + .../Interface/Core/OpcodeDispatcher.cpp | 2 +- FEXCore/Source/Interface/IR/IR.json | 2 +- FEXCore/Source/Interface/IR/IREmitter.h | 4 +- 8 files changed, 81 insertions(+), 63 deletions(-) diff --git a/FEXCore/Source/Interface/Context/Context.h b/FEXCore/Source/Interface/Context/Context.h index 879defe41..2728903c3 100644 --- a/FEXCore/Source/Interface/Context/Context.h +++ b/FEXCore/Source/Interface/Context/Context.h @@ -284,7 +284,7 @@ public: GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst); struct CompileCodeResult { - void* CompiledCode; + CPU::CPUBackend::CompiledCode CompiledCode; fextl::unique_ptr DebugData; uint64_t StartAddr; uint64_t Length; diff --git a/FEXCore/Source/Interface/Core/CPUBackend.h b/FEXCore/Source/Interface/Core/CPUBackend.h index b059b7e1f..ed5fc29d8 100644 --- a/FEXCore/Source/Interface/Core/CPUBackend.h +++ b/FEXCore/Source/Interface/Core/CPUBackend.h @@ -13,6 +13,7 @@ $end_info$ #include #include #include +#include #include @@ -94,15 +95,7 @@ namespace CPU { struct CompiledCode { // Where this code block begins. uint8_t* BlockBegin; - /** - * The function entrypoint to this codeblock. - * - * This may or may not equal `BlockBegin` above. Depending on the CPU backend, it may stick data - * prior to the BlockEntry. - * - * Is actually a function pointer of type `void (FEXCore::Core::ThreadState *Thread)` - */ - uint8_t* BlockEntry; + fextl::map EntryPoints; // The total size of the codeblock from [BlockBegin, BlockBegin+Size). size_t Size; }; diff --git a/FEXCore/Source/Interface/Core/Core.cpp b/FEXCore/Source/Interface/Core/Core.cpp index ccfec2e5d..9685ac38b 100644 --- a/FEXCore/Source/Interface/Core/Core.cpp +++ b/FEXCore/Source/Interface/Core/Core.cpp @@ -750,7 +750,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT auto CompiledCode = Thread->CPUBackend->RelocateJITObjectCode(GuestRIP, CodeCacheEntry); if (CompiledCode) { return { - .CompiledCode = CompiledCode, + .CompiledCode = {}, .DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read .StartAddr = 0, // Unused .Length = 0, // Unused @@ -769,7 +769,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT // Generate IR + Meta Info auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst); if (!IRView) { - return {nullptr, nullptr, 0, 0}; + return {{}, nullptr, 0, 0}; } // Attempt to get the CPU backend to compile this code @@ -780,7 +780,10 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT if (MaxInst != 1) { if (auto Block = Thread->LookupCache->FindBlock(GuestRIP)) { Thread->OpDispatcher->DelayedDisownBuffer(); - return {.CompiledCode = reinterpret_cast(Block), .DebugData = nullptr, .StartAddr = 0, .Length = 0}; + return {.CompiledCode = {.BlockBegin = reinterpret_cast(Block), .EntryPoints = {{GuestRIP, reinterpret_cast(Block)}}}, + .DebugData = nullptr, + .StartAddr = 0, + .Length = 0}; } } @@ -795,10 +798,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT Thread->OpDispatcher->DelayedDisownBuffer(); return { - // FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint - // In the future with code caching getting wired up, we will pass the rest of the data forward. - // TODO: Pass the data forward when code caching is wired up to this. - .CompiledCode = CompiledCode.BlockEntry, + .CompiledCode = std::move(CompiledCode), .DebugData = std::move(DebugData), .StartAddr = StartAddr, .Length = Length, @@ -821,7 +821,8 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_ return HostCode; } - auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst); + auto [CompiledCode, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst); + auto CodePtr = CompiledCode.EntryPoints[GuestRIP]; if (CodePtr == nullptr) { return 0; } else if (!DebugData) { @@ -831,7 +832,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_ // The core managed to compile the code. if (Config.BlockJITNaming()) { - auto FragmentBasePtr = reinterpret_cast(CodePtr); + auto FragmentBasePtr = CompiledCode.BlockBegin; if (DebugData) { auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP); @@ -840,7 +841,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_ for (auto& Subblock : DebugData->Subblocks) { auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset; if (GuestRIPLookup.Entry) { - Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename, + Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename, GuestRIP - GuestRIPLookup.VAFileStart); } else { Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize); @@ -848,10 +849,10 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_ } } else { if (GuestRIPLookup.Entry) { - Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename, + Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename, GuestRIP - GuestRIPLookup.VAFileStart); } else { - Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, DebugData->HostCodeSize); + Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size); } } } @@ -864,8 +865,8 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_ .GuestRIP = GuestRIP, .GuestCodeLength = Length, .GuestCodeHash = 0, - .HostCodeBegin = CodePtr, - .HostCodeLength = DebugData->HostCodeSize, + .HostCodeBegin = CompiledCode.BlockBegin, + .HostCodeLength = CompiledCode.Size, .HostCodeHash = 0, .ThreadJobRefCount = &Thread->ObjectCacheRefCounter, .Relocations = std::move(*DebugData->Relocations), @@ -875,14 +876,16 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_ // Clear any relocations that might have been generated Thread->CPUBackend->ClearRelocations(); - if (IRCaptureCache.PostCompileCode(Thread, CodePtr, GuestRIP, StartAddr, Length, {}, DebugData.get(), false)) { + if (IRCaptureCache.PostCompileCode(Thread, CompiledCode.BlockBegin, GuestRIP, StartAddr, Length, {}, DebugData.get(), false)) { // Early exit return (uintptr_t)CodePtr; } // Insert to lookup cache // Pages containing this block are added via AddBlockExecutableRange before each page gets accessed in the frontend - Thread->LookupCache->AddBlockMapping(GuestRIP, CodePtr); + for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) { + Thread->LookupCache->AddBlockMapping(GuestAddr, HostAddr); + } return (uintptr_t)CodePtr; } @@ -896,7 +899,8 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui // Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled auto lk = GuardSignalDeferringSection(CodeInvalidationMutex, Thread); - auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1); + auto [CompiledCode, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1); + auto CodePtr = CompiledCode.EntryPoints[GuestRIP]; if (CodePtr == nullptr) { return 0; } @@ -980,7 +984,7 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu Entrypoint, [this, GuestThunkEntrypoint](uintptr_t Entrypoint, FEXCore::IR::IREmitter* emit) { auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0, 0, 0); - auto Block = emit->CreateCodeNode(); + auto Block = emit->CreateCodeNode(true, 0); IRHeader.first->Blocks = emit->WrapNode(Block); emit->SetCurrentCodeBlock(Block); diff --git a/FEXCore/Source/Interface/Core/JIT/JIT.cpp b/FEXCore/Source/Interface/Core/JIT/JIT.cpp index 25ea1fb43..2b089cf2e 100644 --- a/FEXCore/Source/Interface/Core/JIT/JIT.cpp +++ b/FEXCore/Source/Interface/Core/JIT/JIT.cpp @@ -741,6 +741,26 @@ void Arm64JITCore::EmitInterruptChecks(bool CheckTF) { #endif } +void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF) { + // Get the address of the JITCodeHeader and store in to the core state. + // Two instruction cost, each 1 cycle. + adr(TMP1, &HeaderLabel); + str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader)); + + EmitInterruptChecks(CheckTF); + + if (SpillSlots) { + const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize; + + if (ARMEmitter::IsImmAddSub(TotalSpillSlotsSize)) { + sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize); + } else { + LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize); + sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0); + } + } +} + CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData, bool CheckTF) { FEXCORE_PROFILE_SCOPED("Arm64::CompileCode"); @@ -751,6 +771,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size this->Entry = Entry; this->DebugData = DebugData; this->IR = IR; + CodeData.EntryPoints.clear(); // Fairly excessive buffer range to make sure we don't overflow uint32_t BufferRange = 0x100 + SSACount * 24; @@ -768,9 +789,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size JITCodeHeader* CodeHeader = GetCursorAddress(); CursorIncrement(sizeof(JITCodeHeader)); -#ifdef VIXL_DISASSEMBLER - const auto DisasmBegin = GetCursorAddress(); -#endif + auto CodeBegin = GetCursorAddress(); // AAPCS64 // r30 = LR @@ -792,34 +811,15 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size // X1-X3 = Temp // X4-r18 = RA - CodeData.BlockEntry = GetCursorAddress(); - - // Get the address of the JITCodeHeader and store in to the core state. - // Two instruction cost, each 1 cycle. - adr(TMP1, &JITCodeHeaderLabel); - str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader)); - - EmitInterruptChecks(CheckTF); - SpillSlots = IR->SpillSlots(); - if (SpillSlots) { - const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize; - - if (ARMEmitter::IsImmAddSub(TotalSpillSlotsSize)) { - sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize); - } else { - LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize); - sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0); - } - } - + bool EmittedEntry = false; PendingTargetLabel = nullptr; for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) { using namespace FEXCore::IR; -#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED auto BlockIROp = BlockHeader->CW(); +#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block"); #endif @@ -832,6 +832,21 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second) { b(PendingTargetLabel); } + + if (BlockIROp->EntryPoint) { + uint64_t BlockStartRIP = Entry + BlockIROp->GuestEntryOffset; + if (EmittedEntry) { + b(&IsTarget->second); + } else { + EmittedEntry = true; + } + + CodeData.EntryPoints.emplace(BlockStartRIP, GetCursorAddress()); + DebugData->GuestOpcodes.push_back({BlockIROp->GuestEntryOffset, GetCursorAddress() - CodeData.BlockBegin}); + + EmitEntryPoint(JITCodeHeaderLabel, CheckTF); + } + PendingTargetLabel = nullptr; Bind(&IsTarget->second); @@ -852,7 +867,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size } } - DebugData->Subblocks.push_back({static_cast(BlockStartHostCode - CodeData.BlockEntry), + DebugData->Subblocks.push_back({static_cast(BlockStartHostCode - CodeData.BlockBegin), static_cast(GetCursorAddress() - BlockStartHostCode)}); } @@ -862,8 +877,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size } PendingTargetLabel = nullptr; - // CodeSize not including the tail data. - const uint64_t CodeOnlySize = GetCursorAddress() - CodeData.BlockBegin; + // CodeSize not including the header or tail data. + const uint64_t CodeOnlySize = GetCursorAddress() - CodeBegin; // Add the JitCodeTail Align(alignof(JITCodeTail)); @@ -960,7 +975,10 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size // Adjust host addresses const auto Delta = GetCursorAddress() - CodeData.BlockBegin; CodeData.BlockBegin += Delta; - CodeData.BlockEntry += Delta; + for (auto& EntryPoint : CodeData.EntryPoints) { + EntryPoint.second += Delta; + } + CodeBegin += Delta; // Copy over CodeBuffer contents memcpy(GetCursorAddress(), TempCodeBuffer, TempSize); @@ -971,7 +989,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size TempAllocator.DelayedDisownBuffer(); - ClearICache(CodeData.BlockBegin, CodeOnlySize); + ClearICache(CodeBegin, CodeOnlySize); #ifdef VIXL_DISASSEMBLER if (Disassemble() & FEXCore::Config::Disassemble::STATS) { @@ -985,7 +1003,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size } if (Disassemble() & FEXCore::Config::Disassemble::BLOCKS) { - const auto DisasmEnd = reinterpret_cast(JITBlockTailLocation); + const auto DisasmBegin = reinterpret_cast(CodeBegin); + const auto DisasmEnd = reinterpret_cast(CodeBegin + CodeOnlySize); LogMan::Msg::IFmt("Disassemble Begin"); for (auto PCToDecode = DisasmBegin; PCToDecode < DisasmEnd; PCToDecode += 4) { DisasmDecoder->Decode(PCToDecode); @@ -1001,7 +1020,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size this->IR = nullptr; - return CodeData; + return std::move(CodeData); } void Arm64JITCore::ResetStack() { diff --git a/FEXCore/Source/Interface/Core/JIT/JITClass.h b/FEXCore/Source/Interface/Core/JIT/JITClass.h index dfee02575..17ab7f92d 100644 --- a/FEXCore/Source/Interface/Core/JIT/JITClass.h +++ b/FEXCore/Source/Interface/Core/JIT/JITClass.h @@ -375,6 +375,8 @@ private: void EmitInterruptChecks(bool CheckTF); + void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF); + // Runtime selection; // Load and store TSO memory style OpType RT_LoadMemTSO; diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index efb5fc558..bdee4b8de 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -3852,7 +3852,7 @@ void OpDispatchBuilder::CMPXCHGPairOp(OpcodeArgs) { void OpDispatchBuilder::CreateJumpBlocks(const fextl::vector* Blocks) { Ref PrevCodeBlock {}; for (auto& Target : *Blocks) { - auto CodeNode = CreateCodeNode(); + auto CodeNode = CreateCodeNode(Target.IsEntryPoint, Target.Entry - Entry); JumpTargets.try_emplace(Target.Entry, JumpTargetInfo {CodeNode, false}); diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index 97da525c7..aed86b5c4 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -171,7 +171,7 @@ "SwitchGen": false, "JITDispatchOverride": "NoOp" }, - "CodeBlock SSA:$Begin, SSA:$Last, u32:$ID": { + "CodeBlock SSA:$Begin, SSA:$Last, u32:$ID, i1:$EntryPoint{false}, u32:$GuestEntryOffset{0}": { "SwitchGen": false, "RAOverride": "0", "JITDispatchOverride": "NoOp" diff --git a/FEXCore/Source/Interface/IR/IREmitter.h b/FEXCore/Source/Interface/IR/IREmitter.h index 04eaf2b0a..284aa3b5a 100644 --- a/FEXCore/Source/Interface/IR/IREmitter.h +++ b/FEXCore/Source/Interface/IR/IREmitter.h @@ -248,11 +248,11 @@ public: * * @return OrderedNode */ - IRPair CreateCodeNode() { + IRPair CreateCodeNode(bool EntryPoint = false, uint32_t GuestEntryOffset = 0) { SetWriteCursor(nullptr); // Orphan from any previous nodes auto ID = ViewIR().GetHeader()->BlockCount++; - auto CodeNode = _CodeBlock(InvalidNode, InvalidNode, ID); + auto CodeNode = _CodeBlock(InvalidNode, InvalidNode, ID, EntryPoint, GuestEntryOffset); CodeBlocks.emplace_back(CodeNode);