FEXCore: Support multiple entrypoints into a multiblock

If a multiblock contains a call instruction, we know at the point
of compilation that the instruction after that call will likely be
jumped to at some point. Avoid redundant recompilation by tracking
such cases and including an entrypoint for that instruction in the
multiblock aswell.
This commit is contained in:
Billy Laws committed 2025-07-10 16:00:24 +01:00
1 parent cdef1ed0c5
commit d41cb3b69d
8 files changed
+81 -63

No files matched your search

+1 -1
View File
@@ -284,7 +284,7 @@ public:
GenerateIRResult GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, bool ExtendedDebugInfo, uint64_t MaxInst);
struct CompileCodeResult {
void* CompiledCode;
CPU::CPUBackend::CompiledCode CompiledCode;
fextl::unique_ptr<FEXCore::Core::DebugData> DebugData;
uint64_t StartAddr;
uint64_t Length;
+2 -9
View File
@@ -13,6 +13,7 @@ $end_info$
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
#include <FEXCore/fextl/map.h>
#include <cstdint>
@@ -94,15 +95,7 @@ namespace CPU {
struct CompiledCode {
// Where this code block begins.
uint8_t* BlockBegin;
/**
* The function entrypoint to this codeblock.
*
* This may or may not equal `BlockBegin` above. Depending on the CPU backend, it may stick data
* prior to the BlockEntry.
*
* Is actually a function pointer of type `void (FEXCore::Core::ThreadState *Thread)`
*/
uint8_t* BlockEntry;
fextl::map<uint64_t, uint8_t*> EntryPoints;
// The total size of the codeblock from [BlockBegin, BlockBegin+Size).
size_t Size;
};
+22 -18
View File
@@ -750,7 +750,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
auto CompiledCode = Thread->CPUBackend->RelocateJITObjectCode(GuestRIP, CodeCacheEntry);
if (CompiledCode) {
return {
.CompiledCode = CompiledCode,
.CompiledCode = {},
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
.StartAddr = 0, // Unused
.Length = 0, // Unused
@@ -769,7 +769,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
// Generate IR + Meta Info
auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst);
if (!IRView) {
return {nullptr, nullptr, 0, 0};
return {{}, nullptr, 0, 0};
}
// Attempt to get the CPU backend to compile this code
@@ -780,7 +780,10 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
if (MaxInst != 1) {
if (auto Block = Thread->LookupCache->FindBlock(GuestRIP)) {
Thread->OpDispatcher->DelayedDisownBuffer();
return {.CompiledCode = reinterpret_cast<uint8_t*>(Block), .DebugData = nullptr, .StartAddr = 0, .Length = 0};
return {.CompiledCode = {.BlockBegin = reinterpret_cast<uint8_t*>(Block), .EntryPoints = {{GuestRIP, reinterpret_cast<uint8_t*>(Block)}}},
.DebugData = nullptr,
.StartAddr = 0,
.Length = 0};
}
}
@@ -795,10 +798,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
Thread->OpDispatcher->DelayedDisownBuffer();
return {
// FEX currently throws away the CPUBackend::CompiledCode object other than the entrypoint
// In the future with code caching getting wired up, we will pass the rest of the data forward.
// TODO: Pass the data forward when code caching is wired up to this.
.CompiledCode = CompiledCode.BlockEntry,
.CompiledCode = std::move(CompiledCode),
.DebugData = std::move(DebugData),
.StartAddr = StartAddr,
.Length = Length,
@@ -821,7 +821,8 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
return HostCode;
}
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
auto [CompiledCode, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
if (CodePtr == nullptr) {
return 0;
} else if (!DebugData) {
@@ -831,7 +832,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
// The core managed to compile the code.
if (Config.BlockJITNaming()) {
auto FragmentBasePtr = reinterpret_cast<uint8_t*>(CodePtr);
auto FragmentBasePtr = CompiledCode.BlockBegin;
if (DebugData) {
auto GuestRIPLookup = SyscallHandler->LookupAOTIRCacheEntry(Thread, GuestRIP);
@@ -840,7 +841,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
for (auto& Subblock : DebugData->Subblocks) {
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
if (GuestRIPLookup.Entry) {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename,
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
GuestRIP - GuestRIPLookup.VAFileStart);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
@@ -848,10 +849,10 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
}
} else {
if (GuestRIPLookup.Entry) {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename,
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, CompiledCode.Size, GuestRIPLookup.Entry->Filename,
GuestRIP - GuestRIPLookup.VAFileStart);
} else {
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, DebugData->HostCodeSize);
Symbols.Register(Thread->SymbolBuffer.get(), FragmentBasePtr, GuestRIP, CompiledCode.Size);
}
}
}
@@ -864,8 +865,8 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
.GuestRIP = GuestRIP,
.GuestCodeLength = Length,
.GuestCodeHash = 0,
.HostCodeBegin = CodePtr,
.HostCodeLength = DebugData->HostCodeSize,
.HostCodeBegin = CompiledCode.BlockBegin,
.HostCodeLength = CompiledCode.Size,
.HostCodeHash = 0,
.ThreadJobRefCount = &Thread->ObjectCacheRefCounter,
.Relocations = std::move(*DebugData->Relocations),
@@ -875,14 +876,16 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
// Clear any relocations that might have been generated
Thread->CPUBackend->ClearRelocations();
if (IRCaptureCache.PostCompileCode(Thread, CodePtr, GuestRIP, StartAddr, Length, {}, DebugData.get(), false)) {
if (IRCaptureCache.PostCompileCode(Thread, CompiledCode.BlockBegin, GuestRIP, StartAddr, Length, {}, DebugData.get(), false)) {
// Early exit
return (uintptr_t)CodePtr;
}
// Insert to lookup cache
// Pages containing this block are added via AddBlockExecutableRange before each page gets accessed in the frontend
Thread->LookupCache->AddBlockMapping(GuestRIP, CodePtr);
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
Thread->LookupCache->AddBlockMapping(GuestAddr, HostAddr);
}
return (uintptr_t)CodePtr;
}
@@ -896,7 +899,8 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1);
auto [CompiledCode, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1);
auto CodePtr = CompiledCode.EntryPoints[GuestRIP];
if (CodePtr == nullptr) {
return 0;
}
@@ -980,7 +984,7 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu
Entrypoint,
[this, GuestThunkEntrypoint](uintptr_t Entrypoint, FEXCore::IR::IREmitter* emit) {
auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0, 0, 0);
auto Block = emit->CreateCodeNode();
auto Block = emit->CreateCodeNode(true, 0);
IRHeader.first->Blocks = emit->WrapNode(Block);
emit->SetCurrentCodeBlock(Block);
+50 -31
View File
@@ -741,6 +741,26 @@ void Arm64JITCore::EmitInterruptChecks(bool CheckTF) {
#endif
}
void Arm64JITCore::EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF) {
// Get the address of the JITCodeHeader and store in to the core state.
// Two instruction cost, each 1 cycle.
adr(TMP1, &HeaderLabel);
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
EmitInterruptChecks(CheckTF);
if (SpillSlots) {
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
if (ARMEmitter::IsImmAddSub(TotalSpillSlotsSize)) {
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
}
}
}
CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR,
FEXCore::Core::DebugData* DebugData, bool CheckTF) {
FEXCORE_PROFILE_SCOPED("Arm64::CompileCode");
@@ -751,6 +771,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
this->Entry = Entry;
this->DebugData = DebugData;
this->IR = IR;
CodeData.EntryPoints.clear();
// Fairly excessive buffer range to make sure we don't overflow
uint32_t BufferRange = 0x100 + SSACount * 24;
@@ -768,9 +789,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
JITCodeHeader* CodeHeader = GetCursorAddress<JITCodeHeader*>();
CursorIncrement(sizeof(JITCodeHeader));
#ifdef VIXL_DISASSEMBLER
const auto DisasmBegin = GetCursorAddress<const vixl::aarch64::Instruction*>();
#endif
auto CodeBegin = GetCursorAddress<uint8_t*>();
// AAPCS64
// r30 = LR
@@ -792,34 +811,15 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// X1-X3 = Temp
// X4-r18 = RA
CodeData.BlockEntry = GetCursorAddress<uint8_t*>();
// Get the address of the JITCodeHeader and store in to the core state.
// Two instruction cost, each 1 cycle.
adr(TMP1, &JITCodeHeaderLabel);
str(TMP1, STATE, offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader));
EmitInterruptChecks(CheckTF);
SpillSlots = IR->SpillSlots();
if (SpillSlots) {
const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize;
if (ARMEmitter::IsImmAddSub(TotalSpillSlotsSize)) {
sub(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::rsp, ARMEmitter::Reg::rsp, TotalSpillSlotsSize);
} else {
LoadConstant(ARMEmitter::Size::i64Bit, TMP1, TotalSpillSlotsSize);
sub(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::rsp, ARMEmitter::XReg::rsp, TMP1, ARMEmitter::ExtendedType::LSL_64, 0);
}
}
bool EmittedEntry = false;
PendingTargetLabel = nullptr;
for (auto [BlockNode, BlockHeader] : IR->GetBlocks()) {
using namespace FEXCore::IR;
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
#endif
@@ -832,6 +832,21 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
if (PendingTargetLabel && PendingTargetLabel != &IsTarget->second) {
b(PendingTargetLabel);
}
if (BlockIROp->EntryPoint) {
uint64_t BlockStartRIP = Entry + BlockIROp->GuestEntryOffset;
if (EmittedEntry) {
b(&IsTarget->second);
} else {
EmittedEntry = true;
}
CodeData.EntryPoints.emplace(BlockStartRIP, GetCursorAddress<uint8_t*>());
DebugData->GuestOpcodes.push_back({BlockIROp->GuestEntryOffset, GetCursorAddress<uint8_t*>() - CodeData.BlockBegin});
EmitEntryPoint(JITCodeHeaderLabel, CheckTF);
}
PendingTargetLabel = nullptr;
Bind(&IsTarget->second);
@@ -852,7 +867,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
}
}
DebugData->Subblocks.push_back({static_cast<uint32_t>(BlockStartHostCode - CodeData.BlockEntry),
DebugData->Subblocks.push_back({static_cast<uint32_t>(BlockStartHostCode - CodeData.BlockBegin),
static_cast<uint32_t>(GetCursorAddress<uint8_t*>() - BlockStartHostCode)});
}
@@ -862,8 +877,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
}
PendingTargetLabel = nullptr;
// CodeSize not including the tail data.
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
// CodeSize not including the header or tail data.
const uint64_t CodeOnlySize = GetCursorAddress<uint8_t*>() - CodeBegin;
// Add the JitCodeTail
Align(alignof(JITCodeTail));
@@ -960,7 +975,10 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// Adjust host addresses
const auto Delta = GetCursorAddress<uint8_t*>() - CodeData.BlockBegin;
CodeData.BlockBegin += Delta;
CodeData.BlockEntry += Delta;
for (auto& EntryPoint : CodeData.EntryPoints) {
EntryPoint.second += Delta;
}
CodeBegin += Delta;
// Copy over CodeBuffer contents
memcpy(GetCursorAddress<uint8_t*>(), TempCodeBuffer, TempSize);
@@ -971,7 +989,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
TempAllocator.DelayedDisownBuffer();
ClearICache(CodeData.BlockBegin, CodeOnlySize);
ClearICache(CodeBegin, CodeOnlySize);
#ifdef VIXL_DISASSEMBLER
if (Disassemble() & FEXCore::Config::Disassemble::STATS) {
@@ -985,7 +1003,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
}
if (Disassemble() & FEXCore::Config::Disassemble::BLOCKS) {
const auto DisasmEnd = reinterpret_cast<const vixl::aarch64::Instruction*>(JITBlockTailLocation);
const auto DisasmBegin = reinterpret_cast<const vixl::aarch64::Instruction*>(CodeBegin);
const auto DisasmEnd = reinterpret_cast<const vixl::aarch64::Instruction*>(CodeBegin + CodeOnlySize);
LogMan::Msg::IFmt("Disassemble Begin");
for (auto PCToDecode = DisasmBegin; PCToDecode < DisasmEnd; PCToDecode += 4) {
DisasmDecoder->Decode(PCToDecode);
@@ -1001,7 +1020,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
this->IR = nullptr;
return CodeData;
return std::move(CodeData);
}
void Arm64JITCore::ResetStack() {
@@ -375,6 +375,8 @@ private:
void EmitInterruptChecks(bool CheckTF);
void EmitEntryPoint(ARMEmitter::BackwardLabel& HeaderLabel, bool CheckTF);
// Runtime selection;
// Load and store TSO memory style
OpType RT_LoadMemTSO;
@@ -3852,7 +3852,7 @@ void OpDispatchBuilder::CMPXCHGPairOp(OpcodeArgs) {
void OpDispatchBuilder::CreateJumpBlocks(const fextl::vector<FEXCore::Frontend::Decoder::DecodedBlocks>* Blocks) {
Ref PrevCodeBlock {};
for (auto& Target : *Blocks) {
auto CodeNode = CreateCodeNode();
auto CodeNode = CreateCodeNode(Target.IsEntryPoint, Target.Entry - Entry);
JumpTargets.try_emplace(Target.Entry, JumpTargetInfo {CodeNode, false});
+1 -1
View File
@@ -171,7 +171,7 @@
"SwitchGen": false,
"JITDispatchOverride": "NoOp"
},
"CodeBlock SSA:$Begin, SSA:$Last, u32:$ID": {
"CodeBlock SSA:$Begin, SSA:$Last, u32:$ID, i1:$EntryPoint{false}, u32:$GuestEntryOffset{0}": {
"SwitchGen": false,
"RAOverride": "0",
"JITDispatchOverride": "NoOp"
+2 -2
View File
@@ -248,11 +248,11 @@ public:
*
* @return OrderedNode
*/
IRPair<IROp_CodeBlock> CreateCodeNode() {
IRPair<IROp_CodeBlock> CreateCodeNode(bool EntryPoint = false, uint32_t GuestEntryOffset = 0) {
SetWriteCursor(nullptr); // Orphan from any previous nodes
auto ID = ViewIR().GetHeader()->BlockCount++;
auto CodeNode = _CodeBlock(InvalidNode, InvalidNode, ID);
auto CodeNode = _CodeBlock(InvalidNode, InvalidNode, ID, EntryPoint, GuestEntryOffset);
CodeBlocks.emplace_back(CodeNode);