Core: Reduce JIT time by sharing CodeBuffers between threads

This is changes the interface of CodeBuffer to that of a partially persistent
data structure based on reference counting:
- Exactly one CodeBuffer is now designated as "active", which means data can
  be *appended* to it
- Lossy modifications to the active CodeBuffer will not invalidate any data
  in use by other threads, which enables save sharing across threads
- Instead, such lossy modifications trigger a new "version" of the data in
  the modifying thread. Old versions of the CodeBuffer persist as read-only
  data for use by the other threads.
- The other threads can update their version of the CodeBuffer. This will
  decrease the reference count and eventually trigger deallocation of the
  old version
This commit is contained in:
Tony Wasserka committed 2025-06-01 22:44:49 +02:00
1 parent ab51958b26
commit 4078840ef1
9 files changed
+206 -85

No files matched your search

+7 -1
View File
@@ -164,7 +164,8 @@ public:
IRCaptureCache.WriteFilesWithCode(Writer);
}
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) override;
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
return CodeInvalidationMutex;
@@ -263,6 +264,10 @@ public:
auto Thread = Frame->Thread;
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
// NOTE: Other threads sharing the same CodeBuffer may reference
// invalidated data ranges through their L1/L2 caches. This is
// not currently a problem since FEX does not repurpose the
// invalidated CodeBuffer memory range currently.
ThreadRemoveCodeEntry(Thread, GuestRIP);
}
@@ -283,6 +288,7 @@ public:
fextl::unique_ptr<FEXCore::Core::DebugData> DebugData;
uint64_t StartAddr;
uint64_t Length;
std::unique_lock<ForkableUniqueMutex> CodeBufferLock;
};
[[nodiscard]]
CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
+71 -41
View File
@@ -6,6 +6,8 @@
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include <cstdint>
#include "LookupCache.h"
#ifndef _WIN32
#include <sys/prctl.h>
#endif
@@ -264,10 +266,10 @@ namespace CPU {
return TotalLUT;
}()};
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
CPUBackend::CPUBackend(CodeBufferManager& manager, FEXCore::Core::InternalThreadState* ThreadState, size_t MaxCodeSize)
: ThreadState(ThreadState)
, InitialCodeSize(InitialCodeSize)
, MaxCodeSize(MaxCodeSize) {
, MaxCodeSize(MaxCodeSize)
, manager(manager) {
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
@@ -307,37 +309,38 @@ namespace CPU {
CPUBackend::~CPUBackend() = default;
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer* {
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
if (CodeBuffers.empty()) {
EmplaceNewCodeBuffer(manager.AllocateNew(InitialCodeSize));
} else {
// If we have more than one code buffer we are tracking then walk them and delete
// This is a cleanup step
CodeBuffers.resize(1);
auto PrevCodeBuffer = CurrentCodeBuffer;
if (CurrentCodeBufferSize != MaxCodeSize) {
CodeBuffers.clear();
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer = manager.StartLargerCodeBuffer(MaxCodeSize);
// Resize the code buffer and reallocate our code size
CurrentCodeBufferSize *= 1.5;
CurrentCodeBufferSize = std::min(CurrentCodeBufferSize, MaxCodeSize);
EmplaceNewCodeBuffer(manager.AllocateNew(CurrentCodeBufferSize));
}
}
} else {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Allocate some new code buffers that we can switch over to instead
EmplaceNewCodeBuffer(manager.AllocateNew(InitialCodeSize));
}
return CodeBuffers.back().get();
RegisterForSignalHandler(PrevCodeBuffer);
return CurrentCodeBuffer.get();
}
void CPUBackend::EmplaceNewCodeBuffer(fextl::shared_ptr<CodeBuffer> Buffer) {
CurrentCodeBufferSize = Buffer->Size;
CodeBuffers.emplace_back(Buffer);
void CPUBackend::RegisterForSignalHandler(fextl::shared_ptr<CodeBuffer> CodeBuffer) {
if (ThreadState->CurrentFrame->SignalHandlerRefCounter != 0) {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Keep a reference to the old code buffer to delay deallocation
SignalHandlerCodeBuffers.push_back(CodeBuffer);
} else {
SignalHandlerCodeBuffers.clear();
}
}
fextl::shared_ptr<CodeBuffer> CPUBackend::CheckCodeBufferUpdate() {
fextl::shared_ptr<CodeBuffer> OldCodeBuffer;
auto NewCodeBuffer = manager.GetLatest();
if (CurrentCodeBuffer != NewCodeBuffer) {
RegisterForSignalHandler(CurrentCodeBuffer);
return std::exchange(CurrentCodeBuffer, NewCodeBuffer);
}
return nullptr;
}
GuestToHostMap& GetLookupCache(const CodeBuffer& Buffer) {
return *Buffer.LookupCache;
}
CodeBuffer::CodeBuffer(size_t Size)
@@ -351,10 +354,11 @@ namespace CPU {
FEXCore::Allocator::ProtectOptions::None)) {
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
}
LookupCache = fextl::make_unique<GuestToHostMap>();
}
CodeBuffer::~CodeBuffer() {
// TODO: Assert that mutex is held?
FEXCore::Allocator::VirtualFree(Ptr, Size);
}
@@ -386,24 +390,50 @@ namespace CPU {
auto Buffer = fextl::make_shared<CodeBuffer>(Size);
// TODO: Re-enable
// if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
// static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
// }
Latest = Buffer;
LatestOffset = 0;
OnCodeBufferAllocated(*Buffer);
return Buffer;
}
fextl::shared_ptr<CodeBuffer> CodeBufferManager::GetLatest() {
if (!Latest) {
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
AllocateNew(INITIAL_CODE_SIZE);
}
return Latest;
}
fextl::shared_ptr<CodeBuffer> CodeBufferManager::StartLargerCodeBuffer(size_t MaxCodeSize) {
if (!Latest) {
// Allocate initial CodeBuffer and return it
return GetLatest();
}
auto NewCodeBufferSize = GetLatest()->Size;
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MaxCodeSize);
return AllocateNew(NewCodeBufferSize);
}
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
// The last page of the code buffer is protected, so we need to exclude it from the valid range
// when checking if the address is in the code buffer.
for (auto& Buffer : CodeBuffers) {
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer->Ptr) + Buffer->Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
if (Address >= reinterpret_cast<uintptr_t>(Buffer->Ptr) && Address < LastPageAddr) {
auto CheckCodeBuffer = [](CodeBuffer& Buffer, uintptr_t Address) {
// The last page of the code buffer is protected, so we need to exclude it from the valid range
// when checking if the address is in the code buffer.
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer.Ptr) + Buffer.Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
return (Address >= reinterpret_cast<uintptr_t>(Buffer.Ptr) && Address < LastPageAddr);
};
if (CheckCodeBuffer(*CurrentCodeBuffer, Address)) {
return true;
}
for (auto& Buffer : SignalHandlerCodeBuffers) {
if (CheckCodeBuffer(*Buffer, Address)) {
return true;
}
}
return false;
}
+48 -11
View File
@@ -9,6 +9,7 @@ $end_info$
#pragma once
#include <FEXCore/Utils/CompilerDefs.h>
#include <FEXCore/Utils/SignalScopeGuards.h>
#include <FEXCore/fextl/memory.h>
#include <FEXCore/fextl/string.h>
#include <FEXCore/fextl/vector.h>
@@ -32,11 +33,15 @@ namespace CodeSerialize {
struct CodeObjectFileSection;
}
struct GuestToHostMap;
namespace CPU {
struct CodeBuffer {
uint8_t* Ptr;
size_t Size;
fextl::unique_ptr<GuestToHostMap> LookupCache;
CodeBuffer(size_t Size);
CodeBuffer(const CodeBuffer&) = delete;
CodeBuffer& operator=(const CodeBuffer&) = delete;
@@ -46,8 +51,36 @@ namespace CPU {
~CodeBuffer();
};
/**
* A manager that coordinates access to the CodeBuffer used for compiling new code across threads.
*
* The CodeBuffer is managed as a partially persistent data structure:
* - Exactly one CodeBuffer is now designated as "active", which means data can be appended to it
* - Lossy modifications to the active CodeBuffer will not invalidate any data in use by other threads (which is what enables save CodeBuffer sharing across threads)
* - Instead, such lossy modifications trigger a new "version" of the data in the modifying thread. Old versions of the CodeBuffer persist as read-only data for use by the other threads.
* - The other threads can update their version of the CodeBuffer. This will decrease the reference count and eventually trigger deallocation of the old version
*/
class CodeBufferManager {
public:
// Get the CodeBuffer that was most recently allocated.
// This is the only CodeBuffer that data may be written to.
fextl::shared_ptr<CodeBuffer> GetLatest();
// Allocate a new CodeBuffer with geometric growth.
// Subsequent calls to GetLatest will point to the returned buffer.
fextl::shared_ptr<CodeBuffer> StartLargerCodeBuffer(size_t MaxCodeSize);
// Write offset into the latest CodeBuffer
std::size_t LatestOffset;
// Protects writes to the latest CodeBuffer
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
virtual void OnCodeBufferAllocated(CodeBuffer&) {};
private:
fextl::shared_ptr<CodeBuffer> Latest;
fextl::shared_ptr<CodeBuffer> AllocateNew(size_t Size);
};
@@ -55,10 +88,9 @@ namespace CPU {
public:
/**
* @param InitialCodeSize - Initial size for the code buffers
* @param MaxCodeSize - Max size for the code buffers
*/
CPUBackend(FEXCore::Core::InternalThreadState*, size_t InitialCodeSize, size_t MaxCodeSize);
CPUBackend(CodeBufferManager&, FEXCore::Core::InternalThreadState*, size_t MaxCodeSize);
virtual ~CPUBackend();
@@ -157,6 +189,11 @@ namespace CPU {
bool IsAddressInCodeBuffer(uintptr_t Address) const;
// Updates the CodeBuffer if needed and returns a reference to the old one.
// The returned reference should be kept alive carefully to avoid early deletion of resources.
[[nodiscard]]
fextl::shared_ptr<CodeBuffer> CheckCodeBufferUpdate();
protected:
// Max spill slot size in bytes. We need at most 32 bytes
// to be able to handle a 256-bit vector store to a slot.
@@ -164,21 +201,21 @@ namespace CPU {
FEXCore::Core::InternalThreadState* ThreadState;
size_t InitialCodeSize, MaxCodeSize;
size_t MaxCodeSize;
[[nodiscard]]
CodeBuffer* GetEmptyCodeBuffer();
// This is the size of the last code buffer we allocated
size_t CurrentCodeBufferSize = 0;
// This is the code buffer containing the main code under execution by this thread.
// CheckCodeBufferUpdate must be used before compiling new code.
fextl::shared_ptr<CodeBuffer> CurrentCodeBuffer;
CodeBufferManager manager; // TODO: Rename
// Old CodeBuffer generations required to be valid until returning from signal handlers
fextl::vector<fextl::shared_ptr<CodeBuffer>> SignalHandlerCodeBuffers;
CodeBufferManager& manager; // TODO: Rename
private:
void EmplaceNewCodeBuffer(fextl::shared_ptr<CodeBuffer> Buffer);
// This is the array of code buffers. Unless signals force us to keep more than
// buffer, there will be only one entry here
fextl::vector<fextl::shared_ptr<CodeBuffer>> CodeBuffers;
void RegisterForSignalHandler(fextl::shared_ptr<CodeBuffer>);
};
} // namespace CPU
+21 -8
View File
@@ -495,7 +495,13 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
}
#endif
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
void ContextImpl::OnCodeBufferAllocated(CPU::CodeBuffer& Buffer) {
if (Config.GlobalJITNaming()) {
Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
}
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer) {
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
if (CodeObjectCacheService) {
@@ -503,10 +509,14 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
// Use the thread's object cache ref counter for this
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
}
auto lk = Thread->LookupCache->AcquireLock();
Thread->LookupCache->ClearCache();
Thread->CPUBackend->ClearCache();
if (NewCodeBuffer) {
// Allocate new CodeBuffer + L3 LookupCache and clear L1+L2 caches
Thread->CPUBackend->ClearCache();
} else {
// Clear L1+L2 cache of this thread, and clear L3 cache across any threads using it
Thread->LookupCache->ClearCache();
}
}
static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP) {
@@ -744,6 +754,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
.StartAddr = 0, // Unused
.Length = 0, // Unused
.CodeBufferLock {} // Unused
};
}
}
@@ -768,6 +779,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
// Attempt to get the CPU backend to compile this code
auto Lock = std::unique_lock {CodeBufferWriteMutex};
auto CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &*IRView, DebugData.get(), TFSet);
// Release the IR
@@ -781,6 +793,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
.DebugData = std::move(DebugData),
.StartAddr = StartAddr,
.Length = Length,
.CodeBufferLock = std::move(Lock),
};
}
@@ -800,7 +813,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
return HostCode;
}
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
auto [CodePtr, DebugData, StartAddr, Length, CodeBufferLock] = CompileCode(Thread, GuestRIP, MaxInst);
if (CodePtr == nullptr) {
return 0;
}
@@ -872,7 +885,7 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1);
auto [CodePtr, DebugData, StartAddr, Length, CodeBufferLock] = CompileCode(Thread, GuestRIP, 1);
if (CodePtr == nullptr) {
return 0;
}
@@ -911,8 +924,8 @@ void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
UpdateAtomicTSOEmulationConfig();
if (Config.TSOAutoMigration) {
// Only the lookup cache is cleared here, so that old code can keep running until next compilation
auto lk = Thread->LookupCache->AcquireLock();
// Only the lookup cache is cleared here, so that old code can keep running until next compilation.
// This will leak previously compiled blocks until the CodeBuffer is cleared for some other reason.
Thread->LookupCache->ClearCache();
}
}
+20 -6
View File
@@ -40,7 +40,6 @@ $end_info$
#include <string.h>
#include <limits>
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
@@ -496,7 +495,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Fram
void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::Ref Node) {}
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
: CPUBackend(*ctx, Thread, MAX_CODE_SIZE)
, Arm64Emitter(ctx)
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128}
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256}
@@ -550,8 +549,8 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
AArch64.LREM = reinterpret_cast<uint64_t>(LREM);
}
// Must be done after Dispatcher init
ClearCache();
CurrentCodeBuffer = manager.GetLatest();
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
// Setup dynamic dispatch.
if (ParanoidTSO()) {
@@ -570,11 +569,15 @@ void Arm64JITCore::EmitDetectionString() {
}
void Arm64JITCore::ClearCache() {
// Get the backing code buffer
// NOTE: Holding on to the reference here is required to ensure validity of the WriteLock mutex
auto PrevCodeBuffer = CurrentCodeBuffer;
std::lock_guard lk(PrevCodeBuffer->LookupCache->WriteLock);
auto CodeBuffer = GetEmptyCodeBuffer();
SetBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
EmitDetectionString();
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache);
}
Arm64JITCore::~Arm64JITCore() {}
@@ -693,7 +696,16 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
// Fairly excessive buffer range to make sure we don't overflow
uint32_t BufferRange = SSACount * 16;
if ((GetCursorOffset() + BufferRange) > (CurrentCodeBufferSize - Utils::FEX_PAGE_SIZE)) {
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
"doesn't match up!\n");
if (auto Prev = CheckCodeBufferUpdate()) {
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache);
}
SetBuffer(CurrentCodeBuffer->Ptr, CurrentCodeBuffer->Size);
SetCursorOffset(manager.LatestOffset);
if ((GetCursorOffset() + BufferRange) > (CurrentCodeBuffer->Size - Utils::FEX_PAGE_SIZE)) {
CTX->ClearCodeCache(ThreadState);
}
@@ -867,6 +879,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
JITBlockTail->Size = CodeData.Size;
manager.LatestOffset = GetCursorOffset();
ClearICache(CodeData.BlockBegin, CodeOnlySize);
#ifdef VIXL_DISASSEMBLER
+10 -3
View File
@@ -65,20 +65,27 @@ LookupCache::~LookupCache() {
}
void LookupCache::ClearL2Cache() {
auto lk = L3.AcquireLock();
auto lk = Shared->AcquireLock();
// Clear out the page memory
// PagePointer and PageMemory are sequential with each other. Clear both at once.
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, false);
AllocateOffset = 0;
}
void LookupCache::ClearThreadLocalCaches() {
auto lk = Shared->AcquireLock();
// Clear L1 and L2 by clearing the full cache.
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
}
void LookupCache::ClearCache() {
auto lk = L3.AcquireLock();
auto lk = Shared->AcquireLock();
// Clear L1 and L2 by clearing the full cache.
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
L3.ClearCache(lk);
Shared->ClearCache(lk);
}
void GuestToHostMap::ClearCache(const LockToken&) {
+26 -12
View File
@@ -61,6 +61,10 @@ struct GuestToHostMap {
// Adds to Guest -> Host code mapping
void AddBlockMapping(uint64_t Address, void* HostCode, const LockToken&) {
// This may replace an existing mapping
// NOTE: Generally no previous entry should exist, however there is one exception:
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
// may already contain the block address. Since is comparatively rare, we'll just leak
// one of the two blocks in this case.
BlockList[Address] = (uintptr_t)HostCode;
}
@@ -102,6 +106,13 @@ public:
LookupCache(FEXCore::Context::ContextImpl* CTX);
~LookupCache();
// Swaps out the underlying GuestToHostMap and clears all associated caches.
// This interface requires the previous CodeBuffer to be provided despite not using it. This ensures the shared write lock is still valid.
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap) {
ClearThreadLocalCaches();
Shared = &NewMap;
}
uintptr_t FindBlock(uint64_t Address) {
// Try L1, no lock needed
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
@@ -110,7 +121,7 @@ public:
}
// L2 and L3 need to be locked
auto lk = L3.AcquireLock();
auto lk = Shared->AcquireLock();
// Try L2
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
@@ -132,7 +143,7 @@ public:
}
// Try L3
auto HostCode = L3.FindBlock(Address, lk);
auto HostCode = Shared->FindBlock(Address, lk);
if (HostCode) {
CacheBlockMapping(Address, HostCode.value());
return HostCode.value();
@@ -142,15 +153,14 @@ public:
return 0;
}
GuestToHostMap L3;
GuestToHostMap* Shared = nullptr;
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
// Appends Block {Address} to CodePages [Start, Start + Length)
// Returns true if new pages are marked as containing code
bool AddBlockExecutableRange(uint64_t Address, uint64_t Start, uint64_t Length) {
auto lk = L3.AcquireLock();
auto lk = Shared->AcquireLock();
bool rv = false;
@@ -165,9 +175,9 @@ public:
// Adds to Guest -> Host code mapping
void AddBlockMapping(uint64_t Address, void* HostCode) {
auto lk = L3.AcquireLock();
auto lk = Shared->AcquireLock();
L3.AddBlockMapping(Address, HostCode, lk);
Shared->AddBlockMapping(Address, HostCode, lk);
// There is no need to update L1 or L2, they will get updated on first lookup
// However, adding to L1 here increases performance
@@ -176,10 +186,13 @@ public:
L1Entry.HostCode = (uintptr_t)HostCode;
}
// NOTE: It's the caller's responsibility to call Erase() for all other
// GuestToHostMaps that share the same LookupCache. Otherwise, the
// L1/L2 caches will contain stale references to deallocated memory.
void Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
auto lk = L3.AcquireLock();
auto lk = Shared->AcquireLock();
L3.Erase(Frame, Address, lk);
Shared->Erase(Frame, Address, lk);
// Do L1
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
@@ -209,12 +222,13 @@ public:
}
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
auto lk = L3.AcquireLock();
L3.AddBlockLink(GuestDestination, HostLink, delinker, lk);
auto lk = Shared->AcquireLock();
Shared->AddBlockLink(GuestDestination, HostLink, delinker, lk);
}
void ClearCache();
void ClearL2Cache();
void ClearThreadLocalCaches();
uintptr_t GetL1Pointer() const {
return L1Pointer;
@@ -237,7 +251,7 @@ public:
// This approach has not been fully vetted yet.
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
auto AcquireLock() {
return L3.AcquireLock();
return Shared->AcquireLock();
}
private:
+2 -2
View File
@@ -171,7 +171,7 @@ public:
FEX_DEFAULT_VISIBILITY virtual void FinalizeAOTIRCache() = 0;
FEX_DEFAULT_VISIBILITY virtual void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) = 0;
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) = 0;
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) = 0;
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) = 0;
FEX_DEFAULT_VISIBILITY virtual FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() = 0;
@@ -181,7 +181,7 @@ public:
ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) = 0;
/**
* @brief Checks if a PC is inside of a thread's JIT code buffer.
* @brief Checks if a PC is inside any code buffer used by the thread's JIT.
*
* @param Thread Which thread's code buffers to check inside of.
* @param Address The PC to check against.
@@ -297,7 +297,7 @@ void ThreadManager::Step() {
// Walk the threads and tell them to clear their caches
// Useful when our block size is set to a large number and we need to step a single instruction
for (auto& Thread : Threads) {
CTX->ClearCodeCache(Thread->Thread);
CTX->ClearCodeCache(Thread->Thread, false);
}
}