mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 17:00:19 +02:00
Core: Reduce JIT time by sharing CodeBuffers between threads
This is changes the interface of CodeBuffer to that of a partially persistent data structure based on reference counting: - Exactly one CodeBuffer is now designated as "active", which means data can be *appended* to it - Lossy modifications to the active CodeBuffer will not invalidate any data in use by other threads, which enables save sharing across threads - Instead, such lossy modifications trigger a new "version" of the data in the modifying thread. Old versions of the CodeBuffer persist as read-only data for use by the other threads. - The other threads can update their version of the CodeBuffer. This will decrease the reference count and eventually trigger deallocation of the old version
This commit is contained in:
1 parent
ab51958b26
commit
4078840ef1
9 files changed
+206
-85
No files matched your search
@@ -164,7 +164,8 @@ public:
|
||||
IRCaptureCache.WriteFilesWithCode(Writer);
|
||||
}
|
||||
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) override;
|
||||
void OnCodeBufferAllocated(CPU::CodeBuffer&) override;
|
||||
void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) override;
|
||||
void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) override;
|
||||
FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() override {
|
||||
return CodeInvalidationMutex;
|
||||
@@ -263,6 +264,10 @@ public:
|
||||
auto Thread = Frame->Thread;
|
||||
auto lk = GuardSignalDeferringSection(static_cast<ContextImpl*>(Thread->CTX)->CodeInvalidationMutex, Thread);
|
||||
|
||||
// NOTE: Other threads sharing the same CodeBuffer may reference
|
||||
// invalidated data ranges through their L1/L2 caches. This is
|
||||
// not currently a problem since FEX does not repurpose the
|
||||
// invalidated CodeBuffer memory range currently.
|
||||
ThreadRemoveCodeEntry(Thread, GuestRIP);
|
||||
}
|
||||
|
||||
@@ -283,6 +288,7 @@ public:
|
||||
fextl::unique_ptr<FEXCore::Core::DebugData> DebugData;
|
||||
uint64_t StartAddr;
|
||||
uint64_t Length;
|
||||
std::unique_lock<ForkableUniqueMutex> CodeBufferLock;
|
||||
};
|
||||
[[nodiscard]]
|
||||
CompileCodeResult CompileCode(FEXCore::Core::InternalThreadState* Thread, uint64_t GuestRIP, uint64_t MaxInst = 0);
|
||||
|
||||
@@ -6,6 +6,8 @@
|
||||
#include "Interface/Core/Dispatcher/Dispatcher.h"
|
||||
#include <cstdint>
|
||||
|
||||
#include "LookupCache.h"
|
||||
|
||||
#ifndef _WIN32
|
||||
#include <sys/prctl.h>
|
||||
#endif
|
||||
@@ -264,10 +266,10 @@ namespace CPU {
|
||||
return TotalLUT;
|
||||
}()};
|
||||
|
||||
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState* ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
|
||||
CPUBackend::CPUBackend(CodeBufferManager& manager, FEXCore::Core::InternalThreadState* ThreadState, size_t MaxCodeSize)
|
||||
: ThreadState(ThreadState)
|
||||
, InitialCodeSize(InitialCodeSize)
|
||||
, MaxCodeSize(MaxCodeSize) {
|
||||
, MaxCodeSize(MaxCodeSize)
|
||||
, manager(manager) {
|
||||
|
||||
auto& Common = ThreadState->CurrentFrame->Pointers.Common;
|
||||
|
||||
@@ -307,37 +309,38 @@ namespace CPU {
|
||||
CPUBackend::~CPUBackend() = default;
|
||||
|
||||
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer* {
|
||||
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
|
||||
if (CodeBuffers.empty()) {
|
||||
EmplaceNewCodeBuffer(manager.AllocateNew(InitialCodeSize));
|
||||
} else {
|
||||
// If we have more than one code buffer we are tracking then walk them and delete
|
||||
// This is a cleanup step
|
||||
CodeBuffers.resize(1);
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
|
||||
if (CurrentCodeBufferSize != MaxCodeSize) {
|
||||
CodeBuffers.clear();
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBuffer = manager.StartLargerCodeBuffer(MaxCodeSize);
|
||||
|
||||
// Resize the code buffer and reallocate our code size
|
||||
CurrentCodeBufferSize *= 1.5;
|
||||
CurrentCodeBufferSize = std::min(CurrentCodeBufferSize, MaxCodeSize);
|
||||
|
||||
EmplaceNewCodeBuffer(manager.AllocateNew(CurrentCodeBufferSize));
|
||||
}
|
||||
}
|
||||
} else {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Allocate some new code buffers that we can switch over to instead
|
||||
EmplaceNewCodeBuffer(manager.AllocateNew(InitialCodeSize));
|
||||
}
|
||||
|
||||
return CodeBuffers.back().get();
|
||||
RegisterForSignalHandler(PrevCodeBuffer);
|
||||
return CurrentCodeBuffer.get();
|
||||
}
|
||||
|
||||
void CPUBackend::EmplaceNewCodeBuffer(fextl::shared_ptr<CodeBuffer> Buffer) {
|
||||
CurrentCodeBufferSize = Buffer->Size;
|
||||
CodeBuffers.emplace_back(Buffer);
|
||||
void CPUBackend::RegisterForSignalHandler(fextl::shared_ptr<CodeBuffer> CodeBuffer) {
|
||||
if (ThreadState->CurrentFrame->SignalHandlerRefCounter != 0) {
|
||||
// We have signal handlers that have generated code
|
||||
// This means that we can not safely clear the code at this point in time
|
||||
// Keep a reference to the old code buffer to delay deallocation
|
||||
SignalHandlerCodeBuffers.push_back(CodeBuffer);
|
||||
} else {
|
||||
SignalHandlerCodeBuffers.clear();
|
||||
}
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CPUBackend::CheckCodeBufferUpdate() {
|
||||
fextl::shared_ptr<CodeBuffer> OldCodeBuffer;
|
||||
auto NewCodeBuffer = manager.GetLatest();
|
||||
if (CurrentCodeBuffer != NewCodeBuffer) {
|
||||
RegisterForSignalHandler(CurrentCodeBuffer);
|
||||
return std::exchange(CurrentCodeBuffer, NewCodeBuffer);
|
||||
}
|
||||
return nullptr;
|
||||
}
|
||||
|
||||
GuestToHostMap& GetLookupCache(const CodeBuffer& Buffer) {
|
||||
return *Buffer.LookupCache;
|
||||
}
|
||||
|
||||
CodeBuffer::CodeBuffer(size_t Size)
|
||||
@@ -351,10 +354,11 @@ namespace CPU {
|
||||
FEXCore::Allocator::ProtectOptions::None)) {
|
||||
LogMan::Msg::EFmt("Failed to mprotect last page of code buffer.");
|
||||
}
|
||||
|
||||
LookupCache = fextl::make_unique<GuestToHostMap>();
|
||||
}
|
||||
|
||||
CodeBuffer::~CodeBuffer() {
|
||||
// TODO: Assert that mutex is held?
|
||||
FEXCore::Allocator::VirtualFree(Ptr, Size);
|
||||
}
|
||||
|
||||
@@ -386,24 +390,50 @@ namespace CPU {
|
||||
|
||||
auto Buffer = fextl::make_shared<CodeBuffer>(Size);
|
||||
|
||||
// TODO: Re-enable
|
||||
// if (static_cast<Context::ContextImpl*>(ThreadState->CTX)->Config.GlobalJITNaming()) {
|
||||
// static_cast<Context::ContextImpl*>(ThreadState->CTX)->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
// }
|
||||
Latest = Buffer;
|
||||
LatestOffset = 0;
|
||||
|
||||
OnCodeBufferAllocated(*Buffer);
|
||||
|
||||
return Buffer;
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CodeBufferManager::GetLatest() {
|
||||
if (!Latest) {
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
AllocateNew(INITIAL_CODE_SIZE);
|
||||
}
|
||||
return Latest;
|
||||
}
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> CodeBufferManager::StartLargerCodeBuffer(size_t MaxCodeSize) {
|
||||
if (!Latest) {
|
||||
// Allocate initial CodeBuffer and return it
|
||||
return GetLatest();
|
||||
}
|
||||
|
||||
auto NewCodeBufferSize = GetLatest()->Size;
|
||||
NewCodeBufferSize = std::min<size_t>(NewCodeBufferSize * 2, MaxCodeSize);
|
||||
return AllocateNew(NewCodeBufferSize);
|
||||
}
|
||||
|
||||
|
||||
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
|
||||
// The last page of the code buffer is protected, so we need to exclude it from the valid range
|
||||
// when checking if the address is in the code buffer.
|
||||
for (auto& Buffer : CodeBuffers) {
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer->Ptr) + Buffer->Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
if (Address >= reinterpret_cast<uintptr_t>(Buffer->Ptr) && Address < LastPageAddr) {
|
||||
auto CheckCodeBuffer = [](CodeBuffer& Buffer, uintptr_t Address) {
|
||||
// The last page of the code buffer is protected, so we need to exclude it from the valid range
|
||||
// when checking if the address is in the code buffer.
|
||||
uintptr_t LastPageAddr = AlignDown(reinterpret_cast<uintptr_t>(Buffer.Ptr) + Buffer.Size - 1, FEXCore::Utils::FEX_PAGE_SIZE);
|
||||
return (Address >= reinterpret_cast<uintptr_t>(Buffer.Ptr) && Address < LastPageAddr);
|
||||
};
|
||||
|
||||
if (CheckCodeBuffer(*CurrentCodeBuffer, Address)) {
|
||||
return true;
|
||||
}
|
||||
for (auto& Buffer : SignalHandlerCodeBuffers) {
|
||||
if (CheckCodeBuffer(*Buffer, Address)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
@@ -9,6 +9,7 @@ $end_info$
|
||||
#pragma once
|
||||
|
||||
#include <FEXCore/Utils/CompilerDefs.h>
|
||||
#include <FEXCore/Utils/SignalScopeGuards.h>
|
||||
#include <FEXCore/fextl/memory.h>
|
||||
#include <FEXCore/fextl/string.h>
|
||||
#include <FEXCore/fextl/vector.h>
|
||||
@@ -32,11 +33,15 @@ namespace CodeSerialize {
|
||||
struct CodeObjectFileSection;
|
||||
}
|
||||
|
||||
struct GuestToHostMap;
|
||||
|
||||
namespace CPU {
|
||||
struct CodeBuffer {
|
||||
uint8_t* Ptr;
|
||||
size_t Size;
|
||||
|
||||
fextl::unique_ptr<GuestToHostMap> LookupCache;
|
||||
|
||||
CodeBuffer(size_t Size);
|
||||
CodeBuffer(const CodeBuffer&) = delete;
|
||||
CodeBuffer& operator=(const CodeBuffer&) = delete;
|
||||
@@ -46,8 +51,36 @@ namespace CPU {
|
||||
~CodeBuffer();
|
||||
};
|
||||
|
||||
/**
|
||||
* A manager that coordinates access to the CodeBuffer used for compiling new code across threads.
|
||||
*
|
||||
* The CodeBuffer is managed as a partially persistent data structure:
|
||||
* - Exactly one CodeBuffer is now designated as "active", which means data can be appended to it
|
||||
* - Lossy modifications to the active CodeBuffer will not invalidate any data in use by other threads (which is what enables save CodeBuffer sharing across threads)
|
||||
* - Instead, such lossy modifications trigger a new "version" of the data in the modifying thread. Old versions of the CodeBuffer persist as read-only data for use by the other threads.
|
||||
* - The other threads can update their version of the CodeBuffer. This will decrease the reference count and eventually trigger deallocation of the old version
|
||||
*/
|
||||
class CodeBufferManager {
|
||||
public:
|
||||
// Get the CodeBuffer that was most recently allocated.
|
||||
// This is the only CodeBuffer that data may be written to.
|
||||
fextl::shared_ptr<CodeBuffer> GetLatest();
|
||||
|
||||
// Allocate a new CodeBuffer with geometric growth.
|
||||
// Subsequent calls to GetLatest will point to the returned buffer.
|
||||
fextl::shared_ptr<CodeBuffer> StartLargerCodeBuffer(size_t MaxCodeSize);
|
||||
|
||||
// Write offset into the latest CodeBuffer
|
||||
std::size_t LatestOffset;
|
||||
|
||||
// Protects writes to the latest CodeBuffer
|
||||
FEXCore::ForkableUniqueMutex CodeBufferWriteMutex;
|
||||
|
||||
virtual void OnCodeBufferAllocated(CodeBuffer&) {};
|
||||
|
||||
private:
|
||||
fextl::shared_ptr<CodeBuffer> Latest;
|
||||
|
||||
fextl::shared_ptr<CodeBuffer> AllocateNew(size_t Size);
|
||||
};
|
||||
|
||||
@@ -55,10 +88,9 @@ namespace CPU {
|
||||
public:
|
||||
|
||||
/**
|
||||
* @param InitialCodeSize - Initial size for the code buffers
|
||||
* @param MaxCodeSize - Max size for the code buffers
|
||||
*/
|
||||
CPUBackend(FEXCore::Core::InternalThreadState*, size_t InitialCodeSize, size_t MaxCodeSize);
|
||||
CPUBackend(CodeBufferManager&, FEXCore::Core::InternalThreadState*, size_t MaxCodeSize);
|
||||
|
||||
virtual ~CPUBackend();
|
||||
|
||||
@@ -157,6 +189,11 @@ namespace CPU {
|
||||
|
||||
bool IsAddressInCodeBuffer(uintptr_t Address) const;
|
||||
|
||||
// Updates the CodeBuffer if needed and returns a reference to the old one.
|
||||
// The returned reference should be kept alive carefully to avoid early deletion of resources.
|
||||
[[nodiscard]]
|
||||
fextl::shared_ptr<CodeBuffer> CheckCodeBufferUpdate();
|
||||
|
||||
protected:
|
||||
// Max spill slot size in bytes. We need at most 32 bytes
|
||||
// to be able to handle a 256-bit vector store to a slot.
|
||||
@@ -164,21 +201,21 @@ namespace CPU {
|
||||
|
||||
FEXCore::Core::InternalThreadState* ThreadState;
|
||||
|
||||
size_t InitialCodeSize, MaxCodeSize;
|
||||
size_t MaxCodeSize;
|
||||
[[nodiscard]]
|
||||
CodeBuffer* GetEmptyCodeBuffer();
|
||||
|
||||
// This is the size of the last code buffer we allocated
|
||||
size_t CurrentCodeBufferSize = 0;
|
||||
// This is the code buffer containing the main code under execution by this thread.
|
||||
// CheckCodeBufferUpdate must be used before compiling new code.
|
||||
fextl::shared_ptr<CodeBuffer> CurrentCodeBuffer;
|
||||
|
||||
CodeBufferManager manager; // TODO: Rename
|
||||
// Old CodeBuffer generations required to be valid until returning from signal handlers
|
||||
fextl::vector<fextl::shared_ptr<CodeBuffer>> SignalHandlerCodeBuffers;
|
||||
|
||||
CodeBufferManager& manager; // TODO: Rename
|
||||
|
||||
private:
|
||||
void EmplaceNewCodeBuffer(fextl::shared_ptr<CodeBuffer> Buffer);
|
||||
|
||||
// This is the array of code buffers. Unless signals force us to keep more than
|
||||
// buffer, there will be only one entry here
|
||||
fextl::vector<fextl::shared_ptr<CodeBuffer>> CodeBuffers;
|
||||
void RegisterForSignalHandler(fextl::shared_ptr<CodeBuffer>);
|
||||
};
|
||||
|
||||
} // namespace CPU
|
||||
|
||||
@@ -495,7 +495,13 @@ void ContextImpl::LockBeforeFork(FEXCore::Core::InternalThreadState* Thread) {
|
||||
}
|
||||
#endif
|
||||
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
|
||||
void ContextImpl::OnCodeBufferAllocated(CPU::CodeBuffer& Buffer) {
|
||||
if (Config.GlobalJITNaming()) {
|
||||
Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
|
||||
}
|
||||
}
|
||||
|
||||
void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer) {
|
||||
FEXCORE_PROFILE_INSTANT("ClearCodeCache");
|
||||
|
||||
if (CodeObjectCacheService) {
|
||||
@@ -503,10 +509,14 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) {
|
||||
// Use the thread's object cache ref counter for this
|
||||
CodeSerialize::CodeObjectSerializeService::WaitForEmptyJobQueue(&Thread->ObjectCacheRefCounter);
|
||||
}
|
||||
auto lk = Thread->LookupCache->AcquireLock();
|
||||
|
||||
Thread->LookupCache->ClearCache();
|
||||
Thread->CPUBackend->ClearCache();
|
||||
if (NewCodeBuffer) {
|
||||
// Allocate new CodeBuffer + L3 LookupCache and clear L1+L2 caches
|
||||
Thread->CPUBackend->ClearCache();
|
||||
} else {
|
||||
// Clear L1+L2 cache of this thread, and clear L3 cache across any threads using it
|
||||
Thread->LookupCache->ClearCache();
|
||||
}
|
||||
}
|
||||
|
||||
static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP) {
|
||||
@@ -744,6 +754,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
.DebugData = nullptr, // nullptr here ensures that code serialization doesn't occur on from cache read
|
||||
.StartAddr = 0, // Unused
|
||||
.Length = 0, // Unused
|
||||
.CodeBufferLock {} // Unused
|
||||
};
|
||||
}
|
||||
}
|
||||
@@ -768,6 +779,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
|
||||
// Attempt to get the CPU backend to compile this code
|
||||
|
||||
auto Lock = std::unique_lock {CodeBufferWriteMutex};
|
||||
auto CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &*IRView, DebugData.get(), TFSet);
|
||||
|
||||
// Release the IR
|
||||
@@ -781,6 +793,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT
|
||||
.DebugData = std::move(DebugData),
|
||||
.StartAddr = StartAddr,
|
||||
.Length = Length,
|
||||
.CodeBufferLock = std::move(Lock),
|
||||
};
|
||||
}
|
||||
|
||||
@@ -800,7 +813,7 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_
|
||||
return HostCode;
|
||||
}
|
||||
|
||||
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
auto [CodePtr, DebugData, StartAddr, Length, CodeBufferLock] = CompileCode(Thread, GuestRIP, MaxInst);
|
||||
if (CodePtr == nullptr) {
|
||||
return 0;
|
||||
}
|
||||
@@ -872,7 +885,7 @@ uintptr_t ContextImpl::CompileSingleStep(FEXCore::Core::CpuStateFrame* Frame, ui
|
||||
// Invalidate might take a unique lock on this, to guarantee that during invalidation no code gets compiled
|
||||
auto lk = GuardSignalDeferringSection<std::shared_lock>(CodeInvalidationMutex, Thread);
|
||||
|
||||
auto [CodePtr, DebugData, StartAddr, Length] = CompileCode(Thread, GuestRIP, 1);
|
||||
auto [CodePtr, DebugData, StartAddr, Length, CodeBufferLock] = CompileCode(Thread, GuestRIP, 1);
|
||||
if (CodePtr == nullptr) {
|
||||
return 0;
|
||||
}
|
||||
@@ -911,8 +924,8 @@ void ContextImpl::MarkMemoryShared(FEXCore::Core::InternalThreadState* Thread) {
|
||||
UpdateAtomicTSOEmulationConfig();
|
||||
|
||||
if (Config.TSOAutoMigration) {
|
||||
// Only the lookup cache is cleared here, so that old code can keep running until next compilation
|
||||
auto lk = Thread->LookupCache->AcquireLock();
|
||||
// Only the lookup cache is cleared here, so that old code can keep running until next compilation.
|
||||
// This will leak previously compiled blocks until the CodeBuffer is cleared for some other reason.
|
||||
Thread->LookupCache->ClearCache();
|
||||
}
|
||||
}
|
||||
|
||||
@@ -40,7 +40,6 @@ $end_info$
|
||||
#include <string.h>
|
||||
#include <limits>
|
||||
|
||||
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
|
||||
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
|
||||
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
|
||||
|
||||
@@ -496,7 +495,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Fram
|
||||
void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::Ref Node) {}
|
||||
|
||||
Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread)
|
||||
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
|
||||
: CPUBackend(*ctx, Thread, MAX_CODE_SIZE)
|
||||
, Arm64Emitter(ctx)
|
||||
, HostSupportsSVE128 {ctx->HostFeatures.SupportsSVE128}
|
||||
, HostSupportsSVE256 {ctx->HostFeatures.SupportsSVE256}
|
||||
@@ -550,8 +549,8 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
|
||||
AArch64.LREM = reinterpret_cast<uint64_t>(LREM);
|
||||
}
|
||||
|
||||
// Must be done after Dispatcher init
|
||||
ClearCache();
|
||||
CurrentCodeBuffer = manager.GetLatest();
|
||||
ThreadState->LookupCache->Shared = CurrentCodeBuffer->LookupCache.get();
|
||||
|
||||
// Setup dynamic dispatch.
|
||||
if (ParanoidTSO()) {
|
||||
@@ -570,11 +569,15 @@ void Arm64JITCore::EmitDetectionString() {
|
||||
}
|
||||
|
||||
void Arm64JITCore::ClearCache() {
|
||||
// Get the backing code buffer
|
||||
// NOTE: Holding on to the reference here is required to ensure validity of the WriteLock mutex
|
||||
auto PrevCodeBuffer = CurrentCodeBuffer;
|
||||
std::lock_guard lk(PrevCodeBuffer->LookupCache->WriteLock);
|
||||
|
||||
auto CodeBuffer = GetEmptyCodeBuffer();
|
||||
SetBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
|
||||
EmitDetectionString();
|
||||
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*PrevCodeBuffer, *CurrentCodeBuffer->LookupCache);
|
||||
}
|
||||
|
||||
Arm64JITCore::~Arm64JITCore() {}
|
||||
@@ -693,7 +696,16 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
// Fairly excessive buffer range to make sure we don't overflow
|
||||
uint32_t BufferRange = SSACount * 16;
|
||||
if ((GetCursorOffset() + BufferRange) > (CurrentCodeBufferSize - Utils::FEX_PAGE_SIZE)) {
|
||||
|
||||
LOGMAN_THROW_A_FMT(CurrentCodeBuffer->LookupCache.get() == ThreadState->LookupCache->Shared, "INVARIANT VIOLATED: SharedLookupCache "
|
||||
"doesn't match up!\n");
|
||||
if (auto Prev = CheckCodeBufferUpdate()) {
|
||||
ThreadState->LookupCache->ChangeGuestToHostMapping(*Prev, *CurrentCodeBuffer->LookupCache);
|
||||
}
|
||||
|
||||
SetBuffer(CurrentCodeBuffer->Ptr, CurrentCodeBuffer->Size);
|
||||
SetCursorOffset(manager.LatestOffset);
|
||||
if ((GetCursorOffset() + BufferRange) > (CurrentCodeBuffer->Size - Utils::FEX_PAGE_SIZE)) {
|
||||
CTX->ClearCodeCache(ThreadState);
|
||||
}
|
||||
|
||||
@@ -867,6 +879,8 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size
|
||||
|
||||
JITBlockTail->Size = CodeData.Size;
|
||||
|
||||
manager.LatestOffset = GetCursorOffset();
|
||||
|
||||
ClearICache(CodeData.BlockBegin, CodeOnlySize);
|
||||
|
||||
#ifdef VIXL_DISASSEMBLER
|
||||
|
||||
@@ -65,20 +65,27 @@ LookupCache::~LookupCache() {
|
||||
}
|
||||
|
||||
void LookupCache::ClearL2Cache() {
|
||||
auto lk = L3.AcquireLock();
|
||||
auto lk = Shared->AcquireLock();
|
||||
// Clear out the page memory
|
||||
// PagePointer and PageMemory are sequential with each other. Clear both at once.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), ctx->Config.VirtualMemSize / 4096 * 8 + CODE_SIZE, false);
|
||||
AllocateOffset = 0;
|
||||
}
|
||||
|
||||
void LookupCache::ClearThreadLocalCaches() {
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
}
|
||||
|
||||
void LookupCache::ClearCache() {
|
||||
auto lk = L3.AcquireLock();
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
// Clear L1 and L2 by clearing the full cache.
|
||||
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
|
||||
|
||||
L3.ClearCache(lk);
|
||||
Shared->ClearCache(lk);
|
||||
}
|
||||
|
||||
void GuestToHostMap::ClearCache(const LockToken&) {
|
||||
|
||||
@@ -61,6 +61,10 @@ struct GuestToHostMap {
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode, const LockToken&) {
|
||||
// This may replace an existing mapping
|
||||
// NOTE: Generally no previous entry should exist, however there is one exception:
|
||||
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
|
||||
// may already contain the block address. Since is comparatively rare, we'll just leak
|
||||
// one of the two blocks in this case.
|
||||
BlockList[Address] = (uintptr_t)HostCode;
|
||||
}
|
||||
|
||||
@@ -102,6 +106,13 @@ public:
|
||||
LookupCache(FEXCore::Context::ContextImpl* CTX);
|
||||
~LookupCache();
|
||||
|
||||
// Swaps out the underlying GuestToHostMap and clears all associated caches.
|
||||
// This interface requires the previous CodeBuffer to be provided despite not using it. This ensures the shared write lock is still valid.
|
||||
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap) {
|
||||
ClearThreadLocalCaches();
|
||||
Shared = &NewMap;
|
||||
}
|
||||
|
||||
uintptr_t FindBlock(uint64_t Address) {
|
||||
// Try L1, no lock needed
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
@@ -110,7 +121,7 @@ public:
|
||||
}
|
||||
|
||||
// L2 and L3 need to be locked
|
||||
auto lk = L3.AcquireLock();
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
// Try L2
|
||||
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
||||
@@ -132,7 +143,7 @@ public:
|
||||
}
|
||||
|
||||
// Try L3
|
||||
auto HostCode = L3.FindBlock(Address, lk);
|
||||
auto HostCode = Shared->FindBlock(Address, lk);
|
||||
if (HostCode) {
|
||||
CacheBlockMapping(Address, HostCode.value());
|
||||
return HostCode.value();
|
||||
@@ -142,15 +153,14 @@ public:
|
||||
return 0;
|
||||
}
|
||||
|
||||
GuestToHostMap L3;
|
||||
GuestToHostMap* Shared = nullptr;
|
||||
|
||||
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
|
||||
|
||||
// Appends Block {Address} to CodePages [Start, Start + Length)
|
||||
// Returns true if new pages are marked as containing code
|
||||
bool AddBlockExecutableRange(uint64_t Address, uint64_t Start, uint64_t Length) {
|
||||
auto lk = L3.AcquireLock();
|
||||
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
bool rv = false;
|
||||
|
||||
@@ -165,9 +175,9 @@ public:
|
||||
|
||||
// Adds to Guest -> Host code mapping
|
||||
void AddBlockMapping(uint64_t Address, void* HostCode) {
|
||||
auto lk = L3.AcquireLock();
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
L3.AddBlockMapping(Address, HostCode, lk);
|
||||
Shared->AddBlockMapping(Address, HostCode, lk);
|
||||
|
||||
// There is no need to update L1 or L2, they will get updated on first lookup
|
||||
// However, adding to L1 here increases performance
|
||||
@@ -176,10 +186,13 @@ public:
|
||||
L1Entry.HostCode = (uintptr_t)HostCode;
|
||||
}
|
||||
|
||||
// NOTE: It's the caller's responsibility to call Erase() for all other
|
||||
// GuestToHostMaps that share the same LookupCache. Otherwise, the
|
||||
// L1/L2 caches will contain stale references to deallocated memory.
|
||||
void Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address) {
|
||||
auto lk = L3.AcquireLock();
|
||||
auto lk = Shared->AcquireLock();
|
||||
|
||||
L3.Erase(Frame, Address, lk);
|
||||
Shared->Erase(Frame, Address, lk);
|
||||
|
||||
// Do L1
|
||||
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
||||
@@ -209,12 +222,13 @@ public:
|
||||
}
|
||||
|
||||
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker) {
|
||||
auto lk = L3.AcquireLock();
|
||||
L3.AddBlockLink(GuestDestination, HostLink, delinker, lk);
|
||||
auto lk = Shared->AcquireLock();
|
||||
Shared->AddBlockLink(GuestDestination, HostLink, delinker, lk);
|
||||
}
|
||||
|
||||
void ClearCache();
|
||||
void ClearL2Cache();
|
||||
void ClearThreadLocalCaches();
|
||||
|
||||
uintptr_t GetL1Pointer() const {
|
||||
return L1Pointer;
|
||||
@@ -237,7 +251,7 @@ public:
|
||||
// This approach has not been fully vetted yet.
|
||||
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
|
||||
auto AcquireLock() {
|
||||
return L3.AcquireLock();
|
||||
return Shared->AcquireLock();
|
||||
}
|
||||
|
||||
private:
|
||||
|
||||
@@ -171,7 +171,7 @@ public:
|
||||
FEX_DEFAULT_VISIBILITY virtual void FinalizeAOTIRCache() = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void WriteFilesWithCode(AOTIRCodeFileWriterFn Writer) = 0;
|
||||
|
||||
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void ClearCodeCache(FEXCore::Core::InternalThreadState* Thread, bool NewCodeBuffer = true) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual void InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState* Thread, uint64_t Start, uint64_t Length) = 0;
|
||||
FEX_DEFAULT_VISIBILITY virtual FEXCore::ForkableSharedMutex& GetCodeInvalidationMutex() = 0;
|
||||
|
||||
@@ -181,7 +181,7 @@ public:
|
||||
ConfigureAOTGen(FEXCore::Core::InternalThreadState* Thread, fextl::set<uint64_t>* ExternalBranches, uint64_t SectionMaxAddress) = 0;
|
||||
|
||||
/**
|
||||
* @brief Checks if a PC is inside of a thread's JIT code buffer.
|
||||
* @brief Checks if a PC is inside any code buffer used by the thread's JIT.
|
||||
*
|
||||
* @param Thread Which thread's code buffers to check inside of.
|
||||
* @param Address The PC to check against.
|
||||
|
||||
@@ -297,7 +297,7 @@ void ThreadManager::Step() {
|
||||
// Walk the threads and tell them to clear their caches
|
||||
// Useful when our block size is set to a large number and we need to step a single instruction
|
||||
for (auto& Thread : Threads) {
|
||||
CTX->ClearCodeCache(Thread->Thread);
|
||||
CTX->ClearCodeCache(Thread->Thread, false);
|
||||
}
|
||||
}
|
||||
|
||||
|
||||
Reference in new issue
Block a user