Files
FEX-Emu--FEX/FEXCore/Source/Interface/Core/LookupCache.cpp
T
Billy Laws cf4478eeee LookupCache: Introduce two-pass code invalidation model
Shared code buffer support introduced the concept of having a single
GuestToHostMaps shared across many threads. In the common case all
threads will share one however if e.g. a resize recently occured and
specific thread is yet to compile any code with the new codebuffer it
will still use the old GuestToHostMap. The current invalidation
approach handles this by repeatedly calling erase for every single
thread's GuestToHostMap, even if it is repeated. An accumulator is used
to ensure when two threads share a map, the L1/L2 cache entries in the
second thread will still be invalidated even if the the iteration for
the first thread removed them from the map.

Unfortunately this is incredibly slow in cases with many threads, as
a significant number of redundant map lookups and L1/L2 cache erasures
on threads that never even observed a given block can occur. Solve this
by introducing a two-pass model:
- First, all active codebuffers (and their associated GuestToHostMaps)
  have their entries invalidated for the given range, these codebuffers
  are tracked internally within FEXCore. It is at this point that delinking
  callbacks are ran.
- Second, each thread will have its caches invalidated. But rather than
  naively invalidating the L1/L2 caches for every invalidated block for
  every thread, threads now track on their own what specific entries
  have been potentially fetched into their L1/L2 caches. This is
  aided by GuestToHostMap now tracking the pages each block touches. (an
  inverse CodePages so to speak).
2025-11-20 00:38:03 +00:00

106 lines
3.9 KiB
C++

// SPDX-License-Identifier: MIT
/*
$info$
tags: glue|block-database
desc: Stores information about blocks, and provides C++ implementations to lookup the blocks
$end_info$
*/
#include <FEXCore/Utils/Allocator.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/HLE/SyscallHandler.h>
#include "Interface/Context/Context.h"
#include "Interface/Core/LookupCache.h"
namespace FEXCore {
GuestToHostMap::GuestToHostMap()
: BlockLinks_mbr {"FEXMem_BlockLinks"} {
BlockLinks_pma = fextl::make_unique<std::pmr::polymorphic_allocator<std::byte>>(&BlockLinks_mbr);
// Setup our PMR map.
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
}
LookupCache::LookupCache(FEXCore::Context::ContextImpl* CTX)
: ctx {CTX} {
TotalCacheSize = ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE + MAX_L1_SIZE;
// Block cache ends up looking like this
// PageMemoryMap[VirtualMemoryRegion >> 12]
// |
// v
// PageMemory[Memory & (VIRTUAL_PAGE_SIZE - 1)]
// |
// v
// Pointer to Code
//
// Allocate a region of memory that we can use to back our block pointers
// We need one pointer per page of virtual memory
// At 64GB of virtual memory this will allocate 128MB of virtual memory space
PagePointer = reinterpret_cast<uintptr_t>(FEXCore::Allocator::VirtualAlloc(TotalCacheSize, false, false));
LOGMAN_THROW_A_FMT(PagePointer != -1ULL, "Failed to allocate PagePointer");
FEXCore::Allocator::VirtualName("FEXMem_Lookup", reinterpret_cast<void*>(PagePointer),
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE);
CTX->SyscallHandler->MarkOvercommitRange(PagePointer, TotalCacheSize);
// Allocate our memory backing our pages
// We need 32KB per guest page (One pointer per byte)
// XXX: We can drop down to 16KB if we store 4byte offsets from the code base
// We currently limit to 128MB of real memory for caching for the total cache size.
// Can end up being inefficient if we compile a small number of blocks per page
PageMemory = PagePointer + ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8;
// L1 Cache
L1Pointer = PageMemory + CODE_SIZE;
FEXCore::Allocator::VirtualName("FEXMem_Lookup_L1", reinterpret_cast<void*>(L1Pointer), MAX_L1_SIZE);
VirtualMemSize = ctx->Config.VirtualMemSize;
if (DynamicL1Cache()) {
// Start at minimum size when dynamic.
L1PointerMask = MIN_L1_ENTRIES - 1;
} else {
// Start at maximum instead.
L1PointerMask = MAX_L1_ENTRIES - 1;
}
}
LookupCache::~LookupCache() {
FEXCore::Allocator::VirtualFree(reinterpret_cast<void*>(PagePointer), TotalCacheSize);
ctx->SyscallHandler->UnmarkOvercommitRange(PagePointer, TotalCacheSize);
// No need to free BlockLinks map.
// These will get freed when their memory allocators are deallocated.
}
void LookupCache::ClearL2Cache(const FEXCore::LookupCacheBaseLockToken& lk) {
// Clear out the page memory
// PagePointer and PageMemory are sequential with each other. Clear both at once.
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer),
ctx->Config.VirtualMemSize / FEXCore::Utils::FEX_PAGE_SIZE * 8 + CODE_SIZE, false);
AllocateOffset = 0;
}
void LookupCache::ClearThreadLocalCaches(const LookupCacheWriteLockToken&) {
// Clear L1 and L2 by clearing the full cache.
FEXCore::Allocator::VirtualDontNeed(reinterpret_cast<void*>(PagePointer), TotalCacheSize, false);
CachedCodePages.clear();
}
void LookupCache::ClearCache(const LookupCacheWriteLockToken& lk) {
// Clear L1 and L2 by clearing the full cache.
ClearThreadLocalCaches(lk);
Shared->ClearCache(lk);
}
void GuestToHostMap::ClearCache(const LookupCacheWriteLockToken&) {
// Allocate a new pointer from the BlockLinks pma again.
BlockLinks = BlockLinks_pma->new_object<BlockLinksMapType>();
// All code is gone, clear the block list
BlockList.clear();
}
} // namespace FEXCore