// SPDX-License-Identifier: MIT #pragma once #include "Interface/Context/Context.h" #include #include #include #include #include #include #include #include #include #include #include #include #include namespace FEXCore { struct LookupCacheBaseLockToken { protected: // Protected constructor - only derived classes can construct LookupCacheBaseLockToken() = default; }; struct LookupCacheWriteLockToken : public LookupCacheBaseLockToken { private: // Only constructible by GuestToHostMap friend struct GuestToHostMap; LookupCacheWriteLockToken(FEXCore::Utils::WritePriorityMutex::Mutex& Mutex) : Lock {Mutex} {} std::lock_guard Lock; }; struct LookupCacheReadLockToken : public LookupCacheBaseLockToken { private: // Only constructible by GuestToHostMap friend struct GuestToHostMap; LookupCacheReadLockToken(FEXCore::Utils::WritePriorityMutex::Mutex& Mutex) : Lock {Mutex} {} std::shared_lock Lock; }; struct GuestToHostMap { FEXCore::Utils::WritePriorityMutex::Mutex Lock {}; [[nodiscard]] LookupCacheWriteLockToken AcquireWriteLock() { return LookupCacheWriteLockToken {Lock}; } [[nodiscard]] LookupCacheReadLockToken AcquireReadLock() { return LookupCacheReadLockToken {Lock}; } struct BlockLinkTag { uint64_t GuestDestination; FEXCore::Context::ExitFunctionLinkData* HostLink; bool operator<(const BlockLinkTag& other) const { if (GuestDestination < other.GuestDestination) { return true; } else if (GuestDestination == other.GuestDestination) { return HostLink < other.HostLink; } else { return false; } } }; // Use a monotonic buffer resource to allocate both the std::pmr::map and its members. // This allows us to quickly clear the block link map by clearing the monotonic allocator. // If we had allocated the block link map without the MBR, then clearing the map would require slowly // walking each block member and destructing objects. // // This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't. fextl::pmr::named_monotonic_page_buffer_resource BlockLinks_mbr; using BlockLinksMapType = std::pmr::map; fextl::unique_ptr> BlockLinks_pma; BlockLinksMapType* BlockLinks; struct BlockEntry { uint64_t HostCode; fextl::vector CodePages; }; fextl::robin_map BlockList; fextl::map> CodePages; GuestToHostMap(); // Adds to Guest -> Host code mapping const BlockEntry& AddBlockMapping(uint64_t Address, const fextl::vector& CodePages, void* HostCode, const LookupCacheWriteLockToken&) { // This may replace an existing mapping // NOTE: Generally no previous entry should exist, however there is one exception: // If the backend updates the active thread's CodeBuffer, the new associated LookupCache // may already contain the block address. Since is comparatively rare, we'll just leak // one of the two blocks in this case. return BlockList.insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, CodePages}).first->second; } const BlockEntry* FindBlock(uint64_t Address, const LookupCacheReadLockToken&) { auto HostCode = BlockList.find(Address); if (HostCode == BlockList.end()) { return nullptr; } return &HostCode->second; } bool Erase(uint64_t Address, const LookupCacheWriteLockToken&) { // Sever any links to this block auto lower = BlockLinks->lower_bound({Address, nullptr}); auto upper = BlockLinks->upper_bound({Address, reinterpret_cast(UINTPTR_MAX)}); for (auto it = lower; it != upper; it = BlockLinks->erase(it)) { it->second(it->first.HostLink); } // Remove from BlockList return BlockList.erase(Address) != 0; } void InvalidateRange(uint64_t Start, uint64_t Length) { auto lk = AcquireWriteLock(); auto lower = CodePages.lower_bound(Start >> 12); auto upper = CodePages.upper_bound((Start + Length - 1) >> 12); for (auto it = lower; it != upper; it++) { for (const auto& Entry : it->second) { Erase(Entry, lk); } } CodePages.erase(lower, upper); } void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken&) { BlockLinks->insert({{GuestDestination, HostLink}, delinker}); } bool AddBlockExecutableRange(const std::ranges::input_range auto& Addresses, uint64_t Start, uint64_t Length, const LookupCacheWriteLockToken&) { bool rv = false; for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) { auto& CodePage = CodePages[CurrentPage]; rv |= CodePage.empty(); CodePage.insert(CodePage.end(), Addresses.begin(), Addresses.end()); } return rv; } void ClearCache(const LookupCacheWriteLockToken&); }; class LookupCache { public: struct LookupCacheEntry { uintptr_t HostCode; uintptr_t GuestCode; }; LookupCache(FEXCore::Context::ContextImpl* CTX); ~LookupCache(); // Swaps out the underlying GuestToHostMap and clears all associated caches. // This interface requires the previous CodeBuffer to be provided despite not using it. This ensures the shared write lock is still valid. void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap, const LookupCacheWriteLockToken& lk) { ClearThreadLocalCaches(lk); Shared = &NewMap; } uintptr_t FindBlock(FEXCore::Core::InternalThreadState* Thread, uint64_t Address) { // Try L1, no lock needed auto& L1Entry = reinterpret_cast(L1Pointer)[Address & L1PointerMask]; if (L1Entry.GuestCode == Address) { return L1Entry.HostCode; } // L2 and L3 need to be locked uintptr_t HostPtr {}; { std::optional> LockTime( Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheReadLockTime : nullptr); auto lk = Shared->AcquireReadLock(); LockTime.reset(); if (!DisableL2Cache()) { // Try L2 const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12; const auto PageOffset = Address & (0x0FFF); const auto Pointers = reinterpret_cast(PagePointer); auto LocalPagePointer = Pointers[PageIndex]; // Do we a page pointer for this address? if (LocalPagePointer) { // Find there pointer for the address in the blocks auto BlockPointers = reinterpret_cast(LocalPagePointer); if (BlockPointers[PageOffset].GuestCode == Address) { L1Entry.GuestCode = Address; L1Entry.HostCode = BlockPointers[PageOffset].HostCode; HostPtr = L1Entry.HostCode; } } } if (!HostPtr) { // Try L3 auto Entry = Shared->FindBlock(Address, lk); if (Entry) { CacheBlockMapping(Address, *Entry, false, lk); HostPtr = Entry->HostCode; } } } if (HostPtr && DynamicL1Cache()) { UpdateDynamicL1Stats(Thread); } FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedCacheMissCount, 1); return HostPtr; } void UpdateDynamicL1Stats(FEXCore::Core::InternalThreadState* Thread) { // If host pointer was found in L2 or L3, then add it to the counter. // Keeping track not L1 misses, but specifically L2/L3 hits. ++L2L3CacheHits; const auto CurrentTime = std::chrono::system_clock::now(); const auto Period = CurrentTime - LastPeriod; if (Period >= SamplePeriod) { // If larger than the sample period then check if we need to increase L1 cache size. const double AveragePerSecond = static_cast(L2L3CacheHits) / static_cast(std::chrono::duration_cast(Period).count()) * 1000.0; if (AveragePerSecond >= DynamicL1CacheIncreaseCountHeuristic()) { if (CurrentL1Entries < MAX_L1_ENTRIES) { CurrentL1Entries <<= 1; L1PointerMask = CurrentL1Entries - 1; // Update the thread's L1 pointer mask to increase how much cache it uses. // Since we're in C-code, this is safe to update here. Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask(); } } else if (AveragePerSecond < DynamicL1CacheDecreaseCountHeuristic()) { if (CurrentL1Entries > MIN_L1_ENTRIES) { CurrentL1Entries >>= 1; L1PointerMask = CurrentL1Entries - 1; // Madvise the entries that we are dropping. Gives the memory back to the OS. LookupCacheEntry* FirstZeroL1Entry = &reinterpret_cast(L1Pointer)[CurrentL1Entries]; size_t ZeroMemorySize = (MAX_L1_ENTRIES - CurrentL1Entries) * sizeof(LookupCacheEntry); FEXCore::Allocator::VirtualDontNeed(FirstZeroL1Entry, ZeroMemorySize, false); // Update the thread's L1 pointer mask to increase how much cache it uses. // Since we're in C-code, this is safe to update here. Thread->CurrentFrame->State.L1Mask = GetScaledL1PointerMask(); } } // Update Last period to start again. LastPeriod = CurrentTime; L2L3CacheHits = 0; } } GuestToHostMap* Shared = nullptr; // Appends a list of Block {Address} to CodePages [Start, Start + Length) // Returns true if new pages are marked as containing code bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, const fextl::set& Addresses, uint64_t Start, uint64_t Length) { std::optional> LockTime( Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr); auto lk = Shared->AcquireWriteLock(); LockTime.reset(); return Shared->AddBlockExecutableRange(Addresses, Start, Length, lk); } // Adds to Guest -> Host code mapping void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, const fextl::vector& CodePages, void* HostCode) { std::optional> LockTime( Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr); auto lk = Shared->AcquireWriteLock(); LockTime.reset(); const auto& Entry = Shared->AddBlockMapping(Address, CodePages, HostCode, lk); // There is no need to update L1 or L2, they will get updated on first lookup // However, adding to L1 here increases performance CacheBlockMapping(Address, Entry, true, lk); } // Invalidates L1/L2 for a given guest block void InvalidateCache(uint64_t Address, const LookupCacheWriteLockToken& lk) { // Do L1 auto& L1Entry = reinterpret_cast(L1Pointer)[Address & L1PointerMask]; if (L1Entry.GuestCode == Address) { L1Entry.GuestCode = 0; // Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer // This is a soft guarantee for cross thread invalidation, as atomics are not used // and it hasn't been thoroughly tested } if (!DisableL2Cache()) { // Do full map Address = Address & (VirtualMemSize - 1); uint64_t PageOffset = Address & (0x0FFF); Address >>= 12; uintptr_t* Pointers = reinterpret_cast(PagePointer); uint64_t LocalPagePointer = Pointers[Address]; if (!LocalPagePointer) { // Page for this code didn't even exist, nothing to do return; } // Page exists, just set the offset to zero auto BlockPointers = reinterpret_cast(LocalPagePointer); BlockPointers[PageOffset].GuestCode = 0; BlockPointers[PageOffset].HostCode = 0; } } // Invalidates all L1/L2 entries for all guest block that intersect the given range bool InvalidateCacheRange(uint64_t Start, uint64_t Length) { auto lk = Shared->AcquireWriteLock(); auto lower = CachedCodePages.lower_bound(Start >> 12); auto upper = CachedCodePages.upper_bound((Start + Length - 1) >> 12); for (auto it = lower; it != upper; it++) { for (const auto& Entry : it->second) { InvalidateCache(Entry, lk); } } bool ret = upper != lower; CachedCodePages.erase(lower, upper); return ret; } void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink, const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken& lk) { Shared->AddBlockLink(GuestDestination, HostLink, delinker, lk); } void ClearCache(const LookupCacheWriteLockToken&); void ClearL2Cache(const LookupCacheBaseLockToken&); void ClearThreadLocalCaches(const LookupCacheWriteLockToken&); uintptr_t GetL1Pointer() const { return L1Pointer; } uintptr_t GetScaledL1PointerMask() const { return L1PointerMask << FEXCore::ilog2(sizeof(LookupCache::LookupCacheEntry)); } uintptr_t GetPagePointer() const { return PagePointer; } uintptr_t GetVirtualMemorySize() const { return VirtualMemSize; } // This needs to be taken before reads or writes to L2, L3, CodePages, // and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to // may only happen during cross thread invalidation (::Erase). // All other operations must be done from the owning thread. // Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash. // This approach has not been fully vetted yet. // Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends. auto AcquireWriteLock() { return Shared->AcquireWriteLock(); } private: void CacheBlockMapping(uint64_t Address, const GuestToHostMap::BlockEntry& Entry, bool L1Only, const LookupCacheBaseLockToken& lk) { for (const auto& CodePage : Entry.CodePages) { CachedCodePages[CodePage >> 12].insert(Address); } // Do L1 auto& L1Entry = reinterpret_cast(L1Pointer)[Address & L1PointerMask]; L1Entry.GuestCode = Address; L1Entry.HostCode = Entry.HostCode; if (!DisableL2Cache() && !L1Only) { // Do ful map auto FullAddress = Address; Address = Address & (VirtualMemSize - 1); uint64_t PageOffset = Address & (0x0FFF); Address >>= 12; uintptr_t* Pointers = reinterpret_cast(PagePointer); uint64_t LocalPagePointer = Pointers[Address]; if (!LocalPagePointer) { // We don't have a page pointer for this address // Allocate one now if we can uintptr_t NewPageBacking = AllocateBackingForPage(); if (!NewPageBacking) { // Couldn't allocate, clear L2 and retry ClearL2Cache(lk); CacheBlockMapping(FullAddress, Entry, false, lk); return; } Pointers[Address] = NewPageBacking; LocalPagePointer = NewPageBacking; } // Add the new pointer to the page block auto BlockPointers = reinterpret_cast(LocalPagePointer); // This silently replaces existing mappings BlockPointers[PageOffset].GuestCode = FullAddress; BlockPointers[PageOffset].HostCode = Entry.HostCode; } } uintptr_t AllocateBackingForPage() { uintptr_t NewBase = AllocateOffset; uintptr_t NewEnd = AllocateOffset + SIZE_PER_PAGE; if (NewEnd >= CODE_SIZE) { // We ran out of block backing space. Need to clear the block cache and tell the JIT cores to clear their caches as well // Tell whatever is calling this that it needs to do it. return 0; } AllocateOffset = NewEnd; return PageMemory + NewBase; } // Maps from a page index to all blocks in the page that have at some point been fetched into L1/L2 fextl::map> CachedCodePages; uintptr_t PagePointer; uintptr_t PageMemory; uintptr_t L1Pointer; uintptr_t L1PointerMask; size_t TotalCacheSize; // Start with 8k entries in L1 to give 128KB of L1 cache to each thread. // Max out at 1 million entries to give each thread 16MB of L1 cache maximum. constexpr static size_t MIN_L1_ENTRIES = 8 * 1024; // Must be a power of 2 constexpr static size_t MAX_L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2 constexpr static size_t CODE_SIZE = 128 * 1024 * 1024; constexpr static size_t SIZE_PER_PAGE = FEXCore::Utils::FEX_PAGE_SIZE * sizeof(LookupCacheEntry); constexpr static size_t MAX_L1_SIZE = MAX_L1_ENTRIES * sizeof(LookupCacheEntry); size_t AllocateOffset {}; FEXCore::Context::ContextImpl* ctx; uint64_t VirtualMemSize {}; size_t CurrentL1Entries = MIN_L1_ENTRIES; uint64_t L2L3CacheHits {}; std::chrono::time_point LastPeriod {}; constexpr static std::chrono::seconds SamplePeriod {1}; FEX_CONFIG_OPT(DynamicL1CacheIncreaseCountHeuristic, DYNAMICL1CACHEINCREASECOUNTHEURISTIC); FEX_CONFIG_OPT(DynamicL1CacheDecreaseCountHeuristic, DYNAMICL1CACHEDECREASECOUNTHEURISTIC); FEX_CONFIG_OPT(DynamicL1Cache, DYNAMICL1CACHE); FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE); }; } // namespace FEXCore