mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-08 04:04:10 +02:00
This saves a whole bunch of memory. Cutting `Just Cause 2`'s title
screen from 1132MB anonymous FEX memory down to 438MB. 629MB in L2
alone.
L2 is primarily a means to reduce overhead in map queries, so it's all
about performance. But because it consumes a lot of people it's kind of
hard.
One idea is that the L2 lookups can be moved to shared data structures,
since we already pull the shared lock when doing an L2 lookup this is
already halfway there.
Side note, we're using unique locks even with read-only code paths
which we can't use the shared lock because this terrible recursive
mutex!
Instead of outright changing L2 behaviour and potentially wrecking
havoc, add a config option for now so testing can happen over time.
before:
```
Total FEX Anon memory resident: 1132 mB
JIT resident: 60 mB
OpDispatcher resident: 97 mB
Frontend resident: 37 mB
CPUBackend resident: 500 kB
Lookup cache resident: 629 mB
Lookup L1 cache resident: 108 mB
ThreadStates resident: 436 kB
```
after:
```
Total FEX Anon memory resident: 438 mB
JIT resident: 62 mB
OpDispatcher resident: 56 mB
Frontend resident: 22 mB
CPUBackend resident: 496 kB
Lookup cache resident: 0 (null)
Lookup L1 cache resident: 109 mB
ThreadStates resident: 436 kB
```
340 lines
12 KiB
C++
340 lines
12 KiB
C++
// SPDX-License-Identifier: MIT
|
|
#pragma once
|
|
#include "Interface/Context/Context.h"
|
|
#include <FEXCore/Utils/LogManager.h>
|
|
#include <FEXCore/fextl/map.h>
|
|
#include <FEXCore/fextl/memory_resource.h>
|
|
#include <FEXCore/fextl/robin_map.h>
|
|
#include <FEXCore/fextl/vector.h>
|
|
#include <FEXCore/fextl/memory_resource.h>
|
|
|
|
#include <cstdint>
|
|
#include <stddef.h>
|
|
#include <utility>
|
|
#include <mutex>
|
|
|
|
namespace FEXCore {
|
|
|
|
struct LookupCacheWriteLockToken {
|
|
private:
|
|
// Only constructible by GuestToHostMap
|
|
friend struct GuestToHostMap;
|
|
LookupCacheWriteLockToken(std::mutex& Mutex)
|
|
: Lock {Mutex} {}
|
|
std::lock_guard<std::mutex> Lock;
|
|
};
|
|
|
|
struct GuestToHostMap {
|
|
std::mutex WriteLock;
|
|
|
|
[[nodiscard]]
|
|
LookupCacheWriteLockToken AcquireWriteLock() {
|
|
return LookupCacheWriteLockToken {WriteLock};
|
|
}
|
|
|
|
struct BlockLinkTag {
|
|
uint64_t GuestDestination;
|
|
FEXCore::Context::ExitFunctionLinkData* HostLink;
|
|
|
|
bool operator<(const BlockLinkTag& other) const {
|
|
if (GuestDestination < other.GuestDestination) {
|
|
return true;
|
|
} else if (GuestDestination == other.GuestDestination) {
|
|
return HostLink < other.HostLink;
|
|
} else {
|
|
return false;
|
|
}
|
|
}
|
|
};
|
|
|
|
// Use a monotonic buffer resource to allocate both the std::pmr::map and its members.
|
|
// This allows us to quickly clear the block link map by clearing the monotonic allocator.
|
|
// If we had allocated the block link map without the MBR, then clearing the map would require slowly
|
|
// walking each block member and destructing objects.
|
|
//
|
|
// This makes `BlockLinks` look like a raw pointer that could memory leak, but since it is backed by the MBR, it won't.
|
|
fextl::pmr::named_monotonic_page_buffer_resource BlockLinks_mbr;
|
|
using BlockLinksMapType = std::pmr::map<BlockLinkTag, FEXCore::Context::BlockDelinkerFunc>;
|
|
fextl::unique_ptr<std::pmr::polymorphic_allocator<std::byte>> BlockLinks_pma;
|
|
BlockLinksMapType* BlockLinks;
|
|
|
|
fextl::robin_map<uint64_t, uint64_t> BlockList;
|
|
|
|
fextl::map<uint64_t, fextl::vector<uint64_t>> CodePages;
|
|
|
|
GuestToHostMap();
|
|
|
|
// Adds to Guest -> Host code mapping
|
|
void AddBlockMapping(uint64_t Address, void* HostCode, const LookupCacheWriteLockToken&) {
|
|
// This may replace an existing mapping
|
|
// NOTE: Generally no previous entry should exist, however there is one exception:
|
|
// If the backend updates the active thread's CodeBuffer, the new associated LookupCache
|
|
// may already contain the block address. Since is comparatively rare, we'll just leak
|
|
// one of the two blocks in this case.
|
|
BlockList[Address] = (uintptr_t)HostCode;
|
|
}
|
|
|
|
std::optional<uintptr_t> FindBlock(uint64_t Address, const LookupCacheWriteLockToken&) {
|
|
auto HostCode = BlockList.find(Address);
|
|
if (HostCode == BlockList.end()) {
|
|
return std::nullopt;
|
|
}
|
|
return HostCode->second;
|
|
}
|
|
|
|
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LookupCacheWriteLockToken&) {
|
|
// Sever any links to this block
|
|
auto lower = BlockLinks->lower_bound({Address, nullptr});
|
|
auto upper = BlockLinks->upper_bound({Address, reinterpret_cast<FEXCore::Context::ExitFunctionLinkData*>(UINTPTR_MAX)});
|
|
for (auto it = lower; it != upper; it = BlockLinks->erase(it)) {
|
|
it->second(Frame, it->first.HostLink);
|
|
}
|
|
|
|
// Remove from BlockList
|
|
return BlockList.erase(Address) != 0;
|
|
}
|
|
|
|
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
|
|
const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken&) {
|
|
BlockLinks->insert({{GuestDestination, HostLink}, delinker});
|
|
}
|
|
|
|
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length, const LookupCacheWriteLockToken&) {
|
|
bool rv = false;
|
|
|
|
for (auto CurrentPage = Start >> 12, EndPage = (Start + Length - 1) >> 12; CurrentPage <= EndPage; CurrentPage++) {
|
|
auto& CodePage = CodePages[CurrentPage];
|
|
rv |= CodePage.empty();
|
|
CodePage.insert(CodePage.end(), Addresses.begin(), Addresses.end());
|
|
}
|
|
|
|
return rv;
|
|
}
|
|
|
|
void ClearCache(const LookupCacheWriteLockToken&);
|
|
};
|
|
|
|
class LookupCache {
|
|
public:
|
|
struct LookupCacheEntry {
|
|
uintptr_t HostCode;
|
|
uintptr_t GuestCode;
|
|
};
|
|
|
|
LookupCache(FEXCore::Context::ContextImpl* CTX);
|
|
~LookupCache();
|
|
|
|
// Swaps out the underlying GuestToHostMap and clears all associated caches.
|
|
// This interface requires the previous CodeBuffer to be provided despite not using it. This ensures the shared write lock is still valid.
|
|
void ChangeGuestToHostMapping([[maybe_unused]] CPU::CodeBuffer& Prev, GuestToHostMap& NewMap, const LookupCacheWriteLockToken& lk) {
|
|
ClearThreadLocalCaches(lk);
|
|
Shared = &NewMap;
|
|
}
|
|
|
|
uintptr_t FindBlock(uint64_t Address) {
|
|
// Try L1, no lock needed
|
|
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
|
if (L1Entry.GuestCode == Address) {
|
|
return L1Entry.HostCode;
|
|
}
|
|
|
|
// L2 and L3 need to be locked
|
|
auto lk = Shared->AcquireWriteLock();
|
|
|
|
if (!DisableL2Cache()) {
|
|
// Try L2
|
|
const auto PageIndex = (Address & (VirtualMemSize - 1)) >> 12;
|
|
const auto PageOffset = Address & (0x0FFF);
|
|
|
|
const auto Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
|
auto LocalPagePointer = Pointers[PageIndex];
|
|
|
|
// Do we a page pointer for this address?
|
|
if (LocalPagePointer) {
|
|
// Find there pointer for the address in the blocks
|
|
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
|
|
|
if (BlockPointers[PageOffset].GuestCode == Address) {
|
|
L1Entry.GuestCode = Address;
|
|
L1Entry.HostCode = BlockPointers[PageOffset].HostCode;
|
|
return L1Entry.HostCode;
|
|
}
|
|
}
|
|
}
|
|
|
|
// Try L3
|
|
auto HostCode = Shared->FindBlock(Address, lk);
|
|
if (HostCode) {
|
|
CacheBlockMapping(Address, HostCode.value(), lk);
|
|
return HostCode.value();
|
|
}
|
|
|
|
// Failed to find
|
|
return 0;
|
|
}
|
|
|
|
GuestToHostMap* Shared = nullptr;
|
|
|
|
// Appends a list of Block {Address} to CodePages [Start, Start + Length)
|
|
// Returns true if new pages are marked as containing code
|
|
bool AddBlockExecutableRange(const fextl::set<uint64_t>& Addresses, uint64_t Start, uint64_t Length) {
|
|
auto lk = Shared->AcquireWriteLock();
|
|
return Shared->AddBlockExecutableRange(Addresses, Start, Length, lk);
|
|
}
|
|
|
|
// Adds to Guest -> Host code mapping
|
|
void AddBlockMapping(uint64_t Address, void* HostCode) {
|
|
auto lk = Shared->AcquireWriteLock();
|
|
|
|
Shared->AddBlockMapping(Address, HostCode, lk);
|
|
|
|
// There is no need to update L1 or L2, they will get updated on first lookup
|
|
// However, adding to L1 here increases performance
|
|
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
|
L1Entry.GuestCode = Address;
|
|
L1Entry.HostCode = (uintptr_t)HostCode;
|
|
}
|
|
|
|
// NOTE: It's the caller's responsibility to call Erase() for all other
|
|
// GuestToHostMaps that share the same LookupCache. Otherwise, the
|
|
// L1/L2 caches will contain stale references to deallocated memory.
|
|
bool Erase(FEXCore::Core::CpuStateFrame* Frame, uint64_t Address, const LookupCacheWriteLockToken& lk) {
|
|
bool ErasedAny = Shared->Erase(Frame, Address, lk);
|
|
|
|
// Do L1
|
|
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
|
if (L1Entry.GuestCode == Address) {
|
|
L1Entry.GuestCode = 0;
|
|
ErasedAny = true;
|
|
// Leave L1Entry.HostCode as is, so that concurrent lookups won't read a null pointer
|
|
// This is a soft guarantee for cross thread invalidation, as atomics are not used
|
|
// and it hasn't been thoroughly tested
|
|
}
|
|
|
|
if (!DisableL2Cache()) {
|
|
// Do full map
|
|
Address = Address & (VirtualMemSize - 1);
|
|
uint64_t PageOffset = Address & (0x0FFF);
|
|
Address >>= 12;
|
|
|
|
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
|
uint64_t LocalPagePointer = Pointers[Address];
|
|
if (!LocalPagePointer) {
|
|
// Page for this code didn't even exist, nothing to do
|
|
return ErasedAny;
|
|
}
|
|
|
|
// Page exists, just set the offset to zero
|
|
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
|
BlockPointers[PageOffset].GuestCode = 0;
|
|
BlockPointers[PageOffset].HostCode = 0;
|
|
}
|
|
return true;
|
|
}
|
|
|
|
void AddBlockLink(uint64_t GuestDestination, FEXCore::Context::ExitFunctionLinkData* HostLink,
|
|
const FEXCore::Context::BlockDelinkerFunc& delinker, const LookupCacheWriteLockToken& lk) {
|
|
Shared->AddBlockLink(GuestDestination, HostLink, delinker, lk);
|
|
}
|
|
|
|
void ClearCache(const LookupCacheWriteLockToken&);
|
|
void ClearL2Cache(const LookupCacheWriteLockToken&);
|
|
void ClearThreadLocalCaches(const LookupCacheWriteLockToken&);
|
|
|
|
uintptr_t GetL1Pointer() const {
|
|
return L1Pointer;
|
|
}
|
|
uintptr_t GetPagePointer() const {
|
|
return PagePointer;
|
|
}
|
|
uintptr_t GetVirtualMemorySize() const {
|
|
return VirtualMemSize;
|
|
}
|
|
|
|
constexpr static size_t L1_ENTRIES = 1 * 1024 * 1024; // Must be a power of 2
|
|
constexpr static size_t L1_ENTRIES_MASK = L1_ENTRIES - 1;
|
|
|
|
// This needs to be taken before reads or writes to L2, L3, CodePages,
|
|
// and before writes to L1. Concurrent access from a thread that this LookupCache doesn't belong to
|
|
// may only happen during cross thread invalidation (::Erase).
|
|
// All other operations must be done from the owning thread.
|
|
// Some care is taken so that L1 lookups can be done without locks, and even tearing is unlikely to lead to a crash.
|
|
// This approach has not been fully vetted yet.
|
|
// Also note that L1 lookups might be inlined in the JIT Dispatcher and/or block ends.
|
|
auto AcquireWriteLock() {
|
|
return Shared->AcquireWriteLock();
|
|
}
|
|
|
|
private:
|
|
void CacheBlockMapping(uint64_t Address, uintptr_t HostCode, const LookupCacheWriteLockToken& lk) {
|
|
// Do L1
|
|
auto& L1Entry = reinterpret_cast<LookupCacheEntry*>(L1Pointer)[Address & L1_ENTRIES_MASK];
|
|
L1Entry.GuestCode = Address;
|
|
L1Entry.HostCode = HostCode;
|
|
|
|
if (!DisableL2Cache()) {
|
|
// Do ful map
|
|
auto FullAddress = Address;
|
|
Address = Address & (VirtualMemSize - 1);
|
|
|
|
uint64_t PageOffset = Address & (0x0FFF);
|
|
Address >>= 12;
|
|
|
|
uintptr_t* Pointers = reinterpret_cast<uintptr_t*>(PagePointer);
|
|
uint64_t LocalPagePointer = Pointers[Address];
|
|
if (!LocalPagePointer) {
|
|
// We don't have a page pointer for this address
|
|
// Allocate one now if we can
|
|
uintptr_t NewPageBacking = AllocateBackingForPage();
|
|
if (!NewPageBacking) {
|
|
// Couldn't allocate, clear L2 and retry
|
|
ClearL2Cache(lk);
|
|
CacheBlockMapping(Address, HostCode, lk);
|
|
return;
|
|
}
|
|
Pointers[Address] = NewPageBacking;
|
|
LocalPagePointer = NewPageBacking;
|
|
}
|
|
|
|
// Add the new pointer to the page block
|
|
auto BlockPointers = reinterpret_cast<LookupCacheEntry*>(LocalPagePointer);
|
|
|
|
// This silently replaces existing mappings
|
|
BlockPointers[PageOffset].GuestCode = FullAddress;
|
|
BlockPointers[PageOffset].HostCode = HostCode;
|
|
}
|
|
}
|
|
|
|
uintptr_t AllocateBackingForPage() {
|
|
uintptr_t NewBase = AllocateOffset;
|
|
uintptr_t NewEnd = AllocateOffset + SIZE_PER_PAGE;
|
|
|
|
if (NewEnd >= CODE_SIZE) {
|
|
// We ran out of block backing space. Need to clear the block cache and tell the JIT cores to clear their caches as well
|
|
// Tell whatever is calling this that it needs to do it.
|
|
return 0;
|
|
}
|
|
|
|
AllocateOffset = NewEnd;
|
|
return PageMemory + NewBase;
|
|
}
|
|
|
|
uintptr_t PagePointer;
|
|
uintptr_t PageMemory;
|
|
uintptr_t L1Pointer;
|
|
|
|
size_t TotalCacheSize;
|
|
|
|
constexpr static size_t CODE_SIZE = 128 * 1024 * 1024;
|
|
constexpr static size_t SIZE_PER_PAGE = 4096 * sizeof(LookupCacheEntry);
|
|
constexpr static size_t L1_SIZE = L1_ENTRIES * sizeof(LookupCacheEntry);
|
|
|
|
size_t AllocateOffset {};
|
|
|
|
FEXCore::Context::ContextImpl* ctx;
|
|
uint64_t VirtualMemSize {};
|
|
|
|
FEX_CONFIG_OPT(DisableL2Cache, DISABLEL2CACHE);
|
|
};
|
|
} // namespace FEXCore
|