From b3f902166b75ce355e352f4da68feefb2cb38a79 Mon Sep 17 00:00:00 2001 From: "Pierre-Loup A. Griffais" Date: Sun, 30 Aug 2026 00:03:35 -0700 Subject: [PATCH] DiskCache: get rid of more extra copies/allocs on Lookup Reorganize disk format a bit so that entrypoints and guest pages can be used as is from the original blob allocation, with some in-place relocation. --- FEXCore/Source/Interface/Core/CPUBackend.h | 6 +-- FEXCore/Source/Interface/Core/Core.cpp | 29 ++++++----- FEXCore/Source/Interface/Core/DiskCache.cpp | 55 +++++++++++--------- FEXCore/Source/Interface/Core/JIT/JIT.cpp | 5 +- FEXCore/Source/Interface/Core/JIT/JITClass.h | 2 +- FEXCore/Source/Interface/Core/LookupCache.h | 11 ++-- FEXCore/include/FEXCore/Core/DiskCache.h | 12 ++--- 7 files changed, 61 insertions(+), 59 deletions(-) diff --git a/FEXCore/Source/Interface/Core/CPUBackend.h b/FEXCore/Source/Interface/Core/CPUBackend.h index 19d82eac0..0183694cd 100644 --- a/FEXCore/Source/Interface/Core/CPUBackend.h +++ b/FEXCore/Source/Interface/Core/CPUBackend.h @@ -20,10 +20,6 @@ $end_info$ #include #include -namespace FEXCore::DiskCache { -struct BlobEntryPoint; -} - namespace FEXCore::CPU { union Relocation; } @@ -122,7 +118,7 @@ namespace CPU { virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0; - virtual CompiledCode LoadCachedCode(std::span HostBytes, std::span EntryPoints) { + virtual CompiledCode LoadCachedCode(std::span HostBytes) { return {}; } diff --git a/FEXCore/Source/Interface/Core/Core.cpp b/FEXCore/Source/Interface/Core/Core.cpp index ec8b4b770..95cbdd843 100644 --- a/FEXCore/Source/Interface/Core/Core.cpp +++ b/FEXCore/Source/Interface/Core/Core.cpp @@ -868,26 +868,29 @@ uintptr_t ContextImpl::CompileBlock(FEXCore::Core::CpuStateFrame* Frame, uint64_ CodeCache.ApplyPackedCodeRelocations(GuestRIP, std::as_writable_bytes(Hit->HostCode), Hit->SmallRelocs, Hit->ThunkRelocs, false); if (DiskCacheHitRelocationsApplied && LoadDiskCacheCode) { - auto LoadedCode = Thread->CPUBackend->LoadCachedCode(Hit->HostCode, Hit->EntryPoints); + auto LoadedCode = Thread->CPUBackend->LoadCachedCode(Hit->HostCode); if (LoadedCode.BlockBegin) { - - // annoying to unpack a different copy here, maybe better way to do this - fextl::set EntryPoints; - for (auto [GuestOffset, HostAddr] : LoadedCode.EntryPoints) { - EntryPoints.insert(GuestOffset + Region->FileStartVA); - } - for (auto CodePage : Hit->GuestPages) { - if (Thread->LookupCache->AddBlockExecutableRange(Thread, EntryPoints, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) { + for (auto& CodePage : Hit->GuestPages) { + if (Thread->LookupCache->AddBlockExecutableRange(Thread, Hit->EntryPointRIPs, CodePage, FEXCore::Utils::FEX_PAGE_SIZE)) { SyscallHandler->MarkGuestExecutableRange(Thread, CodePage, FEXCore::Utils::FEX_PAGE_SIZE); } } - for (auto [GuestOffset, HostAddr] : LoadedCode.EntryPoints) { - Thread->LookupCache->AddBlockMapping(Thread, GuestOffset + Region->FileStartVA, Hit->GuestPages, HostAddr); + + LOGMAN_THROW_A_FMT(Hit->EntryPointRIPs.size() == Hit->EntryPointHostOffsets.size(), "Mismatched Disk Cache entrypoint pairs!"); + + uintptr_t CachedHostCode = 0; + for (size_t i = 0; i < Hit->EntryPointRIPs.size(); i++) { + void* HostAddr = LoadedCode.BlockBegin + Hit->EntryPointHostOffsets[i]; + Thread->LookupCache->AddBlockMapping(Thread, Hit->EntryPointRIPs[i], Hit->GuestPages, HostAddr); + if (Hit->EntryPointRIPs[i] == GuestRIP) { + CachedHostCode = reinterpret_cast(HostAddr); + } } - uint64_t ModuleOffset = GuestRIP - Region->FileStartVA; + LOGMAN_THROW_A_FMT(CachedHostCode != 0, "Couldn't find GuestRIP in Disk Cache entrypoints!"); + FEXCORE_PROFILE_INSTANT_INCREMENT(Thread, AccumulatedDiskCacheHitCount, 1); - return reinterpret_cast(LoadedCode.EntryPoints[ModuleOffset]); + return CachedHostCode; } } } diff --git a/FEXCore/Source/Interface/Core/DiskCache.cpp b/FEXCore/Source/Interface/Core/DiskCache.cpp index e2b13c291..7e2c1a7a1 100644 --- a/FEXCore/Source/Interface/Core/DiskCache.cpp +++ b/FEXCore/Source/Interface/Core/DiskCache.cpp @@ -331,7 +331,7 @@ namespace DiskCache { fextl::string SerializedConfig = FEXCore::Config::SerializeForCache(); struct __attribute__((packed)) { - uint8_t FormatVersion; + uint16_t FormatVersion; uint8_t Is64BitMode; uint64_t HostFeaturesHash; } BucketHeader = {FormatVersion, CTX->Config.Is64BitMode, CTX->HostFeatures.HashForCaching()}; @@ -444,9 +444,9 @@ namespace DiskCache { // LogMan::Msg::IFmt("hash ok! length {:d}", Header.GuestSize); // this seems to be a full hit, lastly, check the entry is big enough to have everything (except maybe GuestCode) - uint32_t SizeNeeded = sizeof(Header) + Header.HostSize + Header.EntryPointCount * sizeof(BlobEntryPoint); + uint32_t SizeNeeded = sizeof(Header) + Header.HostSize + Header.EntryPointCount * (sizeof(uint64_t) + sizeof(uint32_t)); SizeNeeded += Header.SmallRelocCount * sizeof(BlobSmallRelocation) + Header.ThunkRelocCount * sizeof(BlobThunkRelocation) + - Header.TouchedGuestPagesCount * sizeof(int64_t); + Header.TouchedGuestPagesCount * sizeof(uint64_t); if (Entry.Size < SizeNeeded) { return std::nullopt; } @@ -455,17 +455,22 @@ namespace DiskCache { HitData.HostCode = {HitData.Blob.data() + BlobOffset, Header.HostSize}; BlobOffset += Header.HostSize; - HitData.EntryPoints = {reinterpret_cast(HitData.Blob.data() + BlobOffset), Header.EntryPointCount}; - BlobOffset += Header.EntryPointCount * sizeof(BlobEntryPoint); + HitData.GuestPages = {reinterpret_cast(HitData.Blob.data() + BlobOffset), Header.TouchedGuestPagesCount}; + BlobOffset += Header.TouchedGuestPagesCount * sizeof(uint64_t); + HitData.EntryPointRIPs = {reinterpret_cast(HitData.Blob.data() + BlobOffset), Header.EntryPointCount}; + BlobOffset += Header.EntryPointCount * sizeof(uint64_t); + HitData.EntryPointHostOffsets = {reinterpret_cast(HitData.Blob.data() + BlobOffset), Header.EntryPointCount}; + BlobOffset += Header.EntryPointCount * sizeof(uint32_t); HitData.SmallRelocs = {reinterpret_cast(HitData.Blob.data() + BlobOffset), Header.SmallRelocCount}; BlobOffset += Header.SmallRelocCount * sizeof(BlobSmallRelocation); HitData.ThunkRelocs = {reinterpret_cast(HitData.Blob.data() + BlobOffset), Header.ThunkRelocCount}; BlobOffset += Header.ThunkRelocCount * sizeof(BlobThunkRelocation); - auto* PageOffsets = reinterpret_cast(HitData.Blob.data() + BlobOffset); - HitData.GuestPages.reserve(Header.TouchedGuestPagesCount); - for (uint32_t i = 0; i < Header.TouchedGuestPagesCount; ++i) { - HitData.GuestPages.push_back(GuestRIP + PageOffsets[i]); + for (auto& PageOffset : HitData.GuestPages) { + PageOffset += GuestRIP; + } + for (auto& EntryPointRip : HitData.EntryPointRIPs) { + EntryPointRip += GuestRIP; } return HitData; @@ -518,7 +523,6 @@ namespace DiskCache { uint32_t SmallRelocCount = 0; uint32_t ThunkRelocCount = 0; for (const auto& Reloc : Relocations) { - // relocs aren't cleared every time if IsGeneratingCache, so filter just in case if (Reloc.Header.Type == CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE) { ThunkRelocCount++; } else { @@ -531,11 +535,12 @@ namespace DiskCache { const size_t HeaderOffset = 0; const size_t HostCodeOffset = HeaderOffset + sizeof(BlobFixedHeader); - const size_t EntryPointsOffset = HostCodeOffset + CompiledCode.Size; - const size_t SmallRelocsOffset = EntryPointsOffset + EntryPointCount * sizeof(BlobEntryPoint); + const size_t TouchedGuestPagesOffset = HostCodeOffset + CompiledCode.Size; + const size_t EntryPointRIPsOffset = TouchedGuestPagesOffset + TouchedGuestPagesCount * sizeof(uint64_t); + const size_t EntryPointHostOffsetsOffset = EntryPointRIPsOffset + EntryPointCount * sizeof(uint64_t); + const size_t SmallRelocsOffset = EntryPointHostOffsetsOffset + EntryPointCount * sizeof(uint32_t); const size_t ThunkRelocsOffset = SmallRelocsOffset + SmallRelocCount * sizeof(BlobSmallRelocation); - const size_t TouchedGuestPagesOffset = ThunkRelocsOffset + ThunkRelocCount * sizeof(BlobThunkRelocation); - const size_t GuestCodeOffset = TouchedGuestPagesOffset + TouchedGuestPagesCount * sizeof(int64_t); + const size_t GuestCodeOffset = ThunkRelocsOffset + ThunkRelocCount * sizeof(BlobThunkRelocation); const size_t TotalSize = GuestCodeOffset + GuestCode.size(); // we'll copy everything into here and pass it to the Writer, then return to caller quickly @@ -562,11 +567,21 @@ namespace DiskCache { memcpy(BlobData + HeaderOffset, &Header, sizeof(Header)); memcpy(BlobData + HostCodeOffset, CompiledCode.BlockBegin, CompiledCode.Size); + // relocate touched pages relative to GuestRIP + auto* PageOffsets = reinterpret_cast(BlobData + TouchedGuestPagesOffset); + uint32_t PageIdx = 0; + for (auto GuestPage : DecodedBlockInfo->CodePages) { + PageOffsets[PageIdx++] = GuestPage - GuestRIP; + } + // pack and relocate entrypoints - auto* EntryPoints = reinterpret_cast(BlobData + EntryPointsOffset); + auto* EntryRIPs = reinterpret_cast(BlobData + EntryPointRIPsOffset); + auto* EntryHostOffsets = reinterpret_cast(BlobData + EntryPointHostOffsetsOffset); uint32_t EntryIdx = 0; for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) { - EntryPoints[EntryIdx++] = {GuestAddr - Region.FileStartVA, uint32_t(HostAddr - CompiledCode.BlockBegin)}; + EntryRIPs[EntryIdx] = GuestAddr - GuestRIP; + EntryHostOffsets[EntryIdx] = uint32_t(HostAddr - CompiledCode.BlockBegin); + EntryIdx++; } // pack relocations @@ -614,14 +629,6 @@ namespace DiskCache { } } - // relocate touched pages relative to GuestRIP - // in theory we could save some size here, unlikely we need all 64bits - auto* PageOffsets = reinterpret_cast(BlobData + TouchedGuestPagesOffset); - uint32_t PageIdx = 0; - for (auto GuestPage : DecodedBlockInfo->CodePages) { - PageOffsets[PageIdx++] = GuestPage - GuestRIP; - } - memcpy(BlobData + GuestCodeOffset, GuestCode.data(), GuestCode.size()); // hand the rest off to the writer thread diff --git a/FEXCore/Source/Interface/Core/JIT/JIT.cpp b/FEXCore/Source/Interface/Core/JIT/JIT.cpp index 1b3d90981..7330e40a6 100644 --- a/FEXCore/Source/Interface/Core/JIT/JIT.cpp +++ b/FEXCore/Source/Interface/Core/JIT/JIT.cpp @@ -1157,7 +1157,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size return std::move(CodeData); } -CPUBackend::CompiledCode Arm64JITCore::LoadCachedCode(std::span HostBytes, std::span EntryPoints) { +CPUBackend::CompiledCode Arm64JITCore::LoadCachedCode(std::span HostBytes) { // we stored it aligned, better still be? LOGMAN_THROW_A_FMT(HostBytes.size() % 16 == 0, "Needs to be 16B aligned!"); auto AllocatedInfo = AllocateCodeBufferInSharedCache(HostBytes.size()); @@ -1170,9 +1170,6 @@ CPUBackend::CompiledCode Arm64JITCore::LoadCachedCode(std::span H Result.BlockBegin = Dest; Result.Size = HostBytes.size(); Result.HostCodeOffset = Dest - CurrentCodeBuffer->GetBufferBase(); - for (const auto& Ep : EntryPoints) { - Result.EntryPoints[Ep.GuestRIP] = Dest + Ep.HostOffset; - } return Result; } diff --git a/FEXCore/Source/Interface/Core/JIT/JITClass.h b/FEXCore/Source/Interface/Core/JIT/JITClass.h index e42e6cb3b..9d91f6503 100644 --- a/FEXCore/Source/Interface/Core/JIT/JITClass.h +++ b/FEXCore/Source/Interface/Core/JIT/JITClass.h @@ -55,7 +55,7 @@ public: FEXCore::Core::DebugData* DebugData, bool CheckTF) override; [[nodiscard]] - CPUBackend::CompiledCode LoadCachedCode(std::span HostBytes, std::span EntryPoints) override; + CPUBackend::CompiledCode LoadCachedCode(std::span HostBytes) override; void ClearCache() override; diff --git a/FEXCore/Source/Interface/Core/LookupCache.h b/FEXCore/Source/Interface/Core/LookupCache.h index 02a9aff39..391d48560 100644 --- a/FEXCore/Source/Interface/Core/LookupCache.h +++ b/FEXCore/Source/Interface/Core/LookupCache.h @@ -13,6 +13,7 @@ #include #include +#include #include #include #include @@ -93,13 +94,15 @@ struct GuestToHostMap { GuestToHostMap(); // Adds to Guest -> Host code mapping - const BlockEntry& AddBlockMapping(uint64_t Address, const fextl::vector& CodePages, void* HostCode, const LookupCacheWriteLockToken&) { + const BlockEntry& AddBlockMapping(uint64_t Address, std::span CodePages, void* HostCode, const LookupCacheWriteLockToken&) { // This may replace an existing mapping // NOTE: Generally no previous entry should exist, however there is one exception: // If the backend updates the active thread's CodeBuffer, the new associated LookupCache // may already contain the block address. Since is comparatively rare, we'll just leak // one of the two blocks in this case. - return BlockList.insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, CodePages}).first->second; + return BlockList + .insert_or_assign(Address, BlockEntry {(uintptr_t)HostCode, fextl::vector(CodePages.begin(), CodePages.end())}) + .first->second; } const BlockEntry* FindBlock(uint64_t Address, const LookupCacheReadLockToken&) { @@ -281,7 +284,7 @@ public: // Appends a list of Block {Address} to CodePages [Start, Start + Length) // Returns true if new pages are marked as containing code - bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, const fextl::set& Addresses, uint64_t Start, uint64_t Length) { + bool AddBlockExecutableRange(FEXCore::Core::InternalThreadState* Thread, auto& Addresses, uint64_t Start, uint64_t Length) { std::optional> LockTime( Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr); auto lk = Shared->AcquireWriteLock(); @@ -291,7 +294,7 @@ public: } // Adds to Guest -> Host code mapping - void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, const fextl::vector& CodePages, void* HostCode) { + void AddBlockMapping(FEXCore::Core::InternalThreadState* Thread, uint64_t Address, std::span CodePages, void* HostCode) { std::optional> LockTime( Thread->ThreadStats ? &Thread->ThreadStats->AccumulatedCacheWriteLockTime : nullptr); auto lk = Shared->AcquireWriteLock(); diff --git a/FEXCore/include/FEXCore/Core/DiskCache.h b/FEXCore/include/FEXCore/Core/DiskCache.h index 7eebbf23a..08468fc2a 100644 --- a/FEXCore/include/FEXCore/Core/DiskCache.h +++ b/FEXCore/include/FEXCore/Core/DiskCache.h @@ -62,11 +62,6 @@ namespace DiskCache { XXH128_hash_t GuestHash; }; - struct __attribute__((packed)) BlobEntryPoint { - uint64_t GuestRIP; // todo those have been made relative since i wrote this, can we get away with less size here? - uint32_t HostOffset; - }; - // packed struct for types 0, 2 and 3. type 1 is bigger and separate below struct __attribute__((packed)) BlobSmallRelocation { uint32_t Offset; @@ -95,10 +90,11 @@ namespace DiskCache { struct CodeHitData { fextl::vector Blob; std::span HostCode; - std::span EntryPoints; + std::span GuestPages; + std::span EntryPointRIPs; + std::span EntryPointHostOffsets; std::span SmallRelocs; std::span ThunkRelocs; - fextl::vector GuestPages; // the spans above point to memory owned by the Blob vec, so it's important this can't be copied CodeHitData() = default; @@ -181,7 +177,7 @@ namespace DiskCache { uint64_t MakeBlobKey(const uint64_t ModuleOffset); FEXCore::Context::ContextImpl* CTX; - static const uint8_t FormatVersion = 1; + static const uint16_t FormatVersion = 3; XXH128_hash_t BucketHash; fextl::vector> ROCacheDBs; fextl::unique_ptr RWCacheDB;