Files
FEX-Emu--FEX/FEXCore/Source/Interface/Core/DiskCache.cpp
T
Pierre-Loup A. Griffais 049c932675 DiskCache: support multiple entries per lookup key
The LookupKey based on data available at Lookup time gives a list of possible
candidate entries, which may have different guest code sizes/footprints.

If only one candidate, it's stored in-line in the map like before - if we grow
past that, an additional (multi)map is allocated, sorted by footprint.

The footprint sorting lets us reduce the amount of hashes performed at Lookup
to the strict minimum.

Anon entries keys are based on a limited size decoded prefix, so this is an
important part of geting the best hit rate possible out of all the other work
with anon keys, as there's often multiple candidates with the same prefix.

Cap the max amount of entries per bucket to limit growth.
2026-09-07 22:50:22 -07:00

1053 lines
42 KiB
C++

// SPDX-License-Identifier: MIT
#define XXH_STATIC_LINKING_ONLY
#include "FEXCore/Utils/TypeDefines.h"
#include "Interface/Core/Frontend.h"
#include "FEXCore/Config/Config.h"
#include "FEXCore/fextl/string.h"
#include "FEXHeaderUtils/Filesystem.h"
#include "FEXCore/Core/DiskCache.h"
#include "FEXCore/Core/DiskCacheFileMapper.h"
#include "FEXCore/Utils/LogManager.h"
#include "Interface/Context/Context.h"
#include "FEXCore/HLE/SyscallHandler.h"
#include "FEXCore/Utils/File.h"
#include "FEXCore/fextl/memory.h"
#include <cstdint>
#include <cstring>
#include <atomic>
namespace FEXCore {
namespace DiskCache {
namespace MesaFOZ {
enum { FOSSILIZE_COMPRESSION_NONE = 1, FOSSILIZE_COMPRESSION_DEFLATE = 2 };
enum { FOSSILIZE_FORMAT_VERSION = 6, FOSSILIZE_FORMAT_MIN_COMPAT_VERSION = 5 };
#define FOZ_REF_MAGIC_SIZE 16
static const uint8_t stream_reference_magic_and_version[FOZ_REF_MAGIC_SIZE] = {
0x81, 'F', 'O', 'S', 'S', 'I', 'L', 'I', 'Z', 'E', 'D', 'B', 0, 0, 0, FOSSILIZE_FORMAT_VERSION, /* 4 bytes to use for versioning. */
};
struct __attribute__((packed)) mesa_index_db_file_entry {
uint64_t hash;
uint32_t size;
uint64_t last_access_time;
uint64_t cache_db_file_offset;
};
} // namespace MesaFOZ
struct __attribute__((packed)) IndexExtraBlobHeader {
XXH128_hash_t GuestHash; // StoreCacheBlob will assume it's here, care if moving
uint64_t GuestFootprint;
uint32_t GuestSize;
uint32_t GuestExtentsCount;
};
static FileMapperFunc FileMapper = nullptr;
bool FOZFile::Open(const fextl::string& FOZFileName, bool ReadOnly) {
FileName = FOZFileName;
this->ReadOnly = ReadOnly;
File::FileModes Modes = File::FileModes::READ;
if (!ReadOnly) {
Modes = Modes | File::FileModes::WRITE | File::FileModes::CREATE;
}
FD = fextl::make_unique<File::File>(FileName.c_str(), Modes, false);
if (!FD->IsValid()) {
FD.reset();
return false;
}
bool Valid = false;
bool TookLock = false;
ssize_t Size = FD->Size();
if (Size < FOZ_REF_MAGIC_SIZE && !ReadOnly) {
if (!FD->Lock(OPEN_LOCK_TIMEOUT_MS)) {
FD.reset();
return false;
}
TookLock = true;
// check size again in case someone else made it while we waited
Size = FD->Size();
}
if (Size == 0 && !ReadOnly) {
Valid = FD->PWrite(MesaFOZ::stream_reference_magic_and_version, FOZ_REF_MAGIC_SIZE, 0) == FOZ_REF_MAGIC_SIZE;
} else {
uint8_t magic[FOZ_REF_MAGIC_SIZE];
if (FD->PRead(magic, FOZ_REF_MAGIC_SIZE, 0) == FOZ_REF_MAGIC_SIZE &&
memcmp(magic, MesaFOZ::stream_reference_magic_and_version, FOZ_REF_MAGIC_SIZE - 1) == 0) {
int version = magic[FOZ_REF_MAGIC_SIZE - 1];
Valid = version <= MesaFOZ::FOSSILIZE_FORMAT_VERSION && version >= MesaFOZ::FOSSILIZE_FORMAT_MIN_COMPAT_VERSION;
}
}
if (TookLock) {
FD->Unlock();
}
if (!Valid) {
FD.reset();
}
return Valid;
}
ssize_t FOZFile::Size() {
return FD ? FD->Size() : -1;
}
bool FOZFile::ReadAll(fextl::vector<uint8_t>& Out) {
ssize_t FileSize = Size();
if (FileSize < FOZ_REF_MAGIC_SIZE) {
return false;
}
Out.resize((size_t)FileSize - FOZ_REF_MAGIC_SIZE);
return FD->PRead(Out.data(), Out.size(), FOZ_REF_MAGIC_SIZE) == (ssize_t)Out.size();
}
bool FOZFile::ReadBlob(uint64_t Offset, std::span<uint8_t> OutBlob) {
if (FD->PRead(OutBlob.data(), OutBlob.size(), Offset) != (ssize_t)OutBlob.size()) {
return false;
}
return true;
}
bool FOZFile::WriteBlob(const MesaFOZ::foz_payload_key& Key, std::span<const std::span<const uint8_t>> BlobChunks, uint64_t& OutBlobOffset) {
ssize_t FileSize = FD->Size();
if (FileSize < 0) {
return false;
}
uint64_t WriteOffset = (uint64_t)FileSize;
if (FD->PWrite(Key.bytes, sizeof(Key.bytes), WriteOffset) != sizeof(Key.bytes)) {
return false;
}
WriteOffset += sizeof(Key.bytes);
uint64_t TotalBlobSize = 0;
for (const std::span<const uint8_t>& Chunk : BlobChunks) {
TotalBlobSize += Chunk.size();
}
MesaFOZ::foz_payload_header ScratchHeader {.payload_size = (uint32_t)TotalBlobSize,
.format = MesaFOZ::FOSSILIZE_COMPRESSION_NONE,
.crc = 0, // todo? maybe
.uncompressed_size = (uint32_t)TotalBlobSize};
if (FD->PWrite(&ScratchHeader, sizeof(ScratchHeader), WriteOffset) != sizeof(ScratchHeader)) {
return false;
}
WriteOffset += sizeof(ScratchHeader);
OutBlobOffset = WriteOffset;
for (const std::span<const uint8_t>& Chunk : BlobChunks) {
if (Chunk.size() == 0) {
continue;
}
if (FD->PWrite(Chunk.data(), Chunk.size(), WriteOffset) != (ssize_t)Chunk.size()) {
return false;
}
WriteOffset += Chunk.size();
}
return true;
}
bool IndexedDB::Open(const fextl::string& CacheDBName, bool ReadOnly) {
if (!CacheFOZ.Open(CacheDBName + ".foz", ReadOnly)) {
return false;
}
if (!IndexFOZ.Open(CacheDBName + "_idx.foz", ReadOnly)) {
return false;
}
File::File::FileHandleType CacheFileHandle = CacheFOZ.GetHandle();
if (FileMapper && CacheFileHandle != (File::File::FileHandleType)-1) {
CacheFileMapping = reinterpret_cast<uint8_t*>(FileMapper(CacheFileHandle, ReadOnly ? CacheFOZ.Size() : BIG_MAPPING_SIZE));
CacheFileSize = CacheFOZ.Size();
}
this->ReadOnly = ReadOnly;
return true;
}
void IndexedDB::PopulateIndex(Index& CacheIndex, bool& FoundMetadata) {
fextl::vector<uint8_t> Data;
if (!IndexFOZ.ReadAll(Data)) {
return;
}
ssize_t CacheFOZSize = CacheFOZ.Size();
if (CacheFOZSize < 0) {
return;
}
const uint8_t* IndexDataStart = Data.data();
const size_t IndexDataSize = Data.size();
size_t ReadOffset = 0;
while (ReadOffset + sizeof(MesaFOZ::foz_payload_key) + sizeof(MesaFOZ::foz_payload_header) <= IndexDataSize) {
const auto* FOZKey = reinterpret_cast<const MesaFOZ::foz_payload_key*>(IndexDataStart + ReadOffset);
ReadOffset += sizeof(MesaFOZ::foz_payload_key);
const auto* FOZHeader = reinterpret_cast<const MesaFOZ::foz_payload_header*>(IndexDataStart + ReadOffset);
ReadOffset += sizeof(MesaFOZ::foz_payload_header);
uint64_t IndexBlobSize = sizeof(MesaFOZ::mesa_index_db_file_entry) + sizeof(IndexExtraBlobHeader);
if (FOZHeader->payload_size < IndexBlobSize || ReadOffset + FOZHeader->payload_size > IndexDataSize) {
break;
}
const auto* IndexBlobCommon = reinterpret_cast<const MesaFOZ::mesa_index_db_file_entry*>(IndexDataStart + ReadOffset);
ReadOffset += sizeof(MesaFOZ::mesa_index_db_file_entry);
const auto* IndexBlobExtra = reinterpret_cast<const IndexExtraBlobHeader*>(IndexDataStart + ReadOffset);
ReadOffset += sizeof(IndexExtraBlobHeader);
// skip corrupt (carefully) so we don't have to figure that out in the hot path later
if (IndexBlobCommon->cache_db_file_offset > (uint64_t)CacheFOZSize ||
IndexBlobCommon->size > (uint64_t)CacheFOZSize - IndexBlobCommon->cache_db_file_offset) {
continue;
}
if (IndexBlobExtra->GuestSize + sizeof(BlobFixedHeader) > IndexBlobCommon->size) {
continue;
}
IndexBlobSize += IndexBlobExtra->GuestExtentsCount * sizeof(uint32_t);
if (FOZHeader->payload_size != IndexBlobSize) {
break;
}
const auto* GuestExtents = reinterpret_cast<const uint32_t*>(IndexDataStart + ReadOffset);
ReadOffset += IndexBlobExtra->GuestExtentsCount * sizeof(uint32_t);
if (FOZKey->bytes[39] != 0xFF) {
IndexEntry NewEntry {this, IndexBlobCommon->cache_db_file_offset, IndexBlobCommon->size, IndexBlobExtra->GuestSize,
IndexBlobExtra->GuestHash};
NewEntry.GuestExtents.resize(IndexBlobExtra->GuestExtentsCount);
memcpy(NewEntry.GuestExtents.data(), GuestExtents, IndexBlobExtra->GuestExtentsCount * sizeof(uint32_t));
bool ExtentsValid = true;
for (uint32_t i = 0; i < NewEntry.GuestExtents.size(); i += 2) {
if ((uint64_t)NewEntry.GuestExtents[i] + NewEntry.GuestExtents[i + 1] > IndexBlobExtra->GuestSize) {
ExtentsValid = false;
break;
}
}
if (!ExtentsValid) {
continue;
}
auto It = CacheIndex.find(IndexBlobCommon->hash);
if (It == CacheIndex.end()) {
CacheIndex.emplace(IndexBlobCommon->hash, IndexCacheHead {std::move(NewEntry), IndexBlobExtra->GuestFootprint, nullptr});
} else {
if (!It->second.MoreEntries.get()) {
It->second.MoreEntries = fextl::make_unique<fextl::multimap<uint64_t, IndexEntry>>();
}
It->second.MoreEntries->insert({IndexBlobExtra->GuestFootprint, std::move(NewEntry)});
}
} else {
FoundMetadata = true;
}
}
// could truncate/delete index if we don't end up perfectly at end here
}
bool IndexedDB::ReadCacheBlob(uint64_t Offset, std::span<uint8_t> OutBlob) {
if (CacheFileMapping && (ReadOnly || Offset + OutBlob.size() <= BIG_MAPPING_SIZE)) {
if (Offset + OutBlob.size() > CacheFileSize) {
return false;
}
// todo could reduce copies by having a private mapping for relocs, etc
memcpy(OutBlob.data(), CacheFileMapping + Offset, OutBlob.size());
return true;
} else {
return CacheFOZ.ReadBlob(Offset, OutBlob);
}
}
bool IndexedDB::StoreCacheBlob(const MesaFOZ::foz_payload_key& UniqueKey, uint64_t LookupKey, std::span<const uint8_t> Blob, Index& Index,
std::mutex& IndexMutex, std::span<const uint8_t> IndexBlob) {
if (ReadOnly) {
// shouldn't happen
return false;
}
{
std::lock_guard Guard(IndexMutex);
auto IndexIt = Index.find(LookupKey);
bool Dupe = false;
if (IndexIt != Index.end()) {
if (IndexIt->second.MoreEntries.get()) {
for (auto& [Key, Elem] : *IndexIt->second.MoreEntries) {
if (XXH128_isEqual(Elem.GuestHash, *(const XXH128_hash_t*)(IndexBlob.data()))) {
Dupe = true;
break;
}
}
}
if (!Dupe && XXH128_isEqual(IndexIt->second.MainEntry.GuestHash, *(const XXH128_hash_t*)(IndexBlob.data()))) {
Dupe = true;
}
// could happen if it's seen again while in flight in the store queue
if (Dupe) {
return true;
}
if (IndexIt->second.MoreEntries.get() && IndexIt->second.MoreEntries->size() >= LOOKUP_KEY_MAX_BUCKET_DEPTH) {
return true;
}
}
}
if (!CacheFOZ.Lock(STORE_LOCK_TIMEOUT_MS) || !IndexFOZ.Lock(STORE_LOCK_TIMEOUT_MS)) {
CacheFOZ.Unlock();
IndexFOZ.Unlock();
return false;
}
// write cache side first so we get offset for index
std::span<const uint8_t> BlobChunks[] = {Blob};
uint64_t BlobOffset = 0;
if (!CacheFOZ.WriteBlob(UniqueKey, BlobChunks, BlobOffset)) {
CacheFOZ.Unlock();
IndexFOZ.Unlock();
return false;
}
MesaFOZ::mesa_index_db_file_entry IndexEntry {.hash = LookupKey,
.size = (uint32_t)Blob.size(),
.last_access_time = 0, // todo..
.cache_db_file_offset = BlobOffset};
std::span<const uint8_t> IndexBlobChunks[] = {
{(const uint8_t*)&IndexEntry, sizeof(IndexEntry)},
IndexBlob,
};
uint64_t UnusedIndexBlobOffset = 0;
if (!IndexFOZ.WriteBlob(UniqueKey, IndexBlobChunks, UnusedIndexBlobOffset)) {
CacheFOZ.Unlock();
IndexFOZ.Unlock();
return false;
}
CacheFOZ.Unlock();
IndexFOZ.Unlock();
// publish new file size for memory-mapped reads
if (CacheFileMapping) {
CacheFileSize = BlobOffset + Blob.size();
}
const IndexExtraBlobHeader* IndexBlobHeader = reinterpret_cast<const IndexExtraBlobHeader*>(IndexBlob.data());
struct IndexEntry NewEntry {this, BlobOffset, (uint32_t)Blob.size(), IndexBlobHeader->GuestSize, IndexBlobHeader->GuestHash};
NewEntry.GuestExtents.resize(IndexBlobHeader->GuestExtentsCount);
memcpy(NewEntry.GuestExtents.data(), reinterpret_cast<const uint32_t*>(IndexBlob.data() + sizeof(IndexExtraBlobHeader)),
IndexBlobHeader->GuestExtentsCount * sizeof(uint32_t));
std::lock_guard Guard(IndexMutex);
auto It = Index.find(LookupKey);
if (It == Index.end()) {
Index.emplace(LookupKey, IndexCacheHead {std::move(NewEntry), IndexBlobHeader->GuestFootprint, nullptr});
} else {
if (!It->second.MoreEntries.get()) {
It->second.MoreEntries = fextl::make_unique<fextl::multimap<uint64_t, struct IndexEntry>>();
}
It->second.MoreEntries->insert({IndexBlobHeader->GuestFootprint, std::move(NewEntry)});
}
return true;
}
bool DiskCache::OpenCacheDB(const fextl::string& CacheDBName, bool ReadOnly) {
fextl::unique_ptr<IndexedDB> CurDB;
if (!ReadOnly && RWCacheDB) {
// rw already opened, just support one
return false;
}
CurDB = fextl::make_unique<IndexedDB>();
if (!CurDB) {
return false;
}
if (!CurDB->Open(CacheDBName, ReadOnly)) {
CurDB.reset();
return false;
}
CurDB->PopulateIndex(Index, FoundMetadata);
if (ReadOnly) {
ROCacheDBs.push_back(std::move(CurDB));
} else {
RWCacheDB = std::move(CurDB);
}
return true;
}
FEX_DEFAULT_VISIBILITY void SetFileMapper(FileMapperFunc Func) {
FileMapper = Func;
}
void DiskCache::Init(FEXCore::Context::ContextImpl* CTX) {
this->CTX = CTX;
if (!EnableDiskCache) {
return;
}
fextl::string SerializedConfig = FEXCore::Config::SerializeForCache();
struct __attribute__((packed)) {
uint16_t FormatVersion;
uint8_t Is64BitMode;
uint64_t HostFeaturesHash;
} BucketHeader = {FormatVersion, CTX->Config.Is64BitMode, CTX->HostFeatures.HashForCaching()};
fextl::vector<uint8_t> BucketBytes;
BucketBytes.resize(sizeof(BucketHeader) + SerializedConfig.size());
memcpy(BucketBytes.data(), &BucketHeader, sizeof(BucketHeader));
memcpy(BucketBytes.data() + sizeof(BucketHeader), SerializedConfig.data(), SerializedConfig.size());
BucketHash = XXH3_128bits(BucketBytes.data(), BucketBytes.size());
fextl::string BasePath = BasePathOverride();
if (BasePath.empty()) {
BasePath = FEXCore::Config::GetCacheDirectory() + "DiskCache/";
BasePath += fextl::fmt::format("{:016x}{:016x}", BucketHash.high64, BucketHash.low64) + "/";
}
FHU::Filesystem::CreateDirectories(BasePath);
if (!MapDiskCacheFiles) {
FileMapper = nullptr;
}
fextl::string RWDBBasePath = BasePath + "RWCacheDB";
OpenCacheDB(RWDBBasePath, false);
if (RWCacheDB && !FoundMetadata) {
// we just opened a fresh cache, add a metadata blob
MesaFOZ::foz_payload_key MetadataKey;
memset(MetadataKey.bytes, 0xFF, sizeof(MetadataKey));
IndexExtraBlobHeader MetaDataHeader = {};
RWCacheDB->StoreCacheBlob(MetadataKey, ~0, {BucketBytes.data(), BucketBytes.size()}, Index, IndexLock,
{reinterpret_cast<uint8_t*>(&MetaDataHeader), sizeof(MetaDataHeader)});
Index.erase(~0);
}
std::string_view RONames = RODBNames();
while (!RONames.empty()) {
const auto Delim = RONames.find(',');
const std::string_view ROName = RONames.substr(0, Delim);
if (!ROName.empty()) {
fextl::string RODBBasePath = BasePath;
RODBBasePath += ROName;
OpenCacheDB(RODBBasePath, true);
}
if (Delim == std::string_view::npos) {
break;
}
// advance to next
RONames.remove_prefix(Delim + 1);
}
WritingDiskCache = (bool)RWCacheDB;
ReadingDiskCache = !ROCacheDBs.empty() || RWCacheDB != nullptr;
if (IsWritingDiskCache()) {
FEXCore::Threads::Flags WriterThreadFlags = {.LowPriority = true, .Internal = true};
Writer = fextl::make_unique<WorkQueueThread>(WriterThreadFlags);
}
}
uint64_t DiskCache::MakeLookupKey(Core::InternalThreadState* Thread, const uint64_t CodeKey, bool Writable, bool MonoBackpatcher) {
struct __attribute__((packed)) {
uint64_t CodeKey;
XXH128_hash_t BucketHash;
uint8_t Flags;
} BlobKeyBytes = {CodeKey, BucketHash, 0};
if (Writable) {
BlobKeyBytes.Flags |= 1 << 0;
}
if (CTX->AreMonoHacksActive()) {
BlobKeyBytes.Flags |= 1 << 1;
}
if (Thread->CurrentFrame->State.flags[X86State::RFLAG_TF_RAW_LOC]) {
BlobKeyBytes.Flags |= 1 << 2;
}
if (MonoBackpatcher) {
BlobKeyBytes.Flags |= 1 << 3;
}
return XXH3_64bits(&BlobKeyBytes, sizeof(BlobKeyBytes));
}
std::optional<CodeHitData> DiskCache::Lookup(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region,
uint64_t GuestRIP, std::optional<uint64_t>& GuestCodeKey) {
if (!IsReadingDiskCache()) {
return std::nullopt;
}
if (Region && Region->FileStartVA) {
struct __attribute__((packed)) {
uint64_t GuestOffset;
uint64_t FileId;
} FileBackedKey = {GuestRIP - Region->FileStartVA, Region->FileInfo.FileId};
GuestCodeKey = XXH3_64bits(&FileBackedKey, sizeof(FileBackedKey));
} else {
if (!AnonCaching) {
return std::nullopt;
}
Thread->FrontendDecoder->DecodeLoop(reinterpret_cast<const uint8_t*>(GuestRIP), AnonPrefixGuestBytes);
const auto* BlockInfo = Thread->FrontendDecoder->GetDecodedBlockInfo();
XXH3_state_t HashState;
XXH3_64bits_reset(&HashState);
for (auto& SubBlock : BlockInfo->Blocks) {
if (SubBlock.BlockStatus != Frontend::Decoder::DecodedBlockStatus::SUCCESS) {
return std::nullopt;
}
uint64_t HashStart = SubBlock.Entry;
// skip over masked in-block data and data/etc gaps between blocks
for (auto& DataMask : SubBlock.DataMasks) {
XXH3_64bits_update(&HashState, reinterpret_cast<const uint8_t*>(HashStart), DataMask.FieldAddress - HashStart);
HashStart = DataMask.FieldAddress + DataMask.ValueSize;
}
if (HashStart != SubBlock.Entry + SubBlock.Size) {
XXH3_64bits_update(&HashState, reinterpret_cast<const uint8_t*>(HashStart), SubBlock.Size - (HashStart - SubBlock.Entry));
}
}
GuestCodeKey = XXH3_64bits_digest(&HashState);
// if (TotalSize < AnonPrefixGuestBytes) {
// GuestCodeKey = 0;
// return std::nullopt;
// }
// LogMan::Msg::IFmt("anon lookup! length {:d} {}", GuestCodeKey, TotalSize);
}
auto RangeInfo = CTX->SyscallHandler->QueryGuestExecutableRange(Thread, GuestRIP);
if (RangeInfo.Size == 0 || RangeInfo.Base > GuestRIP) {
return std::nullopt;
}
uint64_t Available = RangeInfo.Base + RangeInfo.Size - GuestRIP;
uint64_t LookupKey =
MakeLookupKey(Thread, *GuestCodeKey, RangeInfo.Writable, GuestRIP == CTX->GetMonoBackPatcherBlock().load(std::memory_order_relaxed));
XXH128_hash_t LiveGuestHash;
uint64_t LastFootprintHashed = 0;
bool FirstHash = true; // 0 is probably a valid footprint hash so we need an extra bit there
bool TriedMainEntry = false;
IndexEntry MainEntry;
uint64_t MainEntryFootprint;
fextl::multimap<uint64_t, IndexEntry>::iterator MoreEntriesIt;
fextl::multimap<uint64_t, IndexEntry>* MapPointer = nullptr;
IndexEntry* EntryUnderReview;
uint64_t CurrentFootprint = 0;
{
std::lock_guard Guard(IndexLock);
auto It = Index.find(LookupKey);
if (It == Index.end()) {
// definite miss
return std::nullopt;
}
MainEntry = It->second.MainEntry;
MainEntryFootprint = It->second.MainEntryFootprint;
MapPointer = It->second.MoreEntries.get();
if (MapPointer) {
MoreEntriesIt = It->second.MoreEntries->begin();
}
if (MapPointer && MoreEntriesIt != It->second.MoreEntries->end()) {
EntryUnderReview = &MoreEntriesIt->second;
CurrentFootprint = MoreEntriesIt->first;
} else {
EntryUnderReview = &MainEntry;
CurrentFootprint = MainEntryFootprint;
TriedMainEntry = true;
}
}
// found a lookup key match, check for guest hash match now
bool FoundMatchingHash = false;
bool Advance = false;
while (!FoundMatchingHash) {
if (Advance) {
std::lock_guard Guard(IndexLock);
EntryUnderReview = nullptr;
if (MapPointer && MoreEntriesIt != MapPointer->end()) {
// if the current footprint is also the main entry's footprint, give main entry a shot next
if (!TriedMainEntry && MoreEntriesIt->first == MainEntryFootprint) {
EntryUnderReview = &MainEntry;
CurrentFootprint = MainEntryFootprint;
TriedMainEntry = true;
} else {
MoreEntriesIt++;
}
}
if (!MapPointer || MoreEntriesIt == MapPointer->end()) {
if (TriedMainEntry) {
// ran out
break;
} else {
EntryUnderReview = &MainEntry;
CurrentFootprint = MainEntryFootprint;
TriedMainEntry = true;
}
}
if (!EntryUnderReview && MapPointer && MoreEntriesIt != MapPointer->end()) {
EntryUnderReview = &MoreEntriesIt->second;
CurrentFootprint = MoreEntriesIt->first;
}
}
Advance = true;
// do we have enough room in our live code to even hash GuestSize worth?
if (Available < EntryUnderReview->GuestSize) {
continue;
}
// if (Entry.GuestExtents.size()) {
// LogMan::Msg::IFmt("lookup! length {:d}", Entry.GuestSize);
// for(uint32_t i = 0; i < Entry.GuestExtents.size(); i+=2 ) {
// LogMan::Msg::IFmt("extent {} {}", Entry.GuestExtents[i], Entry.GuestExtents[i]+Entry.GuestExtents[i+1]);
// }
// }
{
IndexEntry& Entry = *EntryUnderReview;
// only renew the hash if the candidate footprint is different than previous
if (FirstHash || LastFootprintHashed != CurrentFootprint) {
XXH3_state_t HashState;
XXH3_128bits_reset(&HashState);
for (uint32_t i = 0; i < Entry.GuestExtents.size(); i += 2) {
XXH3_128bits_update(&HashState, reinterpret_cast<uint8_t*>(GuestRIP) + Entry.GuestExtents[i], Entry.GuestExtents[i + 1]);
}
LiveGuestHash = XXH3_128bits_digest(&HashState);
LastFootprintHashed = CurrentFootprint;
FirstHash = false;
}
if (XXH128_isEqual(LiveGuestHash, Entry.GuestHash)) {
FoundMatchingHash = true;
break;
} else if (Validation) {
fextl::vector<uint8_t> GuestCode(Entry.GuestSize);
if (Entry.Size >= Entry.GuestSize && Entry.DB->ReadCacheBlob(Entry.Offset + Entry.Size - Entry.GuestSize, GuestCode)) {
const uint8_t* CachedGuest = GuestCode.data();
const uint8_t* LiveGuest = reinterpret_cast<const uint8_t*>(GuestRIP);
uint64_t DiffCount = 0;
uint64_t DiffOffset = 0;
for (uint32_t i = 0; i < Entry.GuestExtents.size(); i += 2) {
uint32_t Begin = Entry.GuestExtents[i];
uint32_t End = Begin + Entry.GuestExtents[i + 1];
bool PreviousByteDiff = false;
for (uint32_t Offset = Begin; Offset < End; Offset++) {
if (LiveGuest[Offset] != CachedGuest[Offset]) {
if (DiffCount == 0) {
DiffOffset = Offset;
}
if (!PreviousByteDiff) {
DiffCount++;
}
PreviousByteDiff = true;
} else {
PreviousByteDiff = false;
}
}
}
if (DiffCount) {
uint64_t DiffStart = DiffOffset >= 8 ? DiffOffset - 8 : 0;
uint64_t DiffContextBytes = std::min<uint64_t>(16, Entry.GuestSize - DiffStart);
auto LiveDump = fmt::join(std::span<const uint8_t> {LiveGuest + DiffStart, DiffContextBytes}, " ");
auto CacheDump = fmt::join(std::span<const uint8_t> {CachedGuest + DiffStart, DiffContextBytes}, " ");
auto KeyPrefix = (Region && Region->FileStartVA) ? "file" : "anon";
LogMan::Msg::IFmt("DiskCache: lookup guest hash mismatch key={}-{:x} gsize={}, ndiff={}, first diff: offset={} "
"live=[{:02x}] "
"cached=[{:02x}]",
KeyPrefix, *GuestCodeKey, Entry.GuestSize, DiffCount, DiffOffset, LiveDump, CacheDump);
} else {
LogMan::Msg::IFmt("DiskCache: guest hash mismatch but no diff?");
}
}
}
}
}
if (!FoundMatchingHash || !EntryUnderReview) {
return std::nullopt;
}
IndexEntry& Entry = *EntryUnderReview;
fextl::vector<uint64_t> GuestPages;
for (uint32_t i = 0; i < Entry.GuestExtents.size(); i += 2) {
uint64_t FirstPage = (Entry.GuestExtents[i] + GuestRIP) & Utils::FEX_PAGE_MASK;
uint64_t LastPage = (Entry.GuestExtents[i] + GuestRIP + Entry.GuestExtents[i + 1] - 1) & Utils::FEX_PAGE_MASK;
for (uint64_t Page = FirstPage; Page <= LastPage; Page += Utils::FEX_PAGE_SIZE) {
if (GuestPages.size() == 0 || Page != GuestPages.back()) {
GuestPages.push_back(Page);
}
}
}
// this seems to be a full hit, pull from disk and check the entry is big enough to have everything (except GuestCode)
CodeHitData HitData;
uint32_t EntrySizeWithoutGuestCode = Entry.Size - Entry.GuestSize;
HitData.Blob.resize(GuestPages.size() * sizeof(uint64_t) + EntrySizeWithoutGuestCode);
memcpy(HitData.Blob.data(), GuestPages.data(), GuestPages.size() * sizeof(uint64_t));
uint32_t BlobOffset = GuestPages.size() * sizeof(uint64_t);
if (!Entry.DB->ReadCacheBlob(Entry.Offset, {HitData.Blob.data() + BlobOffset, EntrySizeWithoutGuestCode})) {
return std::nullopt;
}
if (EntrySizeWithoutGuestCode < sizeof(BlobFixedHeader)) {
return std::nullopt;
}
BlobFixedHeader Header;
memcpy(&Header, HitData.Blob.data() + BlobOffset, sizeof(Header));
BlobOffset += sizeof(Header);
uint32_t SizeNeeded = sizeof(Header) + Header.HostSize + Header.EntryPointCount * (sizeof(uint64_t) + sizeof(uint32_t));
SizeNeeded += Header.SmallRelocCount * sizeof(BlobSmallRelocation) + Header.ThunkRelocCount * sizeof(BlobThunkRelocation);
if (EntrySizeWithoutGuestCode != SizeNeeded) {
return std::nullopt;
}
if (Entry.GuestSize != Header.GuestSize || !XXH128_isEqual(Header.GuestHash, Entry.GuestHash)) {
return std::nullopt;
}
HitData.HostCode = {HitData.Blob.data() + BlobOffset, Header.HostSize};
BlobOffset += Header.HostSize;
HitData.EntryPointRIPs = {reinterpret_cast<uint64_t*>(HitData.Blob.data() + BlobOffset), Header.EntryPointCount};
BlobOffset += Header.EntryPointCount * sizeof(uint64_t);
HitData.EntryPointHostOffsets = {reinterpret_cast<const uint32_t*>(HitData.Blob.data() + BlobOffset), Header.EntryPointCount};
BlobOffset += Header.EntryPointCount * sizeof(uint32_t);
std::span<const BlobSmallRelocation> SmallRelocs = {reinterpret_cast<const BlobSmallRelocation*>(HitData.Blob.data() + BlobOffset),
Header.SmallRelocCount};
BlobOffset += Header.SmallRelocCount * sizeof(BlobSmallRelocation);
std::span<const BlobThunkRelocation> ThunkRelocs = {reinterpret_cast<const BlobThunkRelocation*>(HitData.Blob.data() + BlobOffset),
Header.ThunkRelocCount};
BlobOffset += Header.ThunkRelocCount * sizeof(BlobThunkRelocation);
HitData.GuestPages = {reinterpret_cast<uint64_t*>(HitData.Blob.data()), GuestPages.size()};
for (auto& EntryPointRip : HitData.EntryPointRIPs) {
EntryPointRip += GuestRIP;
}
if (!CTX->CodeCache.ApplyPackedCodeRelocations(GuestRIP, std::as_writable_bytes(HitData.HostCode), SmallRelocs, ThunkRelocs)) {
return std::nullopt;
}
return HitData;
}
void DiskCache::Validate(uint64_t GuestCodeKey, const CodeHitData& Hit, const CPU::CPUBackend::CompiledCode& CompiledCode,
std::optional<ExecutableFileSectionInfo> Region) {
if (!Validation) {
return;
}
auto KeyPrefix = (Region && Region->FileStartVA) ? "file" : "anon";
if (Hit.HostCode.size() != CompiledCode.Size) {
LogMan::Msg::EFmt("DiskCache: validate host size mismatch key={}-{:x} cached={} live={}", KeyPrefix, GuestCodeKey,
Hit.HostCode.size(), CompiledCode.Size);
} else if (memcmp(Hit.HostCode.data(), CompiledCode.BlockBegin, CompiledCode.Size) != 0) {
bool PreviousByteDiff = false;
size_t FirstDiff = 0;
size_t DiffCount = 0;
for (size_t i = 0; i < CompiledCode.Size; i++) {
if (Hit.HostCode[i] != CompiledCode.BlockBegin[i]) {
if (DiffCount == 0) {
FirstDiff = i;
}
if (!PreviousByteDiff) {
DiffCount++;
}
PreviousByteDiff = true;
} else {
PreviousByteDiff = false;
}
}
// align to 4 bytes to read host arm easier
const size_t DiffStart = (FirstDiff & ~3ULL) >= 8 ? (FirstDiff & ~3ULL) - 8 : 0;
const size_t ContextBytes = std::min<size_t>(16, CompiledCode.Size - DiffStart);
LogMan::Msg::EFmt("DiskCache: validate host code mismatch key={}-{:x} size={} firstoffset={} ndiff={} cached=[{:02x}] live=[{:02x}]",
KeyPrefix, GuestCodeKey, CompiledCode.Size, FirstDiff, DiffCount,
fmt::join(std::span<const uint8_t> {Hit.HostCode.data() + DiffStart, ContextBytes}, " "),
fmt::join(std::span<const uint8_t> {CompiledCode.BlockBegin + DiffStart, ContextBytes}, " "));
}
}
struct DiskCache::CacheStoreWorkItem final : WorkQueueThread::WorkItem {
DiskCache* Self;
IndexedDB* DB;
MesaFOZ::foz_payload_key UniqueKey;
uint64_t LookupKey;
fextl::vector<uint8_t> Blob;
fextl::vector<uint8_t> IndexBlob;
CacheStoreWorkItem(DiskCache* Self, IndexedDB* DB, const MesaFOZ::foz_payload_key& UniqueKey, uint64_t LookupKey,
fextl::vector<uint8_t>&& Blob, fextl::vector<uint8_t>&& IndexBlob)
: Self(Self)
, DB(DB)
, UniqueKey(UniqueKey)
, LookupKey(LookupKey)
, Blob(std::move(Blob))
, IndexBlob(std::move(IndexBlob)) {}
void Run() override {
DB->StoreCacheBlob(UniqueKey, LookupKey, Blob, Self->Index, Self->IndexLock, IndexBlob);
}
};
bool DiskCache::Store(Core::InternalThreadState* Thread, std::optional<ExecutableFileSectionInfo> Region, uint64_t GuestRIP,
uint64_t GuestCodeKey, std::span<const uint8_t> GuestCode, const CPU::CPUBackend::CompiledCode& CompiledCode,
std::span<const FEXCore::CPU::Relocation> Relocations, const Frontend::Decoder::DecodedBlockInformation* DecodedBlockInfo) {
if (!IsWritingDiskCache()) {
return false;
}
if (!DecodedBlockInfo) {
return false;
}
// check for any reloc targets outside of our jurisdiction
// todo what are they exactly? caching those blocks is great when it works, so need to figure this out and make finer-grained if we can
if (RelocationFilter && Region) {
for (const auto& Reloc : Relocations) {
if (Reloc.Header.Type != CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL && Reloc.Header.Type != CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE) {
continue;
}
uint64_t Target = Reloc.GuestRIP.GuestRIP;
if (Target >= Region->BeginVA && Target < Region->EndVA) {
continue;
}
auto TargetSection = CTX->SyscallHandler->LookupExecutableFileSection(Thread, Target);
if (!TargetSection || TargetSection->FileInfo.FileId != Region->FileInfo.FileId) {
// we don't know where it's pointing, so we don't know how to encode the offset, so we can't cache atm
return false;
}
}
}
uint32_t SmallRelocCount = 0;
uint32_t ThunkRelocCount = 0;
for (const auto& Reloc : Relocations) {
if (Reloc.Header.Type == CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE) {
ThunkRelocCount++;
} else {
SmallRelocCount++;
}
}
fextl::set<uint64_t> DataMaskAddresses;
fextl::vector<uint32_t> ExactGuestCodeExtents;
uint64_t CurStartExtent = 0, CurEndExtent = 0;
const Frontend::Decoder::DecodedBlocks* LastBlock = nullptr;
for (auto& SubBlock : DecodedBlockInfo->Blocks) {
if (SubBlock.BlockStatus != Frontend::Decoder::DecodedBlockStatus::SUCCESS) {
return false;
}
if (!CurStartExtent) {
CurStartExtent = SubBlock.Entry;
CurEndExtent = SubBlock.Entry + SubBlock.Size;
} else {
LOGMAN_THROW_A_FMT(SubBlock.Entry >= CurEndExtent, "DecodedBlocks not sorted or overlapping?");
if (SubBlock.Entry == CurEndExtent) {
CurEndExtent = SubBlock.Entry + SubBlock.Size;
} else {
ExactGuestCodeExtents.push_back(CurStartExtent - GuestRIP);
ExactGuestCodeExtents.push_back(CurEndExtent - CurStartExtent);
CurStartExtent = SubBlock.Entry;
CurEndExtent = SubBlock.Entry + SubBlock.Size;
}
}
// split extents according to data masks as well
for (auto& Mask : SubBlock.DataMasks) {
if (Mask.FieldAddress > CurStartExtent) {
ExactGuestCodeExtents.push_back(CurStartExtent - GuestRIP);
ExactGuestCodeExtents.push_back(Mask.FieldAddress - CurStartExtent);
}
CurStartExtent = Mask.FieldAddress + Mask.ValueSize;
DataMaskAddresses.insert(Mask.FieldAddress);
}
LastBlock = &SubBlock;
}
if (LastBlock && (CurStartExtent != GuestRIP || CurEndExtent != GuestRIP + GuestCode.size())) {
ExactGuestCodeExtents.push_back(CurStartExtent - GuestRIP);
ExactGuestCodeExtents.push_back(CurEndExtent - CurStartExtent);
}
if (ExactGuestCodeExtents.size() == 0) {
ExactGuestCodeExtents.reserve(2);
ExactGuestCodeExtents.push_back(0);
ExactGuestCodeExtents.push_back(GuestCode.size());
}
uint64_t GuestFootprint = XXH3_64bits(ExactGuestCodeExtents.data(), ExactGuestCodeExtents.size() * sizeof(uint32_t));
// if (ExactGuestCodeExtents.size()) {
// LogMan::Msg::IFmt("store! length {:d}", GuestCode.size());
// for(uint32_t i = 0; i < ExactGuestCodeExtents.size(); i+=2 ) {
// LogMan::Msg::IFmt("extent {} {}", ExactGuestCodeExtents[i], ExactGuestCodeExtents[i]+ExactGuestCodeExtents[i+1]);
// }
// }
const uint32_t EntryPointCount = (uint32_t)CompiledCode.EntryPoints.size();
const size_t HeaderOffset = 0;
const size_t HostCodeOffset = HeaderOffset + sizeof(BlobFixedHeader);
const size_t EntryPointRIPsOffset = HostCodeOffset + CompiledCode.Size;
const size_t EntryPointHostOffsetsOffset = EntryPointRIPsOffset + EntryPointCount * sizeof(uint64_t);
const size_t SmallRelocsOffset = EntryPointHostOffsetsOffset + EntryPointCount * sizeof(uint32_t);
const size_t ThunkRelocsOffset = SmallRelocsOffset + SmallRelocCount * sizeof(BlobSmallRelocation);
const size_t GuestCodeOffset = ThunkRelocsOffset + ThunkRelocCount * sizeof(BlobThunkRelocation);
const size_t TotalSize = GuestCodeOffset + GuestCode.size();
// we'll copy everything into here and pass it to the Writer, then return to caller quickly
fextl::vector<uint8_t> Blob;
Blob.resize(TotalSize);
uint8_t* BlobData = Blob.data();
BlobFixedHeader Header {
.GuestSize = (uint32_t)GuestCode.size(),
.HostSize = (uint32_t)CompiledCode.Size,
.EntryPointCount = EntryPointCount,
.SmallRelocCount = SmallRelocCount,
.ThunkRelocCount = ThunkRelocCount,
};
{
XXH3_state_t HashState;
XXH3_128bits_reset(&HashState);
for (uint32_t i = 0; i < ExactGuestCodeExtents.size(); i += 2) {
XXH3_128bits_update(&HashState, GuestCode.data() + ExactGuestCodeExtents[i], ExactGuestCodeExtents[i + 1]);
}
Header.GuestHash = XXH3_128bits_digest(&HashState);
}
memcpy(BlobData + HeaderOffset, &Header, sizeof(Header));
memcpy(BlobData + HostCodeOffset, CompiledCode.BlockBegin, CompiledCode.Size);
// pack and relocate entrypoints
auto* EntryRIPs = reinterpret_cast<uint64_t*>(BlobData + EntryPointRIPsOffset);
auto* EntryHostOffsets = reinterpret_cast<uint32_t*>(BlobData + EntryPointHostOffsetsOffset);
uint32_t EntryIdx = 0;
for (auto [GuestAddr, HostAddr] : CompiledCode.EntryPoints) {
EntryRIPs[EntryIdx] = GuestAddr - GuestRIP;
EntryHostOffsets[EntryIdx] = uint32_t(HostAddr - CompiledCode.BlockBegin);
EntryIdx++;
}
// pack relocations
auto* SmallRelocs = reinterpret_cast<BlobSmallRelocation*>(BlobData + SmallRelocsOffset);
auto* ThunkRelocs = reinterpret_cast<BlobThunkRelocation*>(BlobData + ThunkRelocsOffset);
uint32_t SmallIdx = 0;
uint32_t ThunkIdx = 0;
for (const auto& Reloc : Relocations) {
switch (Reloc.Header.Type) {
// it's important to zero-init the element completely so we don't have garbage in unused fields
// this way, the caches stay deterministic across machines
case CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
BlobSmallRelocation SmallReloc = {};
SmallReloc.Offset = Reloc.Header.Offset;
SmallReloc.Type = uint8_t(Reloc.Header.Type);
SmallReloc.Named.Symbol = uint32_t(Reloc.NamedSymbolLiteral.Symbol);
SmallRelocs[SmallIdx++] = SmallReloc;
break;
}
case CPU::RelocationTypes::RELOC_GUEST_RIP_LITERAL: {
BlobSmallRelocation SmallReloc = {};
SmallReloc.Offset = Reloc.Header.Offset;
SmallReloc.Type = uint8_t(Reloc.Header.Type);
SmallReloc.RIPLiteral.GuestRIP = Reloc.GuestRIP.GuestRIP - GuestRIP;
SmallRelocs[SmallIdx++] = SmallReloc;
break;
}
case CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
BlobSmallRelocation SmallReloc = {};
SmallReloc.Offset = Reloc.Header.Offset;
SmallReloc.Type = uint8_t(Reloc.Header.Type);
SmallReloc.RIPMove.RegisterIndex = Reloc.GuestRIP.RegisterIndex;
SmallReloc.RIPMove.GuestRIP = Reloc.GuestRIP.GuestRIP - GuestRIP;
SmallRelocs[SmallIdx++] = SmallReloc;
break;
}
case CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_DATA_MOVE:
case CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_MOVE:
case CPU::RelocationTypes::RELOC_GUEST_PATCHABLE_RIP_LITERAL: {
// same data for all, relative vs. not and register vs. literal will depend on type on apply
BlobSmallRelocation SmallReloc = {};
SmallReloc.Offset = Reloc.Header.Offset;
SmallReloc.Type = uint8_t(Reloc.Header.Type);
SmallReloc.PatchableData.RegisterIndex = Reloc.GuestPatchableData.RegisterIndex;
SmallReloc.PatchableData.ValueSize = Reloc.GuestPatchableData.ValueSize;
SmallReloc.PatchableData.SiteOffset = uint32_t(Reloc.GuestPatchableData.SiteAddress - GuestRIP);
SmallRelocs[SmallIdx++] = SmallReloc;
// mark the corresponding data mask consumed - we might not find one if they got removed due to the smc workaround
// the hash will just fail on lookup later
auto It = DataMaskAddresses.find(Reloc.GuestPatchableData.SiteAddress);
if (It != DataMaskAddresses.end()) {
DataMaskAddresses.erase(It);
}
break;
}
case CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
BlobThunkRelocation BigReloc = {};
BigReloc.Offset = Reloc.Header.Offset;
BigReloc.RegisterIndex = Reloc.NamedThunkMove.RegisterIndex;
memcpy(BigReloc.SymbolHash, &Reloc.NamedThunkMove.Symbol, sizeof(BigReloc.SymbolHash));
ThunkRelocs[ThunkIdx++] = BigReloc;
break;
}
}
}
if (!DataMaskAddresses.empty()) {
LogMan::Msg::IFmt("DiskCache: DataMask unaccounted for! {:x}", GuestCodeKey);
// this would mean we omitted contents in the hash that we're not going to patch, which would be loading corrupt code
return false;
}
memcpy(BlobData + GuestCodeOffset, GuestCode.data(), GuestCode.size());
auto RangeInfo = CTX->SyscallHandler->QueryGuestExecutableRange(Thread, GuestRIP);
uint64_t LookupKey =
MakeLookupKey(Thread, GuestCodeKey, RangeInfo.Writable, GuestRIP == CTX->GetMonoBackPatcherBlock().load(std::memory_order_relaxed));
MesaFOZ::foz_payload_key Key = {};
{
XXH3_state_t HashState;
XXH3_128bits_reset(&HashState);
XXH3_128bits_update(&HashState, &Header.GuestHash, sizeof(Header.GuestHash));
XXH3_128bits_update(&HashState, &LookupKey, sizeof(LookupKey));
XXH128_hash_t UniqueKey = XXH3_128bits_digest(&HashState);
fextl::string BlobName = fextl::fmt::format("{:016x}{:016x}", UniqueKey.high64, UniqueKey.low64);
memcpy(Key.bytes, BlobName.data(), BlobName.size());
}
fextl::vector<uint8_t> IndexBlob;
IndexBlob.resize(sizeof(IndexExtraBlobHeader) + ExactGuestCodeExtents.size() * sizeof(uint32_t));
IndexExtraBlobHeader IndexBlobHeader {Header.GuestHash, GuestFootprint, Header.GuestSize, (uint32_t)ExactGuestCodeExtents.size()};
memcpy(IndexBlob.data(), &IndexBlobHeader, sizeof(IndexExtraBlobHeader));
memcpy(IndexBlob.data() + sizeof(IndexExtraBlobHeader), ExactGuestCodeExtents.data(), ExactGuestCodeExtents.size() * sizeof(uint32_t));
// hand the rest off to the writer thread
Writer->QueueWork(fextl::make_unique<CacheStoreWorkItem>(this, RWCacheDB.get(), Key, LookupKey, std::move(Blob), std::move(IndexBlob)));
return true;
}
uint16_t GetFormatVersion() {
return FormatVersion;
}
} // namespace DiskCache
} // namespace FEXCore