From ba93bdd66de2a3e7d99fec551e95c01265790e8a Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Mon, 30 Mar 2026 18:16:46 -0700 Subject: [PATCH] FEXRootFSFetcher: Improve hashing performance Don't use pread, instead map the file and madvise larger blocks. This removes copying overhead as its just mapping file pages in instead. Also splits the implementation of file reading from hashing to make tinkering less involved, as if I want more performance out of this (say due to live hashing) then it's easier to tinker. Improves hashing performance from ~2.2GB/s to ~3.6GB/s on my system, which is CPU bounded by xxhash here. --- Source/Tools/FEXRootFSFetcher/XXFileHash.cpp | 104 +++++++++++++++---- 1 file changed, 86 insertions(+), 18 deletions(-) diff --git a/Source/Tools/FEXRootFSFetcher/XXFileHash.cpp b/Source/Tools/FEXRootFSFetcher/XXFileHash.cpp index b894065bc..f1af52629 100644 --- a/Source/Tools/FEXRootFSFetcher/XXFileHash.cpp +++ b/Source/Tools/FEXRootFSFetcher/XXFileHash.cpp @@ -5,12 +5,74 @@ #include #include #include -#include #include +#include namespace XXFileHash { -// 32MB blocks -constexpr static size_t BLOCK_SIZE = 32 * 1024 * 1024; +class Reader { +public: + Reader(int fd, size_t Size) + : fd {fd} + , Size {Size} {} + + virtual ~Reader() = default; + + bool Initialized() const { + return IsInitialized; + } + + using Callback = std::function; + virtual bool Read(Callback cb) = 0; + +protected: + int fd {}; + size_t Size {}; + bool IsInitialized {}; +}; + +class MemoryReader final : public Reader { +public: + MemoryReader(int fd, size_t Size) + : Reader(fd, Size) { + Ptr = reinterpret_cast(mmap(nullptr, Size, PROT_READ, MAP_SHARED, fd, 0)); + IsInitialized = Ptr != MAP_FAILED; + } + + ~MemoryReader() { + munmap(reinterpret_cast(Ptr), Size); + } + + bool Read(Callback cb) override { + auto ReadPtr = Ptr; + const auto ReadEndPtr = Ptr + Size; + size_t ReadSize {}; + + // Claim sequential access. + ::madvise(reinterpret_cast(ReadPtr), Size, MADV_SEQUENTIAL); + + while (ReadPtr < ReadEndPtr) { + ReadSize = std::min(READ_BLOCK_SIZE, ReadEndPtr - ReadPtr); + + if (!cb(ReadPtr, ReadSize)) { + return false; + } + + // Only allow a single block read to be resident. + ::madvise(reinterpret_cast(ReadPtr), ReadSize, MADV_DONTNEED); + + ReadPtr += ReadSize; + } + + return true; + } + +private: + std::byte* Ptr {}; + + // Only allow 128MB in flight. + constexpr static size_t READ_BLOCK_SIZE = 128 * 1024 * 1024; +}; + std::optional HashFile(const fextl::string& Filepath) { int fd = open(Filepath.c_str(), O_RDONLY); if (fd == -1) { @@ -48,33 +110,39 @@ std::optional HashFile(const fextl::string& Filepath) { return HadError(); } + MemoryReader Read(fd, Size); + + if (!Read.Initialized()) { + return HadError(); + } + + const auto Start = std::chrono::high_resolution_clock::now(); + auto Now = Start; const double SizeD = Size; - std::vector Data(BLOCK_SIZE); - off_t CurrentOffset = 0; - auto Now = std::chrono::high_resolution_clock::now(); + size_t CurrentOffset {}; - // Let the kernel know that we will be reading linearly - posix_fadvise(fd, 0, Size, POSIX_FADV_SEQUENTIAL); - while (CurrentOffset < Size) { - - ssize_t Result = pread(fd, Data.data(), BLOCK_SIZE, CurrentOffset); - if (Result == -1) { - return HadError(); + auto CB_XXH = [&](const void* Data, size_t BlockSize) -> bool { + if (XXH3_64bits_update(State, Data, BlockSize) == XXH_ERROR) { + return false; } - if (XXH3_64bits_update(State, Data.data(), Result) == XXH_ERROR) { - return HadError(); - } auto Cur = std::chrono::high_resolution_clock::now(); auto Dur = Cur - Now; if (Dur >= std::chrono::seconds(1)) { fmt::print("{:.2}% hashed\n", (double)CurrentOffset / SizeD * 100.0); Now = Cur; } - CurrentOffset += Result; + + CurrentOffset += BlockSize; + + return true; + }; + + if (!Read.Read(CB_XXH)) { + return HadError(); } - const XXH64_hash_t Hash = XXH3_64bits_digest(State); + const auto Hash = XXH3_64bits_digest(State); XXH3_freeState(State); close(fd);