Merge pull request #1885 from FEX-Emu/skmp/simpler-memory-stealing

Allocator: Simplify StealMemory, make it less chatty with kernel space
This commit is contained in:
Ryan Houdek authored and GitHub committed 2022-08-20 06:20:36 -07:00
commit 04678f8404
4 files changed
+144 -176

No files matched your search

+102 -79
View File
@@ -6,6 +6,10 @@
#include <FEXHeaderUtils/TypeDefines.h>
#include <array>
#include <asm-generic/errno-base.h>
#include <cctype>
#include <cstdio>
#include <fcntl.h>
#include <sys/mman.h>
#include <sys/user.h>
#ifdef ENABLE_JEMALLOC
@@ -131,91 +135,119 @@ namespace FEXCore::Allocator {
FEX_UNREACHABLE;
}
PtrCache* StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
PtrCache *Cache{};
uint64_t CacheSize{};
uint64_t CurrentCacheOffset = 0;
constexpr std::array<size_t, 10> ReservedVMARegionSizes = {{
// Anything larger than 64GB fails out
64ULL * 1024 * 1024 * 1024, // 64GB
32ULL * 1024 * 1024 * 1024, // 32GB
16ULL * 1024 * 1024 * 1024, // 16GB
4ULL * 1024 * 1024 * 1024, // 4GB
1ULL * 1024 * 1024 * 1024, // 1GB
512ULL * 1024 * 1024, // 512MB
128ULL * 1024 * 1024, // 128MB
32ULL * 1024 * 1024, // 32MB
1ULL * 1024 * 1024, // 1MB
4096ULL // One page
}};
constexpr size_t AllocationSizeMaxIndex = ReservedVMARegionSizes.size() - 1;
uint64_t CurrentSizeIndex = 0;
#define STEAL_LOG(...) // fprintf(stderr, __VA_ARGS__)
int PROT_FLAGS = PROT_READ | PROT_WRITE;
for (size_t MemoryOffset = Begin; MemoryOffset < End;) {
size_t AllocationSize = ReservedVMARegionSizes[CurrentSizeIndex];
size_t MemoryOffsetUpper = MemoryOffset + AllocationSize;
std::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End) {
std::vector<MemoryRegion> Regions;
int MapsFD = open("/proc/self/maps", O_RDONLY);
LogMan::Throw::AFmt(MapsFD != -1, "Failed to open /proc/self/maps");
// If we would go above the upper bound on size then try the next size
if (MemoryOffsetUpper > End) {
++CurrentSizeIndex;
continue;
enum {ParseBegin, ParseEnd, ScanEnd} State = ParseBegin;
uintptr_t RegionBegin = 0;
uintptr_t RegionEnd = 0;
char Buffer[2048];
const char *Cursor;
ssize_t Remaining = 0;
for(;;) {
if (Remaining == 0) {
do {
Remaining = read(MapsFD, Buffer, sizeof(Buffer));
} while ( Remaining == -1 && errno == EAGAIN);
Cursor = Buffer;
}
void *Ptr = ::mmap(reinterpret_cast<void*>(MemoryOffset), AllocationSize, PROT_FLAGS, MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE | MAP_FIXED_NOREPLACE, -1, 0);
if (Remaining == 0 && State == ParseBegin) {
STEAL_LOG("[%d] EndOfFile; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
// If we managed to allocate and not get the address we want then unmap it
// This happens with kernels older than 4.17
if (reinterpret_cast<uintptr_t>(Ptr) + AllocationSize > End) {
::munmap(Ptr, AllocationSize);
Ptr = reinterpret_cast<void*>(~0ULL);
}
auto MapBegin = std::max(RegionEnd, Begin);
auto MapEnd = End;
// If we failed to allocate and we are on the smallest allocation size then just continue onward
// This page was unmappable
if (reinterpret_cast<uintptr_t>(Ptr) == ~0ULL && CurrentSizeIndex == AllocationSizeMaxIndex) {
CurrentSizeIndex = 0;
MemoryOffset += AllocationSize;
continue;
}
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
// Congratulations we were able to map this bit
// Reset and claim it was available
if (reinterpret_cast<uintptr_t>(Ptr) != ~0ULL) {
if (!Cache) {
Cache = reinterpret_cast<PtrCache *>(Ptr);
CacheSize = AllocationSize;
PROT_FLAGS = PROT_NONE;
}
else {
Cache[CurrentCacheOffset] = {
.Ptr = static_cast<uint64_t>(reinterpret_cast<uint64_t>(Ptr)),
.Size = static_cast<uint64_t>(AllocationSize)
};
++CurrentCacheOffset;
if (MapEnd > MapBegin) {
STEAL_LOG(" Reserving\n");
auto MapSize = MapEnd - MapBegin;
auto Alloc = mmap((void*)MapBegin, MapSize, PROT_NONE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0);
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", MapBegin, MapSize);
LogMan::Throw::AFmt(Alloc == (void*)MapBegin, "mmap({},{:x}) returned {} instead of {:x}", Alloc, MapBegin);
Regions.push_back({(void*)MapBegin, MapSize});
}
CurrentSizeIndex = 0;
MemoryOffset += AllocationSize;
close(MapsFD);
return Regions;
}
LogMan::Throw::AFmt(Remaining > 0, "Failed to parse /proc/self/maps");
auto c = *Cursor++;
Remaining--;
if (State == ScanEnd) {
if (c == '\n') {
State = ParseBegin;
}
continue;
}
// Couldn't allocate at this size
// Increase and continue
++CurrentSizeIndex;
if (State == ParseBegin) {
if (c == '-') {
STEAL_LOG("[%d] ParseBegin; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
auto MapBegin = std::max(RegionEnd, Begin);
auto MapEnd = std::min(RegionBegin, End);
STEAL_LOG(" MapBegin: %016lX MapEnd: %016lX\n", MapBegin, MapEnd);
if (MapEnd > MapBegin) {
STEAL_LOG(" Reserving\n");
auto MapSize = MapEnd - MapBegin;
auto Alloc = mmap((void*)MapBegin, MapSize, PROT_NONE, MAP_ANONYMOUS | MAP_NORESERVE | MAP_PRIVATE | MAP_FIXED_NOREPLACE, -1, 0);
LogMan::Throw::AFmt(Alloc != MAP_FAILED, "mmap({:x},{:x}) failed", MapBegin, MapSize);
LogMan::Throw::AFmt(Alloc == (void*)MapBegin, "mmap({},{:x}) returned {} instead of {:x}", Alloc, MapBegin);
Regions.push_back({(void*)MapBegin, MapSize});
}
RegionBegin = 0;
RegionEnd = 0;
State = ParseEnd;
continue;
} else {
LogMan::Throw::AFmt(std::isalpha(c) || std::isdigit(c), "Unexpected char '{}' in ParseBegin", c);
RegionBegin = (RegionBegin << 4) | (c <= '9' ? (c - '0') : (c - 'a' + 10));
}
}
if (State == ParseEnd) {
if (c == ' ') {
STEAL_LOG("[%d] ParseEnd; RegionBegin: %016lX RegionEnd: %016lX\n", __LINE__, RegionBegin, RegionEnd);
State = ScanEnd;
continue;
} else {
LogMan::Throw::AFmt(std::isalpha(c) || std::isdigit(c), "Unexpected char '{}' in ParseEnd", c);
RegionEnd = (RegionEnd << 4) | (c <= '9' ? (c - '0') : (c - 'a' + 10));
}
}
}
Cache[CurrentCacheOffset] = {
.Ptr = static_cast<uint64_t>(reinterpret_cast<uint64_t>(Cache)),
.Size = CacheSize,
};
return Cache;
ERROR_AND_DIE_FMT("unreachable");
}
PtrCache* Steal48BitVA() {
std::vector<MemoryRegion> Steal48BitVA() {
size_t Bits = FEXCore::Allocator::DetermineVASize();
if (Bits < 48) {
return nullptr;
return {};
}
uintptr_t Begin48BitVA = 0x0'8000'0000'0000ULL;
@@ -223,18 +255,9 @@ namespace FEXCore::Allocator {
return StealMemoryRegion(Begin48BitVA, End48BitVA);
}
void ReclaimMemoryRegion(PtrCache* Regions) {
if (Regions == nullptr) {
return;
}
for (size_t i = 0;; ++i) {
void *Ptr = reinterpret_cast<void*>(Regions[i].Ptr);
size_t Size = Regions[i].Size;
::munmap(Ptr, Size);
if (Ptr == Regions) {
break;
}
void ReclaimMemoryRegion(const std::vector<MemoryRegion> &Regions) {
for (const auto &Region: Regions) {
::munmap(Region.Ptr, Region.Size);
}
}
}
+33 -89
View File
@@ -141,13 +141,19 @@ namespace Alloc::OSAllocator {
}
// 32-bit old kernel workarounds
FEXCore::Allocator::PtrCache *Steal32BitIfOldKernel();
std::vector<FEXCore::Allocator::MemoryRegion> Steal32BitIfOldKernel();
};
void OSAllocator_64Bit::DetermineVASize() {
size_t Bits = FEXCore::Allocator::DetermineVASize();
uintptr_t Size = 1ULL << Bits;
UPPER_BOUND = Size;
#if _M_X86_64 // Last page cannot be allocated on x86
UPPER_BOUND -= FHU::FEX_PAGE_SIZE;
#endif
UPPER_BOUND_PAGE = UPPER_BOUND / FHU::FEX_PAGE_SIZE;
}
@@ -490,11 +496,11 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) {
return 0;
}
FEXCore::Allocator::PtrCache *OSAllocator_64Bit::Steal32BitIfOldKernel() {
std::vector<FEXCore::Allocator::MemoryRegion> OSAllocator_64Bit::Steal32BitIfOldKernel() {
// First calculate kernel version
struct utsname buf{};
if (uname(&buf) == -1) {
return nullptr;
return {};
}
int32_t Major{};
@@ -512,7 +518,7 @@ FEXCore::Allocator::PtrCache *OSAllocator_64Bit::Steal32BitIfOldKernel() {
if (Version >= ((4 << 24) | (17 << 16) | 0)) {
// If the kernel is >= 4.17 then it supports MAP_FIXED_NOREPLACE
return nullptr;
return {};
}
constexpr size_t LOWER_BOUND_32 = 0x1'0000;
@@ -523,101 +529,39 @@ FEXCore::Allocator::PtrCache *OSAllocator_64Bit::Steal32BitIfOldKernel() {
OSAllocator_64Bit::OSAllocator_64Bit() {
DetermineVASize();
auto ArrayPtr = Steal32BitIfOldKernel();
auto LowMem = Steal32BitIfOldKernel();
// On allocation try and steal the entire upper 64bits of address space for mapping
constexpr std::array<size_t, 8> ReservedVMARegionSizes = {{
// Anything larger than 64GB fails out
64ULL * 1024 * 1024 * 1024, // 64GB
32ULL * 1024 * 1024 * 1024, // 32GB
16ULL * 1024 * 1024 * 1024, // 16GB
4ULL * 1024 * 1024 * 1024, // 4GB
1ULL * 1024 * 1024 * 1024, // 1GB
512ULL * 1024 * 1024, // 512MB
128ULL * 1024 * 1024, // 128MB
4096ULL // One page
}};
auto Ranges = FEXCore::Allocator::StealMemoryRegion(LOWER_BOUND, UPPER_BOUND);
constexpr size_t AllocationSizeMaxIndex = ReservedVMARegionSizes.size() - 1;
for (auto [Ptr, AllocationSize]: Ranges) {
if (!ObjectAlloc) {
auto MaxSize = std::min(size_t(64) * 1024 * 1024, AllocationSize);
// Have the first region only be 4GB VMA
// Avoids conflicts with some tests
uint64_t CurrentSizeIndex = 3;
ReservedVMARegion *PrevReserved{};
for (size_t MemoryOffset = LOWER_BOUND; MemoryOffset < UPPER_BOUND;) {
size_t AllocationSize = ReservedVMARegionSizes[CurrentSizeIndex];
size_t MemoryOffsetUpper = MemoryOffset + AllocationSize;
// Allocate up to 64 MiB the first allocation for an intrusive allocator
mprotect(Ptr, MaxSize, PROT_READ | PROT_WRITE);
// If we would go above the upper bound on size then try the next size
if (MemoryOffsetUpper > UPPER_BOUND) {
++CurrentSizeIndex;
continue;
}
// This enables the kernel to use transparent large pages in the allocator which can reduce memory pressure
::madvise(Ptr, MaxSize, MADV_HUGEPAGE);
void *Ptr = ::mmap(reinterpret_cast<void*>(MemoryOffset), AllocationSize, PROT_NONE, MAP_FIXED_NOREPLACE | MAP_PRIVATE | MAP_ANONYMOUS | MAP_NORESERVE, -1, 0);
ObjectAlloc = new (Ptr) Alloc::ForwardOnlyIntrusiveArenaAllocator(Ptr, MaxSize);
ReservedRegions = ObjectAlloc->new_construct(ReservedRegions, ObjectAlloc);
LiveRegions = ObjectAlloc->new_construct(LiveRegions, ObjectAlloc);
// If we managed to allocate and not get the address we want then unmap it
// This happens with kernels older than 4.17
if (reinterpret_cast<uintptr_t>(Ptr) != MemoryOffset &&
reinterpret_cast<uintptr_t>(Ptr) < LOWER_BOUND) {
::munmap(Ptr, AllocationSize);
Ptr = reinterpret_cast<void*>(~0ULL);
}
// If we failed to allocate and we are on the smallest allocation size then just continue onward
// This page was unmappable
if (reinterpret_cast<uintptr_t>(Ptr) == ~0ULL && CurrentSizeIndex == AllocationSizeMaxIndex) {
CurrentSizeIndex = 0;
MemoryOffset += AllocationSize;
continue;
}
// Congratulations we were able to map this bit
// Reset and claim it was available
if (reinterpret_cast<uintptr_t>(Ptr) != ~0ULL) {
if (!ObjectAlloc) {
// Steal the first allocation for an intrusive allocator
// Will be mprotected correctly already
mprotect(Ptr, AllocationSize, PROT_READ | PROT_WRITE);
ObjectAlloc = new (Ptr) Alloc::ForwardOnlyIntrusiveArenaAllocator(Ptr, AllocationSize);
ReservedRegions = ObjectAlloc->new_construct(ReservedRegions, ObjectAlloc);
LiveRegions = ObjectAlloc->new_construct(LiveRegions, ObjectAlloc);
if (AllocationSize > MaxSize) {
AllocationSize -= MaxSize;
(uint8_t*&)Ptr += MaxSize;
} else {
continue;
}
else {
// If the allocation size is large than a page, then try allowing it to be a huge page
// This enables the kernel to use transparent large pages in the allocator which can reduce memory pressure
// Considering we are allocating the entire VA space, this is a good thing
// If MADV_HUGEPAGE isn't support then this will fail harmlessly
if (AllocationSize > 4096) {
::madvise(Ptr, AllocationSize, MADV_HUGEPAGE);
}
bool Merged = false;
if (PrevReserved) {
Merged = MergeReservedRegionIfPossible(PrevReserved, reinterpret_cast<uint64_t>(Ptr), AllocationSize);
}
if (!Merged) {
ReservedVMARegion *Region = ObjectAlloc->new_construct<ReservedVMARegion>();
Region->Base = reinterpret_cast<uint64_t>(Ptr);
Region->RegionSize = AllocationSize;
ReservedRegions->emplace_back(Region);
PrevReserved = Region;
}
}
CurrentSizeIndex = 0;
MemoryOffset += AllocationSize;
continue;
}
// Couldn't allocate at this size
// Increase and continue
++CurrentSizeIndex;
ReservedVMARegion *Region = ObjectAlloc->new_construct<ReservedVMARegion>();
Region->Base = reinterpret_cast<uint64_t>(Ptr);
Region->RegionSize = AllocationSize;
ReservedRegions->emplace_back(Region);
}
FEXCore::Allocator::ReclaimMemoryRegion(ArrayPtr);
FEXCore::Allocator::ReclaimMemoryRegion(LowMem);
}
OSAllocator_64Bit::~OSAllocator_64Bit() {
+8 -7
View File
@@ -5,6 +5,7 @@
#include <cstdint>
#include <functional>
#include <sys/types.h>
#include <vector>
namespace FEXCore::Allocator {
using MMAP_Hook = void*(*)(void*, size_t, int, int, int, off_t);
@@ -24,19 +25,19 @@ namespace FEXCore::Allocator {
FEX_DEFAULT_VISIBILITY void ClearHooks();
FEX_DEFAULT_VISIBILITY size_t DetermineVASize();
// 48-bit VA handling
struct PtrCache {
uint64_t Ptr;
uint64_t Size;
struct MemoryRegion {
void *Ptr;
size_t Size;
};
FEX_DEFAULT_VISIBILITY PtrCache* StealMemoryRegion(uintptr_t Begin, uintptr_t End);
FEX_DEFAULT_VISIBILITY void ReclaimMemoryRegion(PtrCache* Regions);
FEX_DEFAULT_VISIBILITY std::vector<MemoryRegion> StealMemoryRegion(uintptr_t Begin, uintptr_t End);
FEX_DEFAULT_VISIBILITY void ReclaimMemoryRegion(const std::vector<MemoryRegion> & Regions);
// When running a 64-bit executable on ARM then userspace guest only gets 47 bits of VA
// This is a feature of x86-64 where the kernel gets a full 128TB of VA space
// x86-64 canonical addresses with bit 48 set will sign extend the address (Ignoring LA57)
// AArch64 canonical addresses are only up to bits 48/52 with the remainder being other things
// Use this to reserve the top 128TB of VA so the guest never see it
// Returns nullptr on host VA < 48bits
FEX_DEFAULT_VISIBILITY PtrCache* Steal48BitVA();
FEX_DEFAULT_VISIBILITY std::vector<MemoryRegion> Steal48BitVA();
}
+1 -1
View File
@@ -331,7 +331,7 @@ int main(int argc, char **argv, char **const envp) {
FEXCore::Config::Set(FEXCore::Config::CONFIG_IS64BIT_MODE, Loader.Is64BitMode() ? "1" : "0");
std::unique_ptr<FEX::HLE::MemAllocator> Allocator;
FEXCore::Allocator::PtrCache *Base48Bit{};
std::vector<FEXCore::Allocator::MemoryRegion> Base48Bit;
if (Loader.Is64BitMode()) {
// Destroy the 48th bit if it exists