Refactors: Move CodeBuffers to CPUBackend, SignalHandlerRefCounter to CpuStateFrame, Arm64 Relocation to ARM64Jit

This commit is contained in:
Stefanos Kornilios Misis Poiitidis committed 2022-06-10 10:39:42 +03:00
1 parent 8521aaf65c
commit 0afb3caaed
20 files changed
+363 -432

No files matched your search

+4 -1
View File
@@ -81,6 +81,7 @@ set (SRCS
Interface/Core/LookupCache.cpp
Interface/Core/BlockSamplingData.cpp
Interface/Core/Core.cpp
Interface/Core/CPUBackend.cpp
Interface/Core/CPUID.cpp
Interface/Core/Frontend.cpp
Interface/Core/GdbServer.cpp
@@ -206,7 +207,9 @@ if (ENABLE_JIT_ARM64)
Interface/Core/JIT/Arm64/MemoryOps.cpp
Interface/Core/JIT/Arm64/MiscOps.cpp
Interface/Core/JIT/Arm64/MoveOps.cpp
Interface/Core/JIT/Arm64/VectorOps.cpp)
Interface/Core/JIT/Arm64/VectorOps.cpp
Interface/Core/JIT/Arm64/Arm64Relocations.cpp
)
endif()
set (LIBS vixl dl xxhash tiny-json)
@@ -309,19 +309,6 @@ void Arm64Emitter::PopDynamicRegsAndLR() {
add(sp, sp, SPOffset);
}
void Arm64Emitter::ResetStack() {
if (SpillSlots == 0)
return;
if (IsImmAddSub(SpillSlots * 16)) {
add(sp, sp, SpillSlots * 16);
} else {
// Too big to fit in a 12bit immediate
LoadConstant(x0, SpillSlots * 16);
add(sp, sp, x0);
}
}
void Arm64Emitter::Align16B() {
uint64_t CurrentOffset = GetCursorAddress<uint64_t>();
for (uint64_t i = (16 - (CurrentOffset & 0xF)); i != 0; i -= 4) {
@@ -329,122 +316,4 @@ void Arm64Emitter::Align16B() {
}
}
uint64_t Arm64Emitter::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
switch (Op) {
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
return Dispatcher->ExitFunctionLinkerAddress;
break;
default:
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
break;
}
return ~0ULL;
}
void Arm64Emitter::InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum) {
Relocation MoveABI{};
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint64_t>();
MoveABI.NamedThunkMove.Offset = CurrentCursor - GuestEntry;
MoveABI.NamedThunkMove.Symbol = Sum;
MoveABI.NamedThunkMove.RegisterIndex = Reg.GetCode();
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
LoadConstant(Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
Relocations.emplace_back(MoveABI);
}
Arm64Emitter::NamedSymbolLiteralPair Arm64Emitter::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
uint64_t Pointer = GetNamedSymbolLiteral(Op);
Arm64Emitter::NamedSymbolLiteralPair Lit {
.Lit = Literal(Pointer),
.MoveABI = {
.NamedSymbolLiteral = {
.Header = {
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
},
.Symbol = Op,
.Offset = 0,
},
},
};
return Lit;
}
void Arm64Emitter::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint64_t>();
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - GuestEntry;
place(&Lit.Lit);
Relocations.emplace_back(Lit.MoveABI);
}
void Arm64Emitter::InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant) {
Relocation MoveABI{};
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint64_t>();
MoveABI.GuestRIPMove.Offset = CurrentCursor - GuestEntry;
MoveABI.GuestRIPMove.GuestRIP = Constant;
MoveABI.GuestRIPMove.RegisterIndex = Reg.GetCode();
LoadConstant(Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
Relocations.emplace_back(MoveABI);
}
bool Arm64Emitter::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
size_t DataIndex{};
for (size_t j = 0; j < NumRelocations; ++j) {
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
switch (Reloc->Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
// Relocation occurs at the cursorEntry + offset relative to that cursor
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
// Generate a literal so we can place it
Literal<uint64_t> Lit(Pointer);
place(&Lit);
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
LoadConstant(vixl::aarch64::XRegister(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->NamedThunkMove);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
// XXX: Reenable once the JIT Object Cache is upstream
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
LoadConstant(vixl::aarch64::XRegister(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->GuestRIPMove);
break;
}
}
}
return true;
}
}
@@ -63,8 +63,6 @@ class Arm64Emitter : public vixl::aarch64::Assembler {
protected:
Arm64Emitter(FEXCore::Context::Context *ctx, size_t size);
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
FEXCore::Context::Context *EmitterCTX;
vixl::aarch64::CPU CPU;
void LoadConstant(vixl::aarch64::Register Reg, uint64_t Constant, bool NOPPad = false);
@@ -83,68 +81,7 @@ protected:
void PushCalleeSavedRegisters();
void PopCalleeSavedRegisters();
void ResetStack();
void Align16B();
/**
* @name Relocations
* @{ */
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
/**
* @brief A literal pair relocation object for named symbol literals
*/
struct NamedSymbolLiteralPair {
Literal<uint64_t> Lit;
Relocation MoveABI{};
};
/**
* @brief Inserts a thunk relocation
*
* @param Reg - The GPR to move the thunk handler in to
* @param Sum - The hash of the thunk
*/
void InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum);
/**
* @brief Inserts a guest GPR move relocation
*
* @param Reg - The GPR to move the guest RIP in to
* @param Constant - The guest RIP that will be relocated
*/
void InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant);
/**
* @brief Inserts a named symbol as a literal in memory
*
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
*
* @param Op The named symbol to place
*
* @return A temporary `NamedSymbolLiteralPair`
*/
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
/**
* @brief Place the named symbol literal relocation in memory
*
* @param Lit - Which literal to place
*/
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
std::vector<FEXCore::CPU::Relocation> Relocations;
///< Relocation code loading
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
/** @} */
uint32_t SpillSlots{};
/**
* @brief Current guest RIP entrypoint
*/
uint64_t GuestEntry{};
FEX_CONFIG_OPT(StaticRegisterAllocation, SRA);
};
+74
View File
@@ -0,0 +1,74 @@
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include <FEXCore/Core/CPUBackend.h>
namespace FEXCore {
namespace CPU {
CPUBackend::CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize)
: ThreadState(ThreadState), InitialCodeSize(InitialCodeSize), MaxCodeSize(MaxCodeSize) {}
CPUBackend::~CPUBackend() {
for (auto CodeBuffer : CodeBuffers) {
FreeCodeBuffer(CodeBuffer);
}
CodeBuffers.clear();
}
auto CPUBackend::GetEmptyCodeBuffer() -> CodeBuffer * {
if (ThreadState->CurrentFrame->SignalHandlerRefCounter == 0) {
if (CodeBuffers.empty()) {
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
EmplaceNewCodeBuffer(NewCodeBuffer);
} else {
if (CodeBuffers.size() > 1) {
// If we have more than one code buffer we are tracking then walk them and delete
// This is a cleanup step
for (size_t i = 1; i < CodeBuffers.size(); i++) {
FreeCodeBuffer(CodeBuffers[i]);
}
CodeBuffers.resize(1);
}
// Set the current code buffer to the initial
CurrentCodeBuffer = &CodeBuffers[0];
if (CurrentCodeBuffer->Size != MaxCodeSize) {
FreeCodeBuffer(*CurrentCodeBuffer);
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer->Size *= 1.5;
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MaxCodeSize);
*CurrentCodeBuffer = AllocateNewCodeBuffer(CurrentCodeBuffer->Size);
}
}
} else {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Allocate some new code buffers that we can switch over to instead
auto NewCodeBuffer = AllocateNewCodeBuffer(InitialCodeSize);
EmplaceNewCodeBuffer(NewCodeBuffer);
}
return CurrentCodeBuffer;
}
auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
CodeBuffer Buffer;
Buffer.Size = Size;
Buffer.Ptr = static_cast<uint8_t *>(
FEXCore::Allocator::mmap(nullptr, Buffer.Size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
LOGMAN_THROW_A_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
Dispatcher->RegisterCodeBuffer(Buffer.Ptr, Buffer.Size);
if (ThreadState->CTX->Config.GlobalJITNaming()) {
ThreadState->CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
return Buffer;
}
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
Dispatcher->RemoveCodeBuffer(Buffer.Ptr);
}
}
}
@@ -408,10 +408,9 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
mov(STATE, x0);
// Make sure to adjust the refcounter so we don't clear the cache now
LoadConstant(x0, reinterpret_cast<uint64_t>(&SignalHandlerRefCounter));
ldr(w2, MemOperand(x0));
ldr(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
add(w2, w2, 1);
str(w2, MemOperand(x0));
str(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
// Now push the callback return trampoline to the guest stack
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
@@ -280,7 +280,7 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
// Ref count our faults
// We use this to track if it is safe to clear cache
++SignalHandlerRefCounter;
++ThreadState->CurrentFrame->SignalHandlerRefCounter;
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
// Set the new PC
@@ -640,7 +640,7 @@ bool Dispatcher::HandleSIGILL(int Signal, void *info, void *ucontext) {
// Ref count our faults
// We use this to track if it is safe to clear cache
--SignalHandlerRefCounter;
--ThreadState->CurrentFrame->SignalHandlerRefCounter;
return true;
}
@@ -649,7 +649,7 @@ bool Dispatcher::HandleSIGILL(int Signal, void *info, void *ucontext) {
// Ref count our faults
// We use this to track if it is safe to clear cache
--SignalHandlerRefCounter;
--ThreadState->CurrentFrame->SignalHandlerRefCounter;
return true;
}
@@ -681,7 +681,7 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
// Ref count our faults
// We use this to track if it is safe to clear cache
++SignalHandlerRefCounter;
++ThreadState->CurrentFrame->SignalHandlerRefCounter;
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
return true;
@@ -694,7 +694,7 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
ArchHelpers::Context::SetSp(ucontext, Frame->ReturningStackLocation);
// Our ref counting doesn't matter anymore
SignalHandlerRefCounter = 0;
ThreadState->CurrentFrame->SignalHandlerRefCounter = 0;
// Set the new PC
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
@@ -727,7 +727,7 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
// Ref count our faults
// We use this to track if it is safe to clear cache
--SignalHandlerRefCounter;
--ThreadState->CurrentFrame->SignalHandlerRefCounter;
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
return true;
@@ -55,7 +55,6 @@ public:
/** @} */
uint32_t SignalHandlerRefCounter{};
struct SynchronousFaultDataStruct {
bool FaultToTopAndGeneratedException{};
uint32_t TrapNo;
@@ -345,7 +345,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
// XXX: XMM?
// Make sure to adjust the refcounter so we don't clear the cache now
add(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SignalHandlerRefCountPointer)], 1);
add(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)], 1);
// Now push the callback return trampoline to the guest stack
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
@@ -42,9 +42,6 @@ public:
private:
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *State;
std::unique_ptr<Dispatcher> Dispatcher{};
};
template<typename T>
@@ -0,0 +1,130 @@
/*
$info$
tags: backend|arm64
desc: relocation logic of the arm64 splatter backend
$end_info$
*/
#include "Interface/Core/JIT/Arm64/JITClass.h"
#include "Interface/HLE/Thunks/Thunks.h"
namespace FEXCore::CPU {
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
switch (Op) {
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
return Dispatcher->ExitFunctionLinkerAddress;
break;
default:
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
break;
}
return ~0ULL;
}
void Arm64JITCore::InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum) {
Relocation MoveABI{};
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint64_t>();
MoveABI.NamedThunkMove.Offset = CurrentCursor - GuestEntry;
MoveABI.NamedThunkMove.Symbol = Sum;
MoveABI.NamedThunkMove.RegisterIndex = Reg.GetCode();
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Sum));
LoadConstant(Reg, Pointer, EmitterCTX->Config.CacheObjectCodeCompilation());
Relocations.emplace_back(MoveABI);
}
Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
uint64_t Pointer = GetNamedSymbolLiteral(Op);
Arm64JITCore::NamedSymbolLiteralPair Lit {
.Lit = Literal(Pointer),
.MoveABI = {
.NamedSymbolLiteral = {
.Header = {
.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL,
},
.Symbol = Op,
.Offset = 0,
},
},
};
return Lit;
}
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint64_t>();
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - GuestEntry;
place(&Lit.Lit);
Relocations.emplace_back(Lit.MoveABI);
}
void Arm64JITCore::InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant) {
Relocation MoveABI{};
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint64_t>();
MoveABI.GuestRIPMove.Offset = CurrentCursor - GuestEntry;
MoveABI.GuestRIPMove.GuestRIP = Constant;
MoveABI.GuestRIPMove.RegisterIndex = Reg.GetCode();
LoadConstant(Reg, Constant, EmitterCTX->Config.CacheObjectCodeCompilation());
Relocations.emplace_back(MoveABI);
}
bool Arm64JITCore::ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations) {
size_t DataIndex{};
for (size_t j = 0; j < NumRelocations; ++j) {
const FEXCore::CPU::Relocation *Reloc = reinterpret_cast<const FEXCore::CPU::Relocation *>(&EntryRelocations[DataIndex]);
LOGMAN_THROW_A_FMT((DataIndex % alignof(Relocation)) == 0, "Alignment of relocation wasn't adhered to");
switch (Reloc->Header.Type) {
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_SYMBOL_LITERAL: {
uint64_t Pointer = GetNamedSymbolLiteral(Reloc->NamedSymbolLiteral.Symbol);
// Relocation occurs at the cursorEntry + offset relative to that cursor
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedSymbolLiteral.Offset);
// Generate a literal so we can place it
Literal<uint64_t> Lit(Pointer);
place(&Lit);
DataIndex += sizeof(Reloc->NamedSymbolLiteral);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE: {
uint64_t Pointer = reinterpret_cast<uint64_t>(EmitterCTX->ThunkHandler->LookupThunk(Reloc->NamedThunkMove.Symbol));
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->NamedThunkMove.Offset);
LoadConstant(vixl::aarch64::XRegister(Reloc->NamedThunkMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->NamedThunkMove);
break;
}
case FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE: {
// XXX: Reenable once the JIT Object Cache is upstream
// XXX: Should spin the relocation list, create a list of guest RIP moves, and ask for them all once, reduces lock contention.
uint64_t Pointer = ~0ULL; // EmitterCTX->JITObjectCache->FindRelocatedRIP(Reloc->GuestRIPMove.GuestRIP);
if (Pointer == ~0ULL) {
return false;
}
// Relocation occurs at the cursorEntry + offset relative to that cursor.
GetBuffer()->SetCursorOffset(CursorEntry + Reloc->GuestRIPMove.Offset);
LoadConstant(vixl::aarch64::XRegister(Reloc->GuestRIPMove.RegisterIndex), Pointer, true);
DataIndex += sizeof(Reloc->GuestRIPMove);
break;
}
}
}
return true;
}
}
@@ -46,10 +46,10 @@ DEF_OP(CallbackReturn) {
ResetStack();
// We can now lower the ref counter again
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.SignalHandlerRefCountPointer)));
ldr(w2, MemOperand(x0));
ldr(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
sub(w2, w2, 1);
str(w2, MemOperand(x0));
str(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
// We need to adjust an additional 8 bytes to get back to the original "misaligned" RSP state
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
+22 -77
View File
@@ -367,32 +367,11 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
void Arm64JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
}
Arm64JITCore::CodeBuffer Arm64JITCore::AllocateNewCodeBuffer(size_t Size) {
CodeBuffer Buffer;
Buffer.Size = Size;
Buffer.Ptr = static_cast<uint8_t*>(
FEXCore::Allocator::mmap(nullptr,
Buffer.Size,
PROT_READ | PROT_WRITE | PROT_EXEC,
MAP_PRIVATE | MAP_ANONYMOUS,
-1, 0));
LOGMAN_THROW_A_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
Dispatcher->RegisterCodeBuffer(Buffer.Ptr, Buffer.Size);
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
return Buffer;
}
void Arm64JITCore::FreeCodeBuffer(CodeBuffer Buffer) {
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
Dispatcher->RemoveCodeBuffer(Buffer.Ptr);
}
Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
: Arm64Emitter(ctx, 0)
, CTX {ctx}
, ThreadState {Thread} {
: CPUBackend(Thread, 1024 * 1024 * 16, 1024 * 1024 * 128)
, Arm64Emitter(ctx, 0)
, CTX {ctx} {
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
#if DEBUG
@@ -466,18 +445,11 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
// Fill in the fallback handlers
InterpreterOps::FillFallbackIndexPointers(Pointers.FallbackHandlerPointers);
// Thread Specific
Pointers.SignalHandlerRefCountPointer = reinterpret_cast<uint64_t>(&Dispatcher->SignalHandlerRefCounter);
}
// Can't allocate a code buffer until after dispatcher is created
InitialCodeBuffer = AllocateNewCodeBuffer(Arm64JITCore::INITIAL_CODE_SIZE);
*GetBuffer() = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
// Must be done after Dispatcher init
SetAllowAssembler(true);
EmitDetectionString();
CurrentCodeBuffer = &InitialCodeBuffer;
ClearCache();
}
void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
@@ -521,54 +493,14 @@ void Arm64JITCore::EmitDetectionString() {
void Arm64JITCore::ClearCache() {
// Get the backing code buffer
auto Buffer = GetBuffer();
if (Dispatcher->SignalHandlerRefCounter == 0) {
if (!CodeBuffers.empty()) {
// If we have more than one code buffer we are tracking then walk them and delete
// This is a cleanup step
for (auto CodeBuffer : CodeBuffers) {
FreeCodeBuffer(CodeBuffer);
}
CodeBuffers.clear();
// Set the current code buffer to the initial
*Buffer = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
CurrentCodeBuffer = &InitialCodeBuffer;
}
if (CurrentCodeBuffer->Size == MAX_CODE_SIZE) {
// Rewind to the start of the code cache start
Buffer->Reset();
}
else {
FreeCodeBuffer(InitialCodeBuffer);
// Resize the code buffer and reallocate our code size
InitialCodeBuffer.Size *= 1.5;
InitialCodeBuffer.Size = std::min(InitialCodeBuffer.Size, MAX_CODE_SIZE);
InitialCodeBuffer = AllocateNewCodeBuffer(InitialCodeBuffer.Size);
*Buffer = vixl::CodeBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
}
}
else {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Allocate some new code buffers that we can switch over to instead
auto NewCodeBuffer = Arm64JITCore::AllocateNewCodeBuffer(Arm64JITCore::INITIAL_CODE_SIZE);
EmplaceNewCodeBuffer(NewCodeBuffer);
*Buffer = vixl::CodeBuffer(NewCodeBuffer.Ptr, NewCodeBuffer.Size);
}
auto CodeBuffer = GetEmptyCodeBuffer();
*GetBuffer() = vixl::CodeBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
EmitDetectionString();
}
Arm64JITCore::~Arm64JITCore() {
for (auto CodeBuffer : CodeBuffers) {
FreeCodeBuffer(CodeBuffer);
}
CodeBuffers.clear();
FreeCodeBuffer(InitialCodeBuffer);
}
IR::PhysicalRegister Arm64JITCore::GetPhys(IR::NodeID Node) const {
@@ -887,6 +819,19 @@ uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuSt
return HostCode;
}
void Arm64JITCore::ResetStack() {
if (SpillSlots == 0)
return;
if (IsImmAddSub(SpillSlots * 16)) {
add(sp, sp, SpillSlots * 16);
} else {
// Too big to fit in a 12bit immediate
LoadConstant(x0, SpillSlots * 16);
add(sp, sp, x0);
}
}
std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
return std::make_unique<Arm64JITCore>(ctx, Thread);
}
+64 -30
View File
@@ -37,11 +37,6 @@ using namespace vixl::aarch64;
class Arm64JITCore final : public CPUBackend, public Arm64Emitter {
public:
struct CodeBuffer {
uint8_t *Ptr;
size_t Size;
};
explicit Arm64JITCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread);
~Arm64JITCore() override;
@@ -60,7 +55,8 @@ public:
void ClearCache() override;
static constexpr size_t INITIAL_CODE_SIZE = 1024 * 1024 * 16;
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
// We don't want to move above 128MB atm because that means we will have to encode longer jumps
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const override {
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher);
@@ -73,10 +69,8 @@ public:
private:
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
Label *PendingTargetLabel;
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *ThreadState;
FEXCore::IR::IRListView const *IR;
uint64_t Entry;
@@ -147,28 +141,6 @@ private:
vixl::aarch64::Decoder Decoder;
#endif
void EmplaceNewCodeBuffer(CodeBuffer Buffer) {
CurrentCodeBuffer = &CodeBuffers.emplace_back(Buffer);
}
void FreeCodeBuffer(CodeBuffer Buffer);
// This is the initial code buffer that we will fall back to
// In a program without signals and code clearing, we will typically
// only have this code buffer
CodeBuffer InitialCodeBuffer{};
// This is the array of /additional/ code buffers that we may need to allocate
// Allocation only occurs when we've hit signals and need to clear code cache
// For code safety we can't delete code buffers until outside of all signals
std::vector<CodeBuffer> CodeBuffers{};
// This is the current code buffer that we are tracking
CodeBuffer *CurrentCodeBuffer{};
// We don't want to mvoe above 128MB atm because that means we will have to encode longer jumps
static constexpr size_t MAX_CODE_SIZE = 1024 * 1024 * 128;
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096 * 2;
#if DEBUG
vixl::aarch64::Disassembler Disasm;
#endif
@@ -180,6 +152,68 @@ private:
IR::RegisterAllocationPass *RAPass;
IR::RegisterAllocationData *RAData;
void ResetStack();
/**
* @name Relocations
* @{ */
uint64_t GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
/**
* @brief A literal pair relocation object for named symbol literals
*/
struct NamedSymbolLiteralPair {
Literal<uint64_t> Lit;
Relocation MoveABI{};
};
/**
* @brief Inserts a thunk relocation
*
* @param Reg - The GPR to move the thunk handler in to
* @param Sum - The hash of the thunk
*/
void InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const IR::SHA256Sum &Sum);
/**
* @brief Inserts a guest GPR move relocation
*
* @param Reg - The GPR to move the guest RIP in to
* @param Constant - The guest RIP that will be relocated
*/
void InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Constant);
/**
* @brief Inserts a named symbol as a literal in memory
*
* Need to use `PlaceNamedSymbolLiteral` with the return value to place the literal in the desired location
*
* @param Op The named symbol to place
*
* @return A temporary `NamedSymbolLiteralPair`
*/
NamedSymbolLiteralPair InsertNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op);
/**
* @brief Place the named symbol literal relocation in memory
*
* @param Lit - Which literal to place
*/
void PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit);
std::vector<FEXCore::CPU::Relocation> Relocations;
///< Relocation code loading
bool ApplyRelocations(uint64_t GuestEntry, uint64_t CodeEntry, uint64_t CursorEntry, size_t NumRelocations, const char* EntryRelocations);
/** @} */
uint32_t SpillSlots{};
/**
* @brief Current guest RIP entrypoint
*/
uint64_t GuestEntry{};
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
void RegisterALUHandlers();
@@ -53,7 +53,7 @@ DEF_OP(CallbackReturn) {
}
// Make sure to adjust the refcounter so we don't clear the cache now
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.X86.SignalHandlerRefCountPointer)], 1);
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)], 1);
// We need to adjust an additional 8 bytes to get back to the original "misaligned" RSP state
add(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])], 8);
+10 -76
View File
@@ -55,26 +55,6 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) {
namespace FEXCore::CPU {
CodeBuffer AllocateNewCodeBuffer(FEXCore::Context::Context *CTX, size_t Size) {
CodeBuffer Buffer;
Buffer.Size = Size;
Buffer.Ptr = static_cast<uint8_t*>(
FEXCore::Allocator::mmap(nullptr,
Buffer.Size,
PROT_READ | PROT_WRITE | PROT_EXEC,
MAP_PRIVATE | MAP_ANONYMOUS,
-1, 0));
LOGMAN_THROW_A_FMT(Buffer.Ptr != reinterpret_cast<uint8_t*>(~0ULL), "Couldn't allocate code buffer");
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
return Buffer;
}
void FreeCodeBuffer(CodeBuffer Buffer) {
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
}
void X86JITCore::PushRegs() {
sub(rsp, 16 * RAXMM_x.size());
for (size_t i = 0; i < RAXMM_x.size(); ++i) {
@@ -320,14 +300,10 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
void X86JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
}
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, CodeBuffer Buffer)
: CodeGenerator(Buffer.Size, Buffer.Ptr, nullptr)
, CTX {ctx}
, ThreadState {Thread}
, InitialCodeBuffer {Buffer}
{
CurrentCodeBuffer = &InitialCodeBuffer;
EmitDetectionString();
X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
: CPUBackend(Thread, 1024 * 1024 * 16, 1024 * 1024 * 256)
, CodeGenerator(0, this, nullptr) // this is not used here
, CTX {ctx} {
RAPass = Thread->PassManager->GetPass<IR::RegisterAllocationPass>("RA");
@@ -384,10 +360,10 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
// Fill in the fallback handlers
InterpreterOps::FillFallbackIndexPointers(Pointers.FallbackHandlerPointers);
// Thread Specific
Pointers.SignalHandlerRefCountPointer = reinterpret_cast<uint64_t>(&Dispatcher->SignalHandlerRefCounter);
}
// Must be done after Dispatcher init
ClearCache();
}
void X86JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
@@ -412,13 +388,7 @@ void X86JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
}
X86JITCore::~X86JITCore() {
for (auto CodeBuffer : CodeBuffers) {
FreeCodeBuffer(CodeBuffer);
}
CodeBuffers.clear();
FreeCodeBuffer(InitialCodeBuffer);
}
void X86JITCore::EmitDetectionString() {
@@ -429,44 +399,8 @@ void X86JITCore::EmitDetectionString() {
}
void X86JITCore::ClearCache() {
if (Dispatcher->SignalHandlerRefCounter == 0) {
if (!CodeBuffers.empty()) {
// If we have more than one code buffer we are tracking then walk them and delete
// This is a cleanup step
for (auto CodeBuffer : CodeBuffers) {
FreeCodeBuffer(CodeBuffer);
}
CodeBuffers.clear();
// Set the current code buffer to the initial
setNewBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
CurrentCodeBuffer = &InitialCodeBuffer;
}
if (CurrentCodeBuffer->Size == MAX_CODE_SIZE) {
// Rewind to the start of the code cache start
reset();
}
else {
FreeCodeBuffer(InitialCodeBuffer);
// Resize the code buffer and reallocate our code size
CurrentCodeBuffer->Size *= 1.5;
CurrentCodeBuffer->Size = std::min(CurrentCodeBuffer->Size, MAX_CODE_SIZE);
InitialCodeBuffer = AllocateNewCodeBuffer(CTX, CurrentCodeBuffer->Size);
setNewBuffer(InitialCodeBuffer.Ptr, InitialCodeBuffer.Size);
}
}
else {
// We have signal handlers that have generated code
// This means that we can not safely clear the code at this point in time
// Allocate some new code buffers that we can switch over to instead
auto NewCodeBuffer = AllocateNewCodeBuffer(CTX, X86JITCore::INITIAL_CODE_SIZE);
EmplaceNewCodeBuffer(NewCodeBuffer);
setNewBuffer(NewCodeBuffer.Ptr, NewCodeBuffer.Size);
}
auto CodeBuffer = GetEmptyCodeBuffer();
setNewBuffer(CodeBuffer->Ptr, CodeBuffer->Size);
EmitDetectionString();
}
@@ -830,7 +764,7 @@ uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateF
}
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
return std::make_unique<X86JITCore>(ctx, Thread, AllocateNewCodeBuffer(ctx, X86JITCore::INITIAL_CODE_SIZE));
return std::make_unique<X86JITCore>(ctx, Thread);
}
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX) {
@@ -25,13 +25,6 @@ using namespace Xbyak;
#include <tuple>
namespace FEXCore::CPU {
struct CodeBuffer {
uint8_t *Ptr;
size_t Size;
};
[[nodiscard]] CodeBuffer AllocateNewCodeBuffer(size_t Size);
void FreeCodeBuffer(CodeBuffer Buffer);
// Temp registers
// rax, rcx, rdx, rsi, r8, r9,
@@ -58,8 +51,7 @@ const std::array<Xbyak::Xmm, 11> RAXMM_x = { xmm1, xmm2, xmm3, xmm4, xmm5, xmm6
class X86JITCore final : public CPUBackend, public Xbyak::CodeGenerator {
public:
explicit X86JITCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread,
CodeBuffer Buffer);
FEXCore::Core::InternalThreadState *Thread);
~X86JITCore() override;
[[nodiscard]] std::string GetName() override { return "JIT"; }
@@ -150,9 +142,7 @@ private:
Label* PendingTargetLabel{};
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *ThreadState;
FEXCore::IR::IRListView const *IR;
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
uint64_t Entry;
std::unordered_map<IR::NodeID, Label> JumpTargets;
@@ -210,27 +200,11 @@ private:
bool GetSamplingData {true};
#endif
void EmplaceNewCodeBuffer(CodeBuffer Buffer) {
CurrentCodeBuffer = &CodeBuffers.emplace_back(Buffer);
}
static uint64_t ExitFunctionLink(X86JITCore* code, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
void EmitDetectionString();
// This is the initial code buffer that we will fall back to
// In a program without signals and code clearing, we will typically
// only have this code buffer
CodeBuffer InitialCodeBuffer{};
// This is the array of /additional/ code buffers that we may need to allocate
// Allocation only occurs when we've hit signals and need to clear code cache
// For code safety we can't delete code buffers until outside of all signals
std::vector<CodeBuffer> CodeBuffers{};
// This is the current code buffer that we are tracking
CodeBuffer *CurrentCodeBuffer{};
uint32_t SpillSlots{};
using SetCC = void (X86JITCore::*)(const Operand& op);
using CMovCC = void (X86JITCore::*)(const Reg& reg, const Operand& op);
+38 -4
View File
@@ -11,6 +11,8 @@ $end_info$
#include <cstdint>
#include <string>
#include <memory>
#include <vector>
namespace FEXCore {
@@ -23,6 +25,7 @@ namespace Core {
struct DebugData;
struct ThreadState;
struct CpuStateFrame;
struct InternalThreadState;
}
namespace CodeSerialize {
@@ -30,13 +33,22 @@ namespace CodeSerialize {
}
namespace CPU {
class InterpreterCore;
class JITCore;
class LLVMCore;
class Dispatcher;
class CPUBackend {
public:
virtual ~CPUBackend() = default;
struct CodeBuffer {
uint8_t *Ptr;
size_t Size;
};
/**
* @param InitialCodeSize - Initial size for the code buffers
* @param MaxCodeSize - Max size for the code buffers
*/
CPUBackend(FEXCore::Core::InternalThreadState *ThreadState, size_t InitialCodeSize, size_t MaxCodeSize);
virtual ~CPUBackend();
/**
* @return The name of this backend
*/
@@ -122,7 +134,29 @@ class LLVMCore;
JITCallback CallbackPtr{};
protected:
FEXCore::Core::InternalThreadState *ThreadState;
size_t InitialCodeSize, MaxCodeSize;
[[nodiscard]] CodeBuffer *GetEmptyCodeBuffer();
// This is the current code buffer that we are tracking
CodeBuffer *CurrentCodeBuffer{};
AsmDispatch DispatchPtr{};
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
private:
CodeBuffer AllocateNewCodeBuffer(size_t Size);
void FreeCodeBuffer(CodeBuffer Buffer);
void EmplaceNewCodeBuffer(CodeBuffer Buffer) {
CurrentCodeBuffer = &CodeBuffers.emplace_back(Buffer);
}
// This is the array of code buffers. Unless signals force us to keep more than
// buffer, there will be only one entry here
std::vector<CodeBuffer> CodeBuffers{};
};
}
+4 -3
View File
@@ -110,7 +110,6 @@ namespace FEXCore::Core {
uint64_t FallbackHandlerPointers[FallbackHandlerIndex::OPINDEX_MAX];
// Thread Specific
uint64_t SignalHandlerRefCountPointer{};
/**
* @name Dispatcher pointers
@@ -143,7 +142,6 @@ namespace FEXCore::Core {
uint64_t FallbackHandlerPointers[FallbackHandlerIndex::OPINDEX_MAX];
// Thread Specific
uint64_t SignalHandlerRefCountPointer{};
/**
* @name Dispatcher pointers
@@ -177,8 +175,11 @@ namespace FEXCore::Core {
* ARM64:
* - Bit 15: In syscall
* - Bit 14-0: Number of static registers spilled
*/
*/
uint64_t InSyscallInfo{};
uint32_t SignalHandlerRefCounter{};
InternalThreadState* Thread;
// Pointers that the JIT needs to load to remove relocations
@@ -72,7 +72,7 @@ namespace FEXCore::Core {
};
struct InternalThreadState {
FEXCore::Core::CpuStateFrame* CurrentFrame = &BaseFrameState;
FEXCore::Core::CpuStateFrame* const CurrentFrame = &BaseFrameState;
struct {
std::atomic_bool Running {false};
+2 -1
View File
@@ -63,7 +63,8 @@ namespace HostFactory {
}
HostCore::HostCore(FEXCore::Context::Context* CTX, FEXCore::Core::InternalThreadState *Thread, bool Fallback)
: CodeGenerator(4096) {
: CPUBackend(Thread, 0, 0)
, CodeGenerator(4096) {
FEXCore::Context::RegisterHostSignalHandler(CTX, SIGSEGV,
[](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
auto InternalThread = Thread;