Backends: Rework some of the interfaces, unified backend dispatch

This commit is contained in:
Stefanos Kornilios Misis Poiitidis authored and Parallels committed 2022-06-22 22:24:42 +03:00
1 parent 917e69b021
commit 4dbcd34548
19 files changed
+352 -292

No files matched your search

+1
View File
@@ -351,6 +351,7 @@ namespace FEXCore::Context {
std::shared_mutex CustomIRMutex;
std::unordered_map<uint64_t, std::tuple<std::function<void(uintptr_t Entrypoint, FEXCore::IR::IREmitter *)>, void *, void *>> CustomIRHandlers;
FEXCore::CPU::CPUBackendFeatures BackendFeatures;
FEXCore::CPU::DispatcherConfig DispatcherConfig;
};
+14 -9
View File
@@ -200,16 +200,16 @@ namespace FEXCore::Context {
#ifdef INTERPRETER_ENABLED
case FEXCore::Config::CONFIG_INTERPRETER:
FEXCore::CPU::InitializeInterpreterSignalHandlers(this);
FEXCore::CPU::GetInterpreterDispatcherConfig(DispatcherConfig);
BackendFeatures = FEXCore::CPU::GetInterpreterBackendFeatures();
break;
#endif
case FEXCore::Config::CONFIG_IRJIT:
#if (_M_X86_64 && JIT_X86_64)
FEXCore::CPU::InitializeX86JITSignalHandlers(this);
FEXCore::CPU::GetX86JITDispatcherConfig(DispatcherConfig);
BackendFeatures = FEXCore::CPU::GetX86JITBackendFeatures();
#elif (_M_ARM_64 && JIT_ARM64)
FEXCore::CPU::InitializeArm64JITSignalHandlers(this);
FEXCore::CPU::GetArm64JITDispatcherConfig(DispatcherConfig);
BackendFeatures = FEXCore::CPU::GetArm64JITBackendFeatures();
#else
ERROR_AND_DIE_FMT("FEXCore has been compiled without a viable JIT core");
#endif
@@ -222,6 +222,8 @@ namespace FEXCore::Context {
break;
}
DispatcherConfig.StaticRegisterAllocation = Config.StaticRegisterAllocation && BackendFeatures.SupportsStaticRegisterAllocation;
#if (_M_X86_64)
Dispatcher = FEXCore::CPU::Dispatcher::CreateX86(this, DispatcherConfig);
#elif (_M_ARM_64)
@@ -527,7 +529,7 @@ namespace FEXCore::Context {
Thread->CTX = this;
bool DoSRA = Config.StaticRegisterAllocation && DispatcherConfig.SupportsStaticRegisterAllocation;
bool DoSRA = DispatcherConfig.StaticRegisterAllocation;
Thread->PassManager->AddDefaultPasses(this, Config.Core == FEXCore::Config::CONFIG_IRJIT, DoSRA);
Thread->PassManager->AddDefaultValidationPasses();
@@ -943,7 +945,7 @@ namespace FEXCore::Context {
}
// Attempt to get the CPU backend to compile this code
return {
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData),
.CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, IRList, DebugData, RAData, GetGdbServerStatus()),
.IRData = IRList,
.DebugData = DebugData,
.RAData = RAData,
@@ -999,22 +1001,25 @@ namespace FEXCore::Context {
// The core managed to compile the code.
if (Config.BlockJITNaming()) {
auto FragmentBasePtr = reinterpret_cast<uint8_t *>(CodePtr);
if (DebugData) {
auto GuestRIPLookup = this->SyscallHandler->LookupAOTIRCacheEntry(GuestRIP);
if (DebugData->Subblocks.size()) {
for (auto& Subblock: DebugData->Subblocks) {
auto BlockBasePtr = FragmentBasePtr + Subblock.HostCodeOffset;
if (GuestRIPLookup.Entry) {
Symbols.Register(CodePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename, GuestRIP - GuestRIPLookup.Offset);
Symbols.Register(BlockBasePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename, GuestRIP - GuestRIPLookup.Offset);
} else {
Symbols.Register((void*)Subblock.HostCodeStart, GuestRIP, Subblock.HostCodeSize);
Symbols.Register(BlockBasePtr, GuestRIP, Subblock.HostCodeSize);
}
}
} else {
if (GuestRIPLookup.Entry) {
Symbols.Register(CodePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename, GuestRIP - GuestRIPLookup.Offset);
Symbols.Register(FragmentBasePtr, DebugData->HostCodeSize, GuestRIPLookup.Entry->Filename, GuestRIP - GuestRIPLookup.Offset);
} else {
Symbols.Register(CodePtr, GuestRIP, DebugData->HostCodeSize);
Symbols.Register(FragmentBasePtr, GuestRIP, DebugData->HostCodeSize);
}
}
}
@@ -30,6 +30,9 @@
#include <sys/syscall.h>
#include <unistd.h>
#define STATE_PTR(STATE_TYPE, FIELD) \
MemOperand(STATE, offsetof(FEXCore::Core::STATE_TYPE, FIELD))
namespace FEXCore::CPU {
using namespace vixl;
@@ -38,9 +41,9 @@ using namespace vixl::aarch64;
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
#define STATE x28
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &config)
: FEXCore::CPU::Dispatcher(ctx), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
SRAEnabled = config.SupportsStaticRegisterAllocation && ctx->Config.StaticRegisterAllocation;
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
: FEXCore::CPU::Dispatcher(ctx), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE)
, config(config) {
SetAllowAssembler(true);
DispatchPtr = GetCursorAddress<AsmDispatch>();
@@ -56,7 +59,6 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
Literal l_CTX {reinterpret_cast<uintptr_t>(CTX)};
Literal l_Sleep {reinterpret_cast<uint64_t>(SleepThread)};
Literal l_CompileBlock {GetCompileBlockPtr()};
Literal l_ExitFunctionLink {config.ExitFunctionLink};
// Push all the register we need to save
PushCalleeSavedRegisters();
@@ -69,11 +71,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
// regardless of where we were in the stack
add(x0, sp, 0);
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)));
str(x0, STATE_PTR(CpuStateFrame, ReturningStackLocation));
AbsoluteLoopTopAddressFillSRA = GetCursorAddress<uint64_t>();
if (SRAEnabled) {
if (config.StaticRegisterAllocation) {
FillStaticRegs();
}
@@ -90,11 +92,11 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
// Load in our RIP
// Don't modify x2 since it contains our RIP once the block doesn't exist
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
ldr(x2, STATE_PTR(CpuStateFrame, State.rip));
auto RipReg = x2;
// L1 Cache
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)));
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(x3, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x3, Shift::LSL, 4));
@@ -102,18 +104,14 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
cmp(x0, RipReg);
b(&FullLookup, Condition::ne);
if (!config.InterpreterDispatch) {
br(x3);
} else {
b(&CallBlock);
}
br(x3);
// L1C check failed, do a full lookup
bind(&FullLookup);
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L2Pointer)));
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
// Mask the address by the virtual address size so we can check for aliases
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
@@ -155,46 +153,21 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
// If we've made it here then we have a real compiled block
{
// update L1 cache
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)));
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
and_(x1, RipReg, LookupCache::L1_ENTRIES_MASK);
add(x0, x0, Operand(x1, Shift::LSL, 4));
stp(x3, x2, MemOperand(x0));
// Jump to the block
if (!config.InterpreterDispatch) {
br(x3);
} else {
bind(&CallBlock);
mov(x0, STATE);
mov(x1, x3);
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Interpreter.FragmentExecuter)));
blr(x3);
if (CTX->GetGdbServerStatus()) {
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
ldr(x0, &l_CTX);
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
// If the value == 0 then branch to the top
cbz(x0, &LoopTop);
// Else we need to pause now
b(&ThreadPauseHandler);
} else {
// Unconditionally loop to the top
// We will only stop on error when compiling a block or signal
b(&LoopTop);
}
}
br(x3);
}
}
{
bind(&ExitSpillSRA);
ThreadStopHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
if (SRAEnabled)
if (config.StaticRegisterAllocation)
SpillStaticRegs();
ThreadStopHandlerAddress = GetCursorAddress<uint64_t>();
@@ -209,7 +182,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
constexpr bool SignalSafeCompile = true;
{
ExitFunctionLinkerAddress = GetCursorAddress<uint64_t>();
if (SRAEnabled)
if (config.StaticRegisterAllocation)
SpillStaticRegs();
if (SignalSafeCompile) {
@@ -234,7 +207,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
mov(x0, STATE);
mov(x1, lr);
ldr(x3, &l_ExitFunctionLink);
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
blr(x3);
if (SignalSafeCompile) {
@@ -255,7 +228,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
mov(x0, x4);
}
if (SRAEnabled)
if (config.StaticRegisterAllocation)
FillStaticRegs();
br(x0);
}
@@ -264,7 +237,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
{
bind(&NoBlock);
if (SRAEnabled)
if (config.StaticRegisterAllocation)
SpillStaticRegs();
if (SignalSafeCompile) {
@@ -311,7 +284,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
add(sp, sp, 16);
}
if (SRAEnabled)
if (config.StaticRegisterAllocation)
FillStaticRegs();
b(&LoopTop);
@@ -330,7 +303,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
// Needs to be distinct from the SignalHandlerReturnAddress
UnimplementedInstructionAddress = GetCursorAddress<uint64_t>();
if (SRAEnabled)
if (config.StaticRegisterAllocation)
SpillStaticRegs();
hlt(0);
@@ -341,17 +314,17 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
// Needs to be distinct from the SignalHandlerReturnAddress
OverflowExceptionInstructionAddress = GetCursorAddress<uint64_t>();
if (SRAEnabled)
if (config.StaticRegisterAllocation)
SpillStaticRegs();
LoadConstant(w1, 1);
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)));
strb(w1, STATE_PTR(CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException));
LoadConstant(w1, X86State::X86_TRAPNO_OF);
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)));
str(w1, STATE_PTR(CpuStateFrame, SynchronousFaultData.TrapNo));
LoadConstant(w1, 0x80);
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)));
str(w1, STATE_PTR(CpuStateFrame, SynchronousFaultData.si_code));
LoadConstant(x1, 0);
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)));
str(w1, STATE_PTR(CpuStateFrame, SynchronousFaultData.err_code));
// hlt/udf = SIGILL
// brk = SIGTRAP
@@ -362,7 +335,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
{
ThreadPauseHandlerAddressSpillSRA = GetCursorAddress<uint64_t>();
if (SRAEnabled)
if (config.StaticRegisterAllocation)
SpillStaticRegs();
bind(&ThreadPauseHandler);
@@ -406,27 +379,27 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
mov(STATE, x0);
// Make sure to adjust the refcounter so we don't clear the cache now
ldr(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
ldr(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
add(w2, w2, 1);
str(w2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)));
str(w2, STATE_PTR(CpuStateFrame, SignalHandlerRefCounter));
// Now push the callback return trampoline to the guest stack
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
LoadConstant(x0, CTX->X86CodeGen.CallbackReturn);
ldr(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
ldr(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
sub(x2, x2, 16);
str(x2, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])));
str(x2, STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
// Store the trampoline to the guest stack
// Guest stack is now correctly misaligned after a regular call instruction
str(x0, MemOperand(x2));
// Store RIP to the context state
str(x1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
str(x1, STATE_PTR(CpuStateFrame, State.rip));
// load static regs
if (SRAEnabled)
if (config.StaticRegisterAllocation)
FillStaticRegs();
// Now go back to the regular dispatcher loop
@@ -438,7 +411,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
PushDynamicRegsAndLR();
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUDIV)));
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUDIV));
SpillStaticRegs();
blr(x3);
@@ -457,7 +430,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
PushDynamicRegsAndLR();
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LDIV)));
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LDIV));
SpillStaticRegs();
blr(x3);
@@ -476,7 +449,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
PushDynamicRegsAndLR();
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LUREM)));
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LUREM));
SpillStaticRegs();
blr(x3);
@@ -495,7 +468,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
PushDynamicRegsAndLR();
ldr(x3, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.AArch64.LREM)));
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.AArch64.LREM));
SpillStaticRegs();
blr(x3);
@@ -512,7 +485,6 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
place(&l_CTX);
place(&l_Sleep);
place(&l_CompileBlock);
place(&l_ExitFunctionLink);
FinalizeCode();
@@ -530,6 +502,62 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfi
}
}
size_t Arm64Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
*GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxGDBPauseCheckSize);
aarch64::Label RunBlock;
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(FEXCore::Context::Context::Config.RunningMode) == 4, "This is expected to be size of 4");
ldr(x0, STATE_PTR(CpuStateFrame, Thread)); // Get thread
ldr(x0, MemOperand(x0, offsetof(FEXCore::Core::InternalThreadState, CTX))); // Get Context
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
// If the value == 0 then we don't need to stop
cbz(w0, &RunBlock);
{
// Make sure RIP is syncronized to the context
LoadConstant(x0, GuestRIP);
str(x0, STATE_PTR(CpuStateFrame, State.rip));
// Stop the thread
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
br(x0);
}
bind(&RunBlock);
FinalizeCode();
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, GetBuffer()->GetCursorOffset());
return GetBuffer()->GetCursorOffset();
}
size_t Arm64Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
LOGMAN_THROW_A_FMT(!config.StaticRegisterAllocation, "GenerateInterpreterTrampoline dispatcher does not support SRA");
*GetBuffer() = vixl::CodeBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
aarch64::Label InlineIRData;
mov(x0, STATE);
adr(x1, &InlineIRData);
ldr(x3, STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
blr(x3);
ldr(x0, STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
br(x0);
bind(&InlineIRData);
FinalizeCode();
vixl::aarch64::CPU::EnsureIAndDCacheCoherency(CodeBuffer, GetBuffer()->GetCursorOffset());
return GetBuffer()->GetCursorOffset();
}
void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
for(int i = 0; i < SRA64.size(); i++) {
if (IgnoreMask & (1U << SRA64[i].GetCode())) {
@@ -567,7 +595,7 @@ void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thr
}
}
std::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::Context *CTX, DispatcherConfig &Config) {
std::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
return std::make_unique<Arm64Dispatcher>(CTX, Config);
}
@@ -15,8 +15,10 @@ namespace FEXCore::CPU {
class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
public:
Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &config);
Arm64Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
protected:
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
@@ -27,6 +29,7 @@ class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
uint64_t LDIVHandlerAddress{};
uint64_t LUREMHandlerAddress{};
uint64_t LREMHandlerAddress{};
DispatcherConfig config;
};
}
@@ -26,9 +26,7 @@ struct Context;
namespace FEXCore::CPU {
struct DispatcherConfig {
bool InterpreterDispatch = false;
uintptr_t ExitFunctionLink = 0;
bool SupportsStaticRegisterAllocation = false;
bool StaticRegisterAllocation = false;
};
class Dispatcher {
@@ -67,8 +65,15 @@ public:
virtual void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) = 0;
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, DispatcherConfig &Config);
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, DispatcherConfig &Config);
// These are across all arches for now
static constexpr size_t MaxGDBPauseCheckSize = 128;
static constexpr size_t MaxInterpreterTrampolineSize = 128;
virtual size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) = 0;
virtual size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) = 0;
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, const DispatcherConfig &Config);
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
DispatchPtr(Frame);
@@ -20,17 +20,20 @@
#include <sys/mman.h>
#include <xbyak/xbyak.h>
#define STATE_PTR(STATE_TYPE, FIELD) \
[STATE + offsetof(FEXCore::Core::STATE_TYPE, FIELD)]
namespace FEXCore::CPU {
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
#define STATE r14
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &config)
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config)
: Dispatcher(ctx)
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE,
FEXCore::Allocator::mmap(nullptr, MAX_DISPATCHER_CODE_SIZE, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0),
nullptr) {
LOGMAN_THROW_A_FMT(!config.SupportsStaticRegisterAllocation, "X86 dispatcher does not support SRA");
LOGMAN_THROW_A_FMT(!config.StaticRegisterAllocation, "X86 dispatcher does not support SRA");
using namespace Xbyak;
using namespace Xbyak::util;
@@ -80,11 +83,10 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &c
// Save this stack pointer so we can cleanly shutdown the emulation with a long jump
// regardless of where we were in the stack
mov(qword [rdi + offsetof(FEXCore::Core::CpuStateFrame, ReturningStackLocation)], rsp);
mov(qword STATE_PTR(CpuStateFrame, ReturningStackLocation), rsp);
Label LoopTop;
Label FullLookup;
Label CallBlock;
Label NoBlock;
Label ExitBlock;
Label ThreadPauseHandler;
@@ -94,26 +96,21 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &c
{
// Load our RIP
mov(rdx, qword [STATE + offsetof(FEXCore::Core::CPUState, rip)]);
mov(rdx, qword STATE_PTR(CPUState, rip));
// L1 Cache
mov(r13, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)]);
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
mov(rax, rdx);
and_(rax, LookupCache::L1_ENTRIES_MASK);
shl(rax, 4);
cmp(qword[r13 + rax + 8], rdx);
cmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, GuestCode)], rdx);
jne(FullLookup);
if (!config.InterpreterDispatch) {
jmp(qword[r13 + rax + 0]);
} else {
mov(rax, qword[r13 + rax + 0]);
jmp(CallBlock);
}
jmp(qword[r13 + rax + offsetof(FEXCore::LookupCache::LookupCacheEntry, HostCode)]);
L(FullLookup);
mov(r13, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L2Pointer)]);
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L2Pointer));
// Full lookup
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
@@ -145,7 +142,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &c
je(NoBlock);
// Update L1
mov(r13, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L1Pointer)]);
mov(r13, qword STATE_PTR(CpuStateFrame, Pointers.Common.L1Pointer));
mov(rcx, rdx);
and_(rcx, LookupCache::L1_ENTRIES_MASK);
shl(rcx, 1);
@@ -153,31 +150,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &c
mov(qword[r13 + rcx*8 + 0], rax);
// Real block if we made it here
if (!config.InterpreterDispatch) {
jmp(rax);
} else {
L(CallBlock);
mov(rdi, STATE);
mov(rsi, rax);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Interpreter.FragmentExecuter)]);
if (CTX->GetGdbServerStatus()) {
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
mov(rax, qword [STATE + (offsetof(FEXCore::Core::InternalThreadState, CTX))]);
// If the value == 0 then branch to the top
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
je(LoopTop);
// Else we need to pause now
jmp(ThreadPauseHandler);
ud2();
}
else {
jmp(LoopTop);
}
}
jmp(rax);
}
{
@@ -289,8 +262,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &c
mov(rdi, STATE);
mov(rsi, rax); // rax is set at the block end
mov(rax, config.ExitFunctionLink);
call(rax);
call(qword STATE_PTR(CpuStateFrame, Pointers.Common.ExitFunctionLink));
if (SignalSafeCompile) {
// Now restore the signal mask
@@ -347,7 +319,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &c
// XXX: XMM?
// Make sure to adjust the refcounter so we don't clear the cache now
add(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SignalHandlerRefCounter)], 1);
add(qword STATE_PTR(CpuStateFrame, SignalHandlerRefCounter), 1);
// Now push the callback return trampoline to the guest stack
// Guest will be misaligned because calling a thunk won't correct the guest's stack once we call the callback from the host
@@ -355,12 +327,12 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &c
// Store the trampoline to the guest stack
// Guest stack is now correctly misaligned after a regular call instruction
sub(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])], 16);
mov(rbx, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.gregs[X86State::REG_RSP])]);
sub(qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]), 16);
mov(rbx, qword STATE_PTR(CpuStateFrame, State.gregs[X86State::REG_RSP]));
mov(qword [rbx], rax);
// Store RIP to the context state
mov(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, State.rip)], rsi);
mov(qword STATE_PTR(CpuStateFrame, State.rip), rsi);
// Back to the loop top now
jmp(LoopTop);
@@ -388,10 +360,10 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &c
// int3 = SIGTRAP
// hlt = SIGSEGV
add(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)], 1);
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)], X86State::X86_TRAPNO_OF);
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)], 0);
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)], 0x80);
add(byte STATE_PTR(CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException), 1);
mov(dword STATE_PTR(CpuStateFrame, SynchronousFaultData.TrapNo), X86State::X86_TRAPNO_OF);
mov(dword STATE_PTR(CpuStateFrame, SynchronousFaultData.err_code), 0);
mov(dword STATE_PTR(CpuStateFrame, SynchronousFaultData.si_code), 0x80);
hlt();
}
@@ -433,6 +405,61 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &c
}
size_t X86Dispatcher::GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) {
using namespace Xbyak;
using namespace Xbyak::util;
setNewBuffer(CodeBuffer, MaxGDBPauseCheckSize);
Label RunBlock;
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
mov(rax, reinterpret_cast<uint64_t>(CTX));
// If the value == 0 then we don't need to stop
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
je(RunBlock);
{
// Make sure RIP is syncronized to the context
mov(rax, GuestRIP);
mov(qword STATE_PTR(CpuStateFrame, State.rip), rax);
// Stop the thread
mov(rax, qword STATE_PTR(CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
jmp(rax);
}
L(RunBlock);
ready();
return getSize();
}
size_t X86Dispatcher::GenerateInterpreterTrampoline(uint8_t *CodeBuffer) {
using namespace Xbyak;
using namespace Xbyak::util;
setNewBuffer(CodeBuffer, MaxInterpreterTrampolineSize);
Label InlineIRData;
mov(rdi, STATE);
lea(rsi, ptr[rip + InlineIRData]);
call(qword STATE_PTR(CpuStateFrame, Pointers.Interpreter.FragmentExecuter));
jmp(qword STATE_PTR(CpuStateFrame, Pointers.Common.DispatcherLoopTop));
L(InlineIRData);
ready();
return getSize();
}
X86Dispatcher::~X86Dispatcher() {
FEXCore::Allocator::munmap(top_, MAX_DISPATCHER_CODE_SIZE);
}
@@ -456,7 +483,7 @@ void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Threa
}
}
std::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::Context *CTX, DispatcherConfig &Config) {
std::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::Context *CTX, const DispatcherConfig &Config) {
return std::make_unique<X86Dispatcher>(CTX, Config);
}
@@ -17,8 +17,10 @@ namespace FEXCore::CPU {
class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator {
public:
X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &config);
X86Dispatcher(FEXCore::Context::Context *ctx, const DispatcherConfig &config);
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
size_t GenerateGDBPauseCheck(uint8_t *CodeBuffer, uint64_t GuestRIP) override;
size_t GenerateInterpreterTrampoline(uint8_t *CodeBuffer) override;
virtual ~X86Dispatcher() override;
};
@@ -8,6 +8,7 @@
#include <FEXCore/IR/IntrusiveIRList.h>
namespace FEXCore::CPU {
class Dispatcher;
class X86DispatchGenerator;
class Arm64DispatchGenerator;
@@ -20,7 +21,7 @@ using DestMapType = std::vector<uint32_t>;
class InterpreterCore final : public CPUBackend {
public:
explicit InterpreterCore(FEXCore::Context::Context *ctx,
explicit InterpreterCore(Dispatcher *Dispatch,
FEXCore::Core::InternalThreadState *Thread);
[[nodiscard]] std::string GetName() override { return "Interpreter"; }
@@ -28,7 +29,7 @@ public:
[[nodiscard]] void *CompileCode(uint64_t Entry,
FEXCore::IR::IRListView const *IR,
FEXCore::Core::DebugData *DebugData,
FEXCore::IR::RegisterAllocationData *RAData) override;
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
@@ -40,6 +41,7 @@ public:
private:
size_t BufferUsed;
Dispatcher *Dispatch;
};
template<typename T>
@@ -36,13 +36,12 @@ namespace FEXCore::IR {
namespace FEXCore::CPU {
class CPUBackend;
InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
InterpreterCore::InterpreterCore(Dispatcher *Dispatcher, FEXCore::Core::InternalThreadState *Thread)
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
, Dispatch(Dispatcher)
{
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
Interpreter.FragmentExecuter = reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR);
@@ -59,20 +58,35 @@ void InterpreterCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
#endif
}
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
auto Size = AlignUp(IR->GetInlineSize(), 16);
if ((BufferUsed + Size) > CurrentCodeBuffer->Size) {
const auto IRSize = AlignUp(IR->GetInlineSize(), 16);
const auto MaxSize = IRSize + Dispatcher::MaxInterpreterTrampolineSize + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
if ((BufferUsed + MaxSize) > CurrentCodeBuffer->Size) {
ThreadState->CTX->ClearCodeCache(ThreadState);
}
auto DestBuffer = CurrentCodeBuffer->Ptr + BufferUsed;
const auto BufferStart = CurrentCodeBuffer->Ptr + BufferUsed;;
auto DestBuffer = BufferStart;
if (GDBEnabled) {
const auto GDBSize = Dispatch->GenerateGDBPauseCheck(DestBuffer, Entry);
DestBuffer += GDBSize;
BufferUsed += GDBSize;
}
const auto TrampolineSize = Dispatch->GenerateInterpreterTrampoline(DestBuffer);
DestBuffer += TrampolineSize;
BufferUsed += TrampolineSize;
IR->Serialize(DestBuffer);
DestBuffer += IRSize;
BufferUsed += IRSize;
BufferUsed += Size;
return DestBuffer;
return BufferStart;
}
void InterpreterCore::ClearCache() {
@@ -82,16 +96,14 @@ void InterpreterCore::ClearCache() {
}
std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
return std::make_unique<InterpreterCore>(ctx, Thread);
return std::make_unique<InterpreterCore>(ctx->Dispatcher.get(), Thread);
}
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX) {
InterpreterCore::InitializeSignalHandlers(CTX);
}
void GetInterpreterDispatcherConfig(DispatcherConfig &config) {
config = DispatcherConfig {
.InterpreterDispatch = true
};
CPUBackendFeatures GetInterpreterBackendFeatures() {
return CPUBackendFeatures { };
}
}
@@ -17,6 +17,6 @@ struct DispatcherConfig;
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread);
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX);
void GetInterpreterDispatcherConfig(DispatcherConfig &config);
CPUBackendFeatures GetInterpreterBackendFeatures();
} // namespace FEXCore::CPU
@@ -26,7 +26,7 @@ void Arm64JITCore::InsertNamedThunkRelocation(vixl::aarch64::Register Reg, const
Relocation MoveABI{};
MoveABI.NamedThunkMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_NAMED_THUNK_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint64_t>();
auto CurrentCursor = GetCursorAddress<uint8_t *>();
MoveABI.NamedThunkMove.Offset = CurrentCursor - GuestEntry;
MoveABI.NamedThunkMove.Symbol = Sum;
MoveABI.NamedThunkMove.RegisterIndex = Reg.GetCode();
@@ -57,7 +57,7 @@ Arm64JITCore::NamedSymbolLiteralPair Arm64JITCore::InsertNamedSymbolLiteral(FEXC
void Arm64JITCore::PlaceNamedSymbolLiteral(NamedSymbolLiteralPair &Lit) {
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint64_t>();
auto CurrentCursor = GetCursorAddress<uint8_t *>();
Lit.MoveABI.NamedSymbolLiteral.Offset = CurrentCursor - GuestEntry;
place(&Lit.Lit);
@@ -68,7 +68,7 @@ void Arm64JITCore::InsertGuestRIPMove(vixl::aarch64::Register Reg, uint64_t Cons
Relocation MoveABI{};
MoveABI.GuestRIPMove.Header.Type = FEXCore::CPU::RelocationTypes::RELOC_GUEST_RIP_MOVE;
// Offset is the offset from the entrypoint of the block
auto CurrentCursor = GetCursorAddress<uint64_t>();
auto CurrentCursor = GetCursorAddress<uint8_t *>();
MoveABI.GuestRIPMove.Offset = CurrentCursor - GuestEntry;
MoveABI.GuestRIPMove.GuestRIP = Constant;
MoveABI.GuestRIPMove.RegisterIndex = Reg.GetCode();
+70 -85
View File
@@ -368,6 +368,56 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
}
}
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
auto GuestRip = record[1];
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
if (!HostCode) {
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
Frame->State.rip = GuestRip;
return Frame->Pointers.Common.DispatcherLoopTop;
}
uintptr_t branch = (uintptr_t)(record) - 8;
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
auto offset = HostCode/4 - branch/4;
if (IsInt26(offset)) {
// optimal case - can branch directly
// patch the code
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
emit.b(offset);
emit.FinalizeCode();
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
// Add de-linking handler
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
Literal l_BranchHost{LinkerAddress};
emit.ldr(x0, &l_BranchHost);
emit.blr(x0);
emit.place(&l_BranchHost);
emit.FinalizeCode();
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
});
} else {
// fallback case - do a soft-er link by patching the pointer
record[0] = HostCode;
// Add de-linking handler
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
record[0] = LinkerAddress;
});
}
return HostCode;
}
void Arm64JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
}
@@ -433,6 +483,8 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Arm64JITCore_ExitFunctionLink);
// Fill in the fallback handlers
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers);
@@ -613,7 +665,7 @@ bool Arm64JITCore::IsGPR(IR::NodeID Node) const {
return Class == IR::GPRClass || Class == IR::GPRFixedClass;
}
void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
using namespace aarch64;
JumpTargets.clear();
uint32_t SSACount = IR->GetSSACount();
@@ -628,9 +680,9 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
this->IR = IR;
// Fairly excessive buffer range to make sure we don't overflow
uint32_t BufferRange = SSACount * 16;
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
if ((GetCursorOffset() + BufferRange) > CurrentCodeBuffer->Size) {
ThreadState->CTX->ClearCodeCache(ThreadState);
CTX->ClearCodeCache(ThreadState);
}
// AAPCS64
@@ -653,31 +705,11 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
// X1-X3 = Temp
// X4-r18 = RA
GuestEntry = GetCursorAddress<uint64_t>();
GuestEntry = GetCursorAddress<uint8_t *>();
if (CTX->GetGdbServerStatus()) {
aarch64::Label RunBlock;
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Thread))); // Get thread
ldr(x0, MemOperand(x0, offsetof(FEXCore::Core::InternalThreadState, CTX))); // Get Context
ldr(w0, MemOperand(x0, offsetof(FEXCore::Context::Context, Config.RunningMode)));
// If the value == 0 then we don't need to stop
cbz(w0, &RunBlock);
{
// Make sure RIP is syncronized to the context
LoadConstant(x0, Entry);
str(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, State.rip)));
// Stop the thread
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA)));
br(x0);
}
bind(&RunBlock);
if (GDBEnabled) {
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(GuestEntry, Entry);
GetBuffer()->CursorForward(GDBSize);
}
//LOGMAN_THROW_A_FMT(RAData->HasFullRA(), "Arm64 JIT only works with RA");
@@ -702,7 +734,7 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == IR::OP_CODEBLOCK, "IR type failed to be a code block");
#endif
uintptr_t BlockStartHostCode = GetCursorAddress<uintptr_t>();
auto BlockStartHostCode = GetCursorAddress<uint8_t *>();
{
const auto Node = IR->GetID(BlockNode);
const auto IsTarget = JumpTargets.try_emplace(Node).first;
@@ -726,7 +758,10 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
}
if (DebugData) {
DebugData->Subblocks.push_back({BlockStartHostCode, static_cast<uint32_t>(GetCursorAddress<uintptr_t>() - BlockStartHostCode)});
DebugData->Subblocks.push_back({
static_cast<uint32_t>(BlockStartHostCode - GuestEntry),
static_cast<uint32_t>(GetCursorAddress<uint8_t *>() - BlockStartHostCode)
});
}
}
@@ -739,66 +774,17 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
FinalizeCode();
auto CodeEnd = GetCursorAddress<uint64_t>();
CPU.EnsureIAndDCacheCoherency(reinterpret_cast<void*>(GuestEntry), CodeEnd - reinterpret_cast<uint64_t>(GuestEntry));
auto CodeEnd = GetCursorAddress<uint8_t *>();
CPU.EnsureIAndDCacheCoherency(GuestEntry, CodeEnd - GuestEntry);
if (DebugData) {
DebugData->HostCodeSize = reinterpret_cast<uintptr_t>(CodeEnd) - reinterpret_cast<uintptr_t>(GuestEntry);
DebugData->HostCodeSize = CodeEnd - GuestEntry;
DebugData->Relocations = &Relocations;
}
this->IR = nullptr;
return reinterpret_cast<void*>(GuestEntry);
}
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
auto GuestRip = record[1];
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
if (!HostCode) {
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
Frame->State.rip = GuestRip;
return Frame->Pointers.Common.DispatcherLoopTop;
}
uintptr_t branch = (uintptr_t)(record) - 8;
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
auto offset = HostCode/4 - branch/4;
if (IsInt26(offset)) {
// optimal case - can branch directly
// patch the code
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
emit.b(offset);
emit.FinalizeCode();
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
// Add de-linking handler
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [branch, LinkerAddress]{
vixl::aarch64::Assembler emit((uint8_t*)(branch), 24);
vixl::CodeBufferCheckScope scope(&emit, 24, vixl::CodeBufferCheckScope::kDontReserveBufferSpace, vixl::CodeBufferCheckScope::kNoAssert);
Literal l_BranchHost{LinkerAddress};
emit.ldr(x0, &l_BranchHost);
emit.blr(x0);
emit.place(&l_BranchHost);
emit.FinalizeCode();
vixl::aarch64::CPU::EnsureIAndDCacheCoherency((void*)branch, 24);
});
} else {
// fallback case - do a soft-er link by patching the pointer
record[0] = HostCode;
// Add de-linking handler
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
record[0] = LinkerAddress;
});
}
return HostCode;
return GuestEntry;
}
void Arm64JITCore::ResetStack() {
@@ -822,9 +808,8 @@ void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX) {
Arm64JITCore::InitializeSignalHandlers(CTX);
}
void GetArm64JITDispatcherConfig(DispatcherConfig &config) {
config = DispatcherConfig {
.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Arm64JITCore_ExitFunctionLink),
CPUBackendFeatures GetArm64JITBackendFeatures() {
return CPUBackendFeatures {
.SupportsStaticRegisterAllocation = true
};
}
@@ -52,7 +52,7 @@ public:
[[nodiscard]] void *CompileCode(uint64_t Entry,
FEXCore::IR::IRListView const *IR,
FEXCore::Core::DebugData *DebugData,
FEXCore::IR::RegisterAllocationData *RAData) override;
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
@@ -208,7 +208,7 @@ private:
/**
* @brief Current guest RIP entrypoint
*/
uint64_t GuestEntry{};
uint8_t *GuestEntry{};
using OpHandler = void (Arm64JITCore::*)(IR::IROp_Header *IROp, IR::NodeID Node);
std::array<OpHandler, IR::IROps::OP_LAST + 1> OpHandlers {};
+2 -2
View File
@@ -16,11 +16,11 @@ class CPUBackend;
[[nodiscard]] std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread);
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX);
void GetX86JITDispatcherConfig(DispatcherConfig &config);
CPUBackendFeatures GetX86JITBackendFeatures();
[[nodiscard]] std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread);
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX);
void GetArm64JITDispatcherConfig(DispatcherConfig &config);
CPUBackendFeatures GetArm64JITBackendFeatures();
} // namespace FEXCore::CPU
+31 -46
View File
@@ -300,6 +300,27 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
}
}
static uint64_t X86JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
auto GuestRip = record[1];
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
if (!HostCode) {
Thread->CurrentFrame->State.rip = GuestRip;
return Frame->Pointers.Common.DispatcherLoopTop;
}
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
// undo the link
record[0] = LinkerAddress;
});
record[0] = HostCode;
return HostCode;
}
void X86JITCore::Op_NoOp(IR::IROp_Header *IROp, IR::NodeID Node) {
}
@@ -350,6 +371,7 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
Common.SyscallHandlerObj = reinterpret_cast<uint64_t>(CTX->SyscallHandler);
Common.SyscallHandlerFunc = reinterpret_cast<uint64_t>(FEXCore::Context::HandleSyscall);
Common.ExitFunctionLink = reinterpret_cast<uintptr_t>(&X86JITCore_ExitFunctionLink);
// Fill in the fallback handlers
InterpreterOps::FillFallbackIndexPointers(Common.FallbackHandlerPointers);
@@ -547,7 +569,7 @@ std::tuple<X86JITCore::SetCC, X86JITCore::CMovCC, X86JITCore::JCC> X86JITCore::G
return { &CodeGenerator::sete , &CodeGenerator::cmove , &CodeGenerator::je };
}
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) {
void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, [[maybe_unused]] FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) {
JumpTargets.clear();
uint32_t SSACount = IR->GetSSACount();
@@ -555,32 +577,18 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
this->RAData = RAData;
// Fairly excessive buffer range to make sure we don't overflow
uint32_t BufferRange = SSACount * 16;
uint32_t BufferRange = SSACount * 16 + GDBEnabled * Dispatcher::MaxGDBPauseCheckSize;
if ((getSize() + BufferRange) > CurrentCodeBuffer->Size) {
ThreadState->CTX->ClearCodeCache(ThreadState);
CTX->ClearCodeCache(ThreadState);
}
void *GuestEntry = getCurr<void*>();
auto GuestEntry = getCurr<uint8_t*>();
CursorEntry = getSize();
this->IR = IR;
if (CTX->GetGdbServerStatus()) {
Label RunBlock;
// If we have a gdb server running then run in a less efficient mode that checks if we need to exit
// This happens when single stepping
static_assert(sizeof(CTX->Config.RunningMode) == 4, "This is expected to be size of 4");
mov(rax, reinterpret_cast<uint64_t>(CTX));
// If the value == 0 then branch to the top
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
je(RunBlock);
// Else we need to pause now
mov(rax, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
jmp(rax);
ud2();
L(RunBlock);
if (GDBEnabled) {
auto GDBSize = CTX->Dispatcher->GenerateGDBPauseCheck(GuestEntry, Entry);
setSize(getSize() + GDBSize);
}
LOGMAN_THROW_A_FMT(RAData != nullptr, "Needs RA");
@@ -720,35 +728,12 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
return GuestEntry;
}
static uint64_t X86JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
auto GuestRip = record[1];
auto HostCode = Thread->LookupCache->FindBlock(GuestRip);
if (!HostCode) {
Thread->CurrentFrame->State.rip = GuestRip;
return Frame->Pointers.Common.DispatcherLoopTop;
}
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
// undo the link
record[0] = LinkerAddress;
});
record[0] = HostCode;
return HostCode;
}
std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread) {
return std::make_unique<X86JITCore>(ctx, Thread);
}
void GetX86JITDispatcherConfig(DispatcherConfig &config) {
config = DispatcherConfig {
.ExitFunctionLink = reinterpret_cast<uintptr_t>(&X86JITCore_ExitFunctionLink)
};
CPUBackendFeatures GetX86JITBackendFeatures() {
return CPUBackendFeatures { };
}
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX) {
@@ -60,7 +60,7 @@ public:
[[nodiscard]] void *CompileCode(uint64_t Entry,
FEXCore::IR::IRListView const *IR,
FEXCore::Core::DebugData *DebugData,
FEXCore::IR::RegisterAllocationData *RAData) override;
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) override;
[[nodiscard]] void *MapRegion(void* HostPtr, uint64_t, uint64_t) override { return HostPtr; }
+5 -1
View File
@@ -33,6 +33,10 @@ namespace CodeSerialize {
}
namespace CPU {
struct CPUBackendFeatures {
bool SupportsStaticRegisterAllocation = false;
};
class CPUBackend {
public:
struct CodeBuffer {
@@ -71,7 +75,7 @@ namespace CPU {
[[nodiscard]] virtual void *CompileCode(uint64_t Entry,
FEXCore::IR::IRListView const *IR,
FEXCore::Core::DebugData *DebugData,
FEXCore::IR::RegisterAllocationData *RAData) = 0;
FEXCore::IR::RegisterAllocationData *RAData, bool GDBEnabled) = 0;
/**
* @brief Relocates a block of code from the JIT code object cache
+1
View File
@@ -105,6 +105,7 @@ namespace FEXCore::Core {
uint64_t CPUIDFunction{};
uint64_t SyscallHandlerObj{};
uint64_t SyscallHandlerFunc{};
uint64_t ExitFunctionLink{};
uint64_t FallbackHandlerPointers[FallbackHandlerIndex::OPINDEX_MAX];
@@ -42,7 +42,7 @@ namespace FEXCore::Core {
};
struct DebugDataSubblock {
uintptr_t HostCodeStart;
uint32_t HostCodeOffset;
uint32_t HostCodeSize;
};