Backends+Context: Make Dispatcher per Context from per Thread

This commit is contained in:
Stefanos Kornilios Misis Poiitidis authored and Stefanos Kornilios Mitsis Poiitidis committed 2022-06-16 19:41:42 +00:00
1 parent eac579f714
commit 39f3406fba
25 files changed
+287 -306

No files matched your search

+3 -1
View File
@@ -43,6 +43,7 @@ namespace CPU {
class Arm64JITCore;
class X86JITCore;
class InterpreterCore;
class Dispatcher;
}
namespace HLE {
struct SyscallArguments;
@@ -129,6 +130,7 @@ namespace FEXCore::Context {
FEXCore::CPUIDEmu CPUID;
FEXCore::HLE::SyscallHandler *SyscallHandler{};
std::unique_ptr<FEXCore::ThunkHandler> ThunkHandler;
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
CustomCPUFactoryType CustomCPUFactory;
FEXCore::Context::ExitHandler CustomExitHandler;
@@ -327,7 +329,7 @@ namespace FEXCore::Context {
*
* InitializeCompiler is called inside of CreateThread, so you likely don't need this
*/
void InitializeCompiler(FEXCore::Core::InternalThreadState* State);
void InitializeCompiler(FEXCore::Core::InternalThreadState* Thread);
void WaitForIdleWithTimeout();
+14 -2
View File
@@ -58,7 +58,7 @@ auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
Buffer.Ptr = static_cast<uint8_t *>(
FEXCore::Allocator::mmap(nullptr, Buffer.Size, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0));
LOGMAN_THROW_A_FMT(!!Buffer.Ptr, "Couldn't allocate code buffer");
Dispatcher->RegisterCodeBuffer(Buffer.Ptr, Buffer.Size);
if (ThreadState->CTX->Config.GlobalJITNaming()) {
ThreadState->CTX->Symbols.RegisterJITSpace(Buffer.Ptr, Buffer.Size);
}
@@ -67,7 +67,19 @@ auto CPUBackend::AllocateNewCodeBuffer(size_t Size) -> CodeBuffer {
void CPUBackend::FreeCodeBuffer(CodeBuffer Buffer) {
FEXCore::Allocator::munmap(Buffer.Ptr, Buffer.Size);
Dispatcher->RemoveCodeBuffer(Buffer.Ptr);
}
bool CPUBackend::IsAddressInCodeBuffer(uintptr_t Address) const {
for (auto &Buffer: CodeBuffers) {
auto start = (uintptr_t)Buffer.Ptr;
auto end = start + Buffer.Size;
if (Address >= start && Address < end) {
return true;
}
}
return false;
}
}
+36 -18
View File
@@ -17,6 +17,7 @@ $end_info$
#include "Interface/Core/OpcodeDispatcher.h"
#include "Interface/Core/Interpreter/InterpreterCore.h"
#include "Interface/Core/JIT/JITCore.h"
#include "Interface/Core/Dispatcher/Dispatcher.h"
#include "Interface/HLE/Thunks/Thunks.h"
#include "Interface/IR/Passes/RegisterAllocationPass.h"
#include "Interface/IR/Passes.h"
@@ -194,18 +195,22 @@ namespace FEXCore::Context {
}
FEXCore::Core::InternalThreadState* Context::InitCore(uint64_t InitialRIP, uint64_t StackPointer) {
FEXCore::CPU::DispatcherConfig DispatcherConfig;
// Initialize the CPU core signal handlers
switch (Config.Core) {
#ifdef INTERPRETER_ENABLED
case FEXCore::Config::CONFIG_INTERPRETER:
FEXCore::CPU::InitializeInterpreterSignalHandlers(this);
FEXCore::CPU::GetInterpreterDispatcherConfig(DispatcherConfig);
break;
#endif
case FEXCore::Config::CONFIG_IRJIT:
#if (_M_X86_64 && JIT_X86_64)
FEXCore::CPU::InitializeX86JITSignalHandlers(this);
FEXCore::CPU::GetX86JITDispatcherConfig(DispatcherConfig);
#elif (_M_ARM_64 && JIT_ARM64)
FEXCore::CPU::InitializeArm64JITSignalHandlers(this);
FEXCore::CPU::GetArm64JITDispatcherConfig(DispatcherConfig);
#else
ERROR_AND_DIE_FMT("FEXCore has been compiled without a viable JIT core");
#endif
@@ -229,6 +234,14 @@ namespace FEXCore::Context {
ThunkHandler.reset(FEXCore::ThunkHandler::Create());
#if (_M_X86_64)
Dispatcher = FEXCore::CPU::Dispatcher::CreateX86(this, DispatcherConfig);
#elif (_M_ARM_64)
Dispatcher = FEXCore::CPU::Dispatcher::CreateArm64(this, DispatcherConfig);
#else
ERROR_AND_DIE_FMT("FEXCore has been compiled with an unknown target");
#endif
using namespace FEXCore::Core;
FEXCore::Core::CPUState NewThreadState = CreateDefaultCPUState();
@@ -257,7 +270,7 @@ namespace FEXCore::Context {
}
void Context::HandleCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP) {
Thread->CPUBackend->CallbackPtr(Thread->CurrentFrame, RIP);
Thread->CTX->Dispatcher->ExecuteJITCallback(Thread->CurrentFrame, RIP);
}
void Context::RegisterHostSignalHandler(int Signal, HostSignalDelegatorFunction Func, bool Required) {
@@ -482,17 +495,22 @@ namespace FEXCore::Context {
Thread->StartRunning.NotifyAll();
}
void Context::InitializeCompiler(FEXCore::Core::InternalThreadState* State) {
State->OpDispatcher = std::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
State->OpDispatcher->SetMultiblock(Config.Multiblock);
State->LookupCache = std::make_unique<FEXCore::LookupCache>(this);
State->FrontendDecoder = std::make_unique<FEXCore::Frontend::Decoder>(this);
State->PassManager = std::make_unique<FEXCore::IR::PassManager>();
State->PassManager->RegisterExitHandler([this]() {
void Context::InitializeCompiler(FEXCore::Core::InternalThreadState* Thread) {
Thread->OpDispatcher = std::make_unique<FEXCore::IR::OpDispatchBuilder>(this);
Thread->OpDispatcher->SetMultiblock(Config.Multiblock);
Thread->LookupCache = std::make_unique<FEXCore::LookupCache>(this);
Thread->FrontendDecoder = std::make_unique<FEXCore::Frontend::Decoder>(this);
Thread->PassManager = std::make_unique<FEXCore::IR::PassManager>();
Thread->PassManager->RegisterExitHandler([this]() {
Stop(false /* Ignore current thread */);
});
State->CTX = this;
Thread->CurrentFrame->Pointers.Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
Thread->CurrentFrame->Pointers.Common.L2Pointer = Thread->LookupCache->GetPagePointer();
Dispatcher->InitThreadPointers(Thread);
Thread->CTX = this;
#if _M_ARM_64
bool DoSRA = State->CTX->Config.StaticRegisterAllocation;
@@ -500,32 +518,32 @@ namespace FEXCore::Context {
bool DoSRA = false;
#endif
State->PassManager->AddDefaultPasses(this, Config.Core == FEXCore::Config::CONFIG_IRJIT, DoSRA);
State->PassManager->AddDefaultValidationPasses();
Thread->PassManager->AddDefaultPasses(this, Config.Core == FEXCore::Config::CONFIG_IRJIT, DoSRA);
Thread->PassManager->AddDefaultValidationPasses();
State->PassManager->RegisterSyscallHandler(SyscallHandler);
Thread->PassManager->RegisterSyscallHandler(SyscallHandler);
// Create CPU backend
switch (Config.Core) {
#ifdef INTERPRETER_ENABLED
case FEXCore::Config::CONFIG_INTERPRETER:
State->CPUBackend = FEXCore::CPU::CreateInterpreterCore(this, State);
Thread->CPUBackend = FEXCore::CPU::CreateInterpreterCore(this, Thread);
break;
#endif
case FEXCore::Config::CONFIG_IRJIT:
State->PassManager->InsertRegisterAllocationPass(DoSRA);
Thread->PassManager->InsertRegisterAllocationPass(DoSRA);
#if (_M_X86_64 && JIT_X86_64)
State->CPUBackend = FEXCore::CPU::CreateX86JITCore(this, State);
Thread->CPUBackend = FEXCore::CPU::CreateX86JITCore(this, Thread);
#elif (_M_ARM_64 && JIT_ARM64)
State->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, State);
Thread->CPUBackend = FEXCore::CPU::CreateArm64JITCore(this, Thread);
#else
ERROR_AND_DIE_FMT("FEXCore has been compiled without a viable JIT core");
#endif
break;
case FEXCore::Config::CONFIG_CUSTOM:
State->CPUBackend = CustomCPUFactory(this, State);
Thread->CPUBackend = CustomCPUFactory(this, Thread);
break;
default:
ERROR_AND_DIE_FMT("Unknown core configuration");
@@ -1056,7 +1074,7 @@ namespace FEXCore::Context {
Thread->RunningEvents.Running = true;
Thread->CPUBackend->ExecuteDispatch(Thread->CurrentFrame);
Thread->CTX->Dispatcher->ExecuteDispatch(Thread->CurrentFrame);
Thread->RunningEvents.Running = false;
}
@@ -38,12 +38,12 @@ using namespace vixl::aarch64;
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
#define STATE x28
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
: FEXCore::CPU::Dispatcher(ctx, Thread), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
SRAEnabled = config.StaticRegisterAssignment;
Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &config)
: FEXCore::CPU::Dispatcher(ctx), Arm64Emitter(ctx, MAX_DISPATCHER_CODE_SIZE) {
SRAEnabled = config.SupportsStaticRegisterAllocation && ctx->Config.StaticRegisterAllocation;
SetAllowAssembler(true);
DispatchPtr = GetCursorAddress<CPUBackend::AsmDispatch>();
DispatchPtr = GetCursorAddress<AsmDispatch>();
// while (true) {
// Ptr = FindBlock(RIP)
@@ -53,12 +53,10 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
// Ptr();
// }
Literal l_PagePtr {Thread->LookupCache->GetPagePointer()};
Literal l_CTX {reinterpret_cast<uintptr_t>(CTX)};
Literal l_Sleep {reinterpret_cast<uint64_t>(SleepThread)};
Literal l_CompileBlock {GetCompileBlockPtr()};
Literal l_ExitFunctionLink {config.ExitFunctionLink};
Literal l_ExitFunctionLinkThis {config.ExitFunctionLinkThis};
// Push all the register we need to save
PushCalleeSavedRegisters();
@@ -115,10 +113,10 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
// This is the block cache lookup routine
// It matches what is going on it LookupCache.h::FindBlock
ldr(x0, &l_PagePtr);
ldr(x0, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L2Pointer)));
// Mask the address by the virtual address size so we can check for aliases
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
if (std::popcount(VirtualMemorySize) == 1) {
and_(x3, RipReg, VirtualMemorySize - 1);
}
@@ -233,9 +231,8 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
svc(0);
}
ldr(x0, &l_ExitFunctionLinkThis);
mov(x1, STATE);
mov(x2, lr);
mov(x0, STATE);
mov(x1, lr);
ldr(x3, &l_ExitFunctionLink);
blr(x3);
@@ -347,15 +344,14 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
if (SRAEnabled)
SpillStaticRegs();
LoadConstant(x0, reinterpret_cast<uint64_t>(&SynchronousFaultData));
LoadConstant(w1, 1);
strb(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, FaultToTopAndGeneratedException)));
strb(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)));
LoadConstant(w1, X86State::X86_TRAPNO_OF);
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, TrapNo)));
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)));
LoadConstant(w1, 0x80);
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, si_code)));
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)));
LoadConstant(x1, 0);
str(w1, MemOperand(x0, offsetof(Dispatcher::SynchronousFaultDataStruct, err_code)));
str(w1, MemOperand(STATE, offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)));
// hlt/udf = SIGILL
// brk = SIGTRAP
@@ -401,7 +397,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
// On return to the thunk, the thunk can get whatever its return value is from the thread context depending on ABI handling on its end
// When the thunk itself returns, it'll do its regular return logic there
// void ReentrantCallback(FEXCore::Core::InternalThreadState *Thread, uint64_t RIP);
CallbackPtr = GetCursorAddress<CPUBackend::JITCallback>();
CallbackPtr = GetCursorAddress<JITCallback>();
// We expect the thunk to have previously pushed the registers it was using
PushCalleeSavedRegisters();
@@ -437,14 +433,8 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
b(&LoopTop);
}
// Long division helpers
uint64_t LUDIVHandler{};
uint64_t LDIVHandler{};
uint64_t LUREMHandler{};
uint64_t LREMHandler{};
{
LUDIVHandler = GetCursorAddress<uint64_t>();
LUDIVHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR();
@@ -463,7 +453,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
}
{
LDIVHandler = GetCursorAddress<uint64_t>();
LDIVHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR();
@@ -482,7 +472,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
}
{
LUREMHandler = GetCursorAddress<uint64_t>();
LUREMHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR();
@@ -501,7 +491,7 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
}
{
LREMHandler = GetCursorAddress<uint64_t>();
LREMHandlerAddress = GetCursorAddress<uint64_t>();
PushDynamicRegsAndLR();
@@ -519,12 +509,10 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
ret();
}
place(&l_PagePtr);
place(&l_CTX);
place(&l_Sleep);
place(&l_CompileBlock);
place(&l_ExitFunctionLink);
place(&l_ExitFunctionLinkThis);
FinalizeCode();
@@ -540,10 +528,27 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
if (CTX->Config.GlobalJITNaming()) {
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(DispatchPtr), End - reinterpret_cast<uint64_t>(DispatchPtr));
}
}
// Setup dispatcher specific pointers that need to be accessed from JIT code
void Arm64Dispatcher::SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {
for(int i = 0; i < SRA64.size(); i++) {
if (IgnoreMask & (1U << SRA64[i].GetCode())) {
// Skip this one, it's already spilled
continue;
}
Thread->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
}
for(int i = 0; i < SRAFPR.size(); i++) {
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
memcpy(&Thread->CurrentFrame->State.xmm[i][0], &FPR, sizeof(__uint128_t));
}
}
void Arm64Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
auto &Common = Thread->CurrentFrame->Pointers.Common;
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
@@ -552,29 +557,17 @@ Arm64Dispatcher::Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::
Common.UnimplementedInstructionHandler = UnimplementedInstructionAddress;
Common.OverflowExceptionHandler = OverflowExceptionInstructionAddress;
Common.SignalReturnHandler = SignalHandlerReturnAddress;
Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
auto &AArch64 = ThreadState->CurrentFrame->Pointers.AArch64;
AArch64.LUDIVHandler = LUDIVHandler;
AArch64.LDIVHandler = LDIVHandler;
AArch64.LUREMHandler = LUREMHandler;
AArch64.LREMHandler = LREMHandler;
auto &AArch64 = Thread->CurrentFrame->Pointers.AArch64;
AArch64.LUDIVHandler = LUDIVHandlerAddress;
AArch64.LDIVHandler = LDIVHandlerAddress;
AArch64.LUREMHandler = LUREMHandlerAddress;
AArch64.LREMHandler = LREMHandlerAddress;
}
}
void Arm64Dispatcher::SpillSRA(void *ucontext, uint32_t IgnoreMask) {
for(int i = 0; i < SRA64.size(); i++) {
if (IgnoreMask & (1U << SRA64[i].GetCode())) {
// Skip this one, it's already spilled
continue;
}
ThreadState->CurrentFrame->State.gregs[i] = ArchHelpers::Context::GetArmReg(ucontext, SRA64[i].GetCode());
}
for(int i = 0; i < SRAFPR.size(); i++) {
auto FPR = ArchHelpers::Context::GetArmFPR(ucontext, SRAFPR[i].GetCode());
memcpy(&ThreadState->CurrentFrame->State.xmm[i][0], &FPR, sizeof(__uint128_t));
}
std::unique_ptr<Dispatcher> Dispatcher::CreateArm64(FEXCore::Context::Context *CTX, DispatcherConfig &Config) {
return std::make_unique<Arm64Dispatcher>(CTX, Config);
}
}
}
@@ -15,10 +15,18 @@ namespace FEXCore::CPU {
class Arm64Dispatcher final : public Dispatcher, public Arm64Emitter {
public:
Arm64Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
Arm64Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &config);
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
protected:
void SpillSRA(void *ucontext, uint32_t IgnoreMask) override;
void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) override;
private:
// Long division helpers
uint64_t LUDIVHandlerAddress{};
uint64_t LDIVHandlerAddress{};
uint64_t LUREMHandlerAddress{};
uint64_t LREMHandlerAddress{};
};
}
@@ -39,7 +39,7 @@ void Dispatcher::SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuS
ctx->IdleWaitCV.notify_all();
}
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, void *ucontext) {
ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext) {
// We can end up getting a signal at any point in our host state
// Jump to a handler that saves all state so we can safely return
uint64_t OldSP = ArchHelpers::Context::GetSp(ucontext);
@@ -64,7 +64,7 @@ ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, vo
// Save guest state
// We can't guarantee if registers are in context or host GPRs
// So we need to save everything
memcpy(&Context->GuestState, ThreadState->CurrentFrame, sizeof(FEXCore::Core::CPUState));
memcpy(&Context->GuestState, Thread->CurrentFrame, sizeof(FEXCore::Core::CPUState));
// Set the new SP
ArchHelpers::Context::SetSp(ucontext, NewSP);
@@ -81,13 +81,13 @@ ArchHelpers::Context::ContextBackup* Dispatcher::StoreThreadState(int Signal, vo
Context->SigInfoLocation = 0;
// Store fault to top status and then reset it
Context->FaultToTopAndGeneratedException = SynchronousFaultData.FaultToTopAndGeneratedException;
SynchronousFaultData.FaultToTopAndGeneratedException = false;
Context->FaultToTopAndGeneratedException = Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException;
Thread->CurrentFrame->SynchronousFaultData.FaultToTopAndGeneratedException = false;
return Context;
}
void Dispatcher::RestoreThreadState(void *ucontext) {
void Dispatcher::RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext) {
uint64_t OldSP{};
if (CTX->Config.Core() == FEXCore::Config::CONFIG_IRJIT) {
OldSP = ArchHelpers::Context::GetSp(ucontext);
@@ -102,13 +102,13 @@ void Dispatcher::RestoreThreadState(void *ucontext) {
auto Context = reinterpret_cast<ArchHelpers::Context::ContextBackup*>(NewSP);
// First thing, reset the guest state
memcpy(ThreadState->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
memcpy(Thread->CurrentFrame, &Context->GuestState, sizeof(FEXCore::Core::CPUState));
// Now restore host state
ArchHelpers::Context::RestoreContext(ucontext, Context);
if (Context->UContextLocation) {
auto Frame = ThreadState->CurrentFrame;
auto Frame = Thread->CurrentFrame;
if (Context->Flags &ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT) {
// XXX: Unsupported since it needs state reconstruction
@@ -273,14 +273,14 @@ static uint32_t ConvertSignalToError(int Signal, siginfo_t *HostSigInfo) {
return 0;
}
bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
auto ContextBackup = StoreThreadState(Signal, ucontext);
bool Dispatcher::HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) {
auto ContextBackup = StoreThreadState(Thread, Signal, ucontext);
auto Frame = ThreadState->CurrentFrame;
auto Frame = Thread->CurrentFrame;
// Ref count our faults
// We use this to track if it is safe to clear cache
++ThreadState->CurrentFrame->SignalHandlerRefCounter;
++Thread->CurrentFrame->SignalHandlerRefCounter;
uint64_t OldPC = ArchHelpers::Context::GetPc(ucontext);
// Set the new PC
@@ -299,7 +299,7 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
// We are going to be returning to the top of the dispatcher which will fill again
// Otherwise we might load garbage
if (SRAEnabled) {
if (IsAddressInJITCode(OldPC, false)) {
if (Thread->CPUBackend->IsAddressInCodeBuffer(OldPC)) {
uint32_t IgnoreMask{};
#ifdef _M_ARM_64
if (Frame->InSyscallInfo != 0) {
@@ -323,11 +323,11 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
#endif
// We are in jit, SRA must be spilled
SpillSRA(ucontext, IgnoreMask);
SpillSRA(Thread, ucontext, IgnoreMask);
ContextBackup->Flags |= ArchHelpers::Context::ContextFlags::CONTEXT_FLAG_INJIT;
} else {
if (!IsAddressInJITCode(OldPC, true)) {
if (!IsAddressInDispatcher(OldPC)) {
// This is likely to cause issues but in some cases it isn't fatal
// This can also happen if we have put a signal on hold, then we just reenabled the signal
// So we are in the syscall handler
@@ -409,11 +409,11 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
*guest_siginfo = *HostSigInfo;
if (ContextBackup->FaultToTopAndGeneratedException) {
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = SynchronousFaultData.TrapNo;
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = SynchronousFaultData.err_code;
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
// Overwrite si_code
guest_siginfo->si_code = SynchronousFaultData.si_code;
guest_siginfo->si_code = Thread->CurrentFrame->SynchronousFaultData.si_code;
}
else {
guest_uctx->uc_mcontext.gregs[FEXCore::x86_64::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
@@ -500,9 +500,9 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ES] = Frame->State.es;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_DS] = Frame->State.ds;
if (ContextBackup->FaultToTopAndGeneratedException) {
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = SynchronousFaultData.TrapNo;
guest_siginfo->si_code = SynchronousFaultData.si_code;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = SynchronousFaultData.err_code;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = Frame->SynchronousFaultData.TrapNo;
guest_siginfo->si_code = Frame->SynchronousFaultData.si_code;
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_ERR] = Frame->SynchronousFaultData.err_code;
}
else {
guest_uctx->uc_mcontext.gregs[FEXCore::x86::FEX_REG_TRAPNO] = ConvertSignalToTrapNo(Signal, HostSigInfo);
@@ -633,44 +633,44 @@ bool Dispatcher::HandleGuestSignal(int Signal, void *info, void *ucontext, Guest
return true;
}
bool Dispatcher::HandleSIGILL(int Signal, void *info, void *ucontext) {
bool Dispatcher::HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
if (ArchHelpers::Context::GetPc(ucontext) == SignalHandlerReturnAddress) {
RestoreThreadState(ucontext);
RestoreThreadState(Thread, ucontext);
// Ref count our faults
// We use this to track if it is safe to clear cache
--ThreadState->CurrentFrame->SignalHandlerRefCounter;
--Thread->CurrentFrame->SignalHandlerRefCounter;
return true;
}
if (ArchHelpers::Context::GetPc(ucontext) == PauseReturnInstruction) {
RestoreThreadState(ucontext);
RestoreThreadState(Thread, ucontext);
// Ref count our faults
// We use this to track if it is safe to clear cache
--ThreadState->CurrentFrame->SignalHandlerRefCounter;
--Thread->CurrentFrame->SignalHandlerRefCounter;
return true;
}
return false;
}
bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
FEXCore::Core::SignalEvent SignalReason = ThreadState->SignalReason.load();
auto Frame = ThreadState->CurrentFrame;
bool Dispatcher::HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) {
FEXCore::Core::SignalEvent SignalReason = Thread->SignalReason.load();
auto Frame = Thread->CurrentFrame;
if (SignalReason == FEXCore::Core::SignalEvent::Pause) {
// Store our thread state so we can come back to this
StoreThreadState(Signal, ucontext);
StoreThreadState(Thread, Signal, ucontext);
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
if (SRAEnabled && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
// We are in jit, SRA must be spilled
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddressSpillSRA);
} else {
if (SRAEnabled) {
// We are in non-jit, SRA is already spilled
LOGMAN_THROW_A_FMT(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true),
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
"Signals in dispatcher have unsynchronized context");
}
ArchHelpers::Context::SetPc(ucontext, ThreadPauseHandlerAddress);
@@ -681,9 +681,9 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
// Ref count our faults
// We use this to track if it is safe to clear cache
++ThreadState->CurrentFrame->SignalHandlerRefCounter;
++Thread->CurrentFrame->SignalHandlerRefCounter;
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
return true;
}
@@ -694,16 +694,16 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
ArchHelpers::Context::SetSp(ucontext, Frame->ReturningStackLocation);
// Our ref counting doesn't matter anymore
ThreadState->CurrentFrame->SignalHandlerRefCounter = 0;
Thread->CurrentFrame->SignalHandlerRefCounter = 0;
// Set the new PC
if (SRAEnabled && IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), false)) {
if (SRAEnabled && Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
// We are in jit, SRA must be spilled
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddressSpillSRA);
} else {
if (SRAEnabled) {
// We are in non-jit, SRA is already spilled
LOGMAN_THROW_A_FMT(!IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext), true),
LOGMAN_THROW_A_FMT(!IsAddressInDispatcher(ArchHelpers::Context::GetPc(ucontext)),
"Signals in dispatcher have unsynchronized context");
}
ArchHelpers::Context::SetPc(ucontext, ThreadStopHandlerAddress);
@@ -712,24 +712,24 @@ bool Dispatcher::HandleSignalPause(int Signal, void *info, void *ucontext) {
// We need to be a little bit careful here
// If we were already paused (due to GDB) and we are immediately stopping (due to gdb kill)
// Then we need to ensure we don't double decrement our idle thread counter
if (ThreadState->RunningEvents.ThreadSleeping) {
if (Thread->RunningEvents.ThreadSleeping) {
// If the thread was sleeping then its idle counter was decremented
// Reincrement it here to not break logic
++ThreadState->CTX->IdleWaitRefCount;
++Thread->CTX->IdleWaitRefCount;
}
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
return true;
}
if (SignalReason == FEXCore::Core::SignalEvent::Return) {
RestoreThreadState(ucontext);
RestoreThreadState(Thread, ucontext);
// Ref count our faults
// We use this to track if it is safe to clear cache
--ThreadState->CurrentFrame->SignalHandlerRefCounter;
--Thread->CurrentFrame->SignalHandlerRefCounter;
ThreadState->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
Thread->SignalReason.store(FEXCore::Core::SignalEvent::Nothing);
return true;
}
@@ -748,28 +748,4 @@ uint64_t Dispatcher::GetCompileBlockPtr() {
return CompileBlockPtr.Data;
}
void Dispatcher::RemoveCodeBuffer(uint8_t* start_to_remove) {
for (auto iter = CodeBuffers.begin(); iter != CodeBuffers.end(); ++iter) {
auto [start, end] = *iter;
if (start == reinterpret_cast<uint64_t>(start_to_remove)) {
CodeBuffers.erase(iter);
return;
}
}
}
bool Dispatcher::IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher) const {
for (auto [start, end] : CodeBuffers) {
if (Address >= start && Address < end) {
return true;
}
}
if (IncludeDispatcher && IsAddressInDispatcher(Address)) {
return true;
}
return false;
}
}
@@ -26,17 +26,13 @@ namespace FEXCore::CPU {
struct DispatcherConfig {
bool InterpreterDispatch = false;
uintptr_t ExitFunctionLink = 0;
uintptr_t ExitFunctionLinkThis = 0;
bool StaticRegisterAssignment = false;
bool SupportsStaticRegisterAllocation = false;
};
class Dispatcher {
public:
virtual ~Dispatcher() = default;
CPUBackend::AsmDispatch DispatchPtr;
CPUBackend::JITCallback CallbackPtr;
CPUBackend::IntCallbackReturn ReturnPtr;
/**
* @name Dispatch Helper functions
* @{ */
@@ -50,58 +46,59 @@ public:
uint64_t SignalHandlerReturnAddress{};
uint64_t UnimplementedInstructionAddress{};
uint64_t OverflowExceptionInstructionAddress{};
uint64_t IntCallbackReturnAddress{};
uint64_t PauseReturnInstruction{};
/** @} */
struct SynchronousFaultDataStruct {
bool FaultToTopAndGeneratedException{};
uint32_t TrapNo;
uint32_t err_code;
uint32_t si_code;
} SynchronousFaultData;
uint64_t Start{};
uint64_t End{};
bool HandleGuestSignal(int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
bool HandleSIGILL(int Signal, void *info, void *ucontext);
bool HandleSignalPause(int Signal, void *info, void *ucontext);
bool HandleGuestSignal(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack);
bool HandleSIGILL(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
bool HandleSignalPause(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext);
void RegisterCodeBuffer(uint8_t* start, size_t size) {
CodeBuffers.emplace_back(reinterpret_cast<uint64_t>(start),
reinterpret_cast<uint64_t>(start + size));
}
void RemoveCodeBuffer(uint8_t* start);
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const;
bool IsAddressInDispatcher(uint64_t Address) const {
return Address >= Start && Address < End;
}
protected:
Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
: CTX {ctx}
, ThreadState {Thread} {}
virtual void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) = 0;
ArchHelpers::Context::ContextBackup* StoreThreadState(int Signal, void *ucontext);
void RestoreThreadState(void *ucontext);
static std::unique_ptr<Dispatcher> CreateX86(FEXCore::Context::Context *CTX, DispatcherConfig &Config);
static std::unique_ptr<Dispatcher> CreateArm64(FEXCore::Context::Context *CTX, DispatcherConfig &Config);
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
DispatchPtr(Frame);
}
void ExecuteJITCallback(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP) {
CallbackPtr(Frame, RIP);
}
protected:
Dispatcher(FEXCore::Context::Context *ctx)
: CTX {ctx}
{}
ArchHelpers::Context::ContextBackup* StoreThreadState(FEXCore::Core::InternalThreadState *Thread, int Signal, void *ucontext);
void RestoreThreadState(FEXCore::Core::InternalThreadState *Thread, void *ucontext);
std::stack<uint64_t, std::vector<uint64_t>> SignalFrames;
bool SRAEnabled = false;
virtual void SpillSRA(void *ucontext, uint32_t IgnoreMask) {}
virtual void SpillSRA(FEXCore::Core::InternalThreadState *Thread, void *ucontext, uint32_t IgnoreMask) {}
FEXCore::Context::Context *CTX;
FEXCore::Core::InternalThreadState *ThreadState;
static void SleepThread(FEXCore::Context::Context *ctx, FEXCore::Core::CpuStateFrame *Frame);
static uint64_t GetCompileBlockPtr();
private:
std::vector<std::tuple<uint64_t, uint64_t>> CodeBuffers; // Start, End
using AsmDispatch = void(*)(FEXCore::Core::CpuStateFrame *Frame);
using JITCallback = void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
AsmDispatch DispatchPtr;
JITCallback CallbackPtr;
};
}
@@ -24,15 +24,17 @@ namespace FEXCore::CPU {
static constexpr size_t MAX_DISPATCHER_CODE_SIZE = 4096;
#define STATE r14
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config)
: Dispatcher(ctx, Thread)
X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &config)
: Dispatcher(ctx)
, Xbyak::CodeGenerator(MAX_DISPATCHER_CODE_SIZE,
FEXCore::Allocator::mmap(nullptr, MAX_DISPATCHER_CODE_SIZE, PROT_READ | PROT_WRITE | PROT_EXEC, MAP_PRIVATE | MAP_ANONYMOUS, -1, 0),
nullptr) {
LOGMAN_THROW_A_FMT(!config.SupportsStaticRegisterAllocation, "X86 dispatcher does not support SRA");
using namespace Xbyak;
using namespace Xbyak::util;
DispatchPtr = getCurr<CPUBackend::AsmDispatch>();
DispatchPtr = getCurr<AsmDispatch>();
// Temp registers
// rax, rcx, rdx, rsi, r8, r9,
@@ -111,10 +113,10 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
}
L(FullLookup);
mov(r13, Thread->LookupCache->GetPagePointer());
mov(r13, qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.L2Pointer)]);
// Full lookup
uint64_t VirtualMemorySize = Thread->LookupCache->GetVirtualMemorySize();
uint64_t VirtualMemorySize = CTX->Config.VirtualMemSize;
mov(rax, rdx);
mov(rbx, VirtualMemorySize - 1);
and_(rax, rbx);
@@ -283,10 +285,9 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
mov(rax, r9);
}
// {rdi, rsi, rdx}
mov(rdi, config.ExitFunctionLinkThis);
mov(rsi, STATE);
mov(rdx, rax); // rax is set at the block end
// {rdi, rsi}
mov(rdi, STATE);
mov(rsi, rax); // rax is set at the block end
mov(rax, config.ExitFunctionLink);
call(rax);
@@ -331,7 +332,7 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
}
{
CallbackPtr = getCurr<CPUBackend::JITCallback>();
CallbackPtr = getCurr<JITCallback>();
push(rbx);
push(rbp);
@@ -386,18 +387,17 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
// ud2 = SIGILL
// int3 = SIGTRAP
// hlt = SIGSEGV
mov(rax, reinterpret_cast<uint64_t>(&SynchronousFaultData));
add(byte [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, FaultToTopAndGeneratedException)], 1);
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, TrapNo)], X86State::X86_TRAPNO_OF);
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, err_code)], 0);
mov(dword [rax + offsetof(Dispatcher::SynchronousFaultDataStruct, si_code)], 0x80);
add(byte [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.FaultToTopAndGeneratedException)], 1);
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.TrapNo)], X86State::X86_TRAPNO_OF);
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.err_code)], 0);
mov(dword [STATE + offsetof(FEXCore::Core::CpuStateFrame, SynchronousFaultData.si_code)], 0x80);
hlt();
}
{
ReturnPtr = getCurr<CPUBackend::IntCallbackReturn>();
IntCallbackReturnAddress = getCurr<uint64_t>();
// using CallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
// rdi = thread
@@ -431,23 +431,33 @@ X86Dispatcher::X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::Inte
CTX->Symbols.RegisterJITSpace(reinterpret_cast<void*>(Start), End-Start);
}
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddress;
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddress;
Common.UnimplementedInstructionHandler = UnimplementedInstructionAddress;
Common.OverflowExceptionHandler = OverflowExceptionInstructionAddress;
Common.SignalReturnHandler = SignalHandlerReturnAddress;
Common.L1Pointer = Thread->LookupCache->GetL1Pointer();
}
}
X86Dispatcher::~X86Dispatcher() {
FEXCore::Allocator::munmap(top_, MAX_DISPATCHER_CODE_SIZE);
}
void X86Dispatcher::InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) {
// Setup dispatcher specific pointers that need to be accessed from JIT code
{
auto &Common = Thread->CurrentFrame->Pointers.Common;
Common.DispatcherLoopTop = AbsoluteLoopTopAddress;
Common.DispatcherLoopTopFillSRA = AbsoluteLoopTopAddressFillSRA;
Common.ExitFunctionLinker = ExitFunctionLinkerAddress;
Common.ThreadStopHandlerSpillSRA = ThreadStopHandlerAddress;
Common.ThreadPauseHandlerSpillSRA = ThreadPauseHandlerAddress;
Common.UnimplementedInstructionHandler = UnimplementedInstructionAddress;
Common.OverflowExceptionHandler = OverflowExceptionInstructionAddress;
Common.SignalReturnHandler = SignalHandlerReturnAddress;
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
(uintptr_t&)Interpreter.CallbackReturn = IntCallbackReturnAddress;
}
}
std::unique_ptr<Dispatcher> Dispatcher::CreateX86(FEXCore::Context::Context *CTX, DispatcherConfig &Config) {
return std::make_unique<X86Dispatcher>(CTX, Config);
}
}
@@ -17,7 +17,8 @@ namespace FEXCore::CPU {
class X86Dispatcher final : public Dispatcher, public Xbyak::CodeGenerator {
public:
X86Dispatcher(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread, DispatcherConfig &config);
X86Dispatcher(FEXCore::Context::Context *ctx, DispatcherConfig &config);
void InitThreadPointers(FEXCore::Core::InternalThreadState *Thread) override;
virtual ~X86Dispatcher() override;
};
@@ -34,14 +34,11 @@ public:
[[nodiscard]] bool NeedsOpDispatch() override { return true; }
void CreateAsmDispatch(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread);
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
void ClearCache() override;
private:
FEXCore::Context::Context *CTX;
size_t BufferUsed;
};
@@ -40,34 +40,19 @@ class CPUBackend;
InterpreterCore::InterpreterCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalThreadState *Thread)
: CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE)
, CTX {ctx} {
{
DispatcherConfig config;
config.InterpreterDispatch = true;
#if defined(_M_X86_64)
Dispatcher = std::make_unique<X86Dispatcher>(ctx, Thread, config);
#elif defined(_M_ARM_64)
Dispatcher = std::make_unique<Arm64Dispatcher>(ctx, Thread, config);
#else
#error missing arch
#endif
DispatchPtr = Dispatcher->DispatchPtr;
CallbackPtr = Dispatcher->CallbackPtr;
auto &Interpreter = Thread->CurrentFrame->Pointers.Interpreter;
Interpreter.FragmentExecuter = reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR);
Interpreter.CallbackReturn = Dispatcher->ReturnPtr;
Interpreter.FragmentExecuter = reinterpret_cast<uint64_t>(&InterpreterOps::InterpretIR);
ClearCache();
}
void InterpreterCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
return Thread->CTX->Dispatcher->HandleSignalPause(Thread, Signal, info, ucontext);
}, true);
#ifdef _M_ARM_64
@@ -77,8 +62,7 @@ void InterpreterCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
#endif
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
InterpreterCore *Core = reinterpret_cast<InterpreterCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
return Thread->CTX->Dispatcher->HandleGuestSignal(Thread, Signal, info, ucontext, GuestAction, GuestStack);
};
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
@@ -103,7 +87,8 @@ void *InterpreterCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR:
}
void InterpreterCore::ClearCache() {
auto CodeBuffer = GetEmptyCodeBuffer();
// Calling this one is needed to setup the initial CurrentCodeBuffer
[[maybe_unused]] auto CodeBuffer = GetEmptyCodeBuffer();
BufferUsed = 0;
}
@@ -115,4 +100,9 @@ void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX) {
InterpreterCore::InitializeSignalHandlers(CTX);
}
void GetInterpreterDispatcherConfig(DispatcherConfig &config) {
config = DispatcherConfig {
.InterpreterDispatch = true
};
}
}
@@ -12,10 +12,11 @@ namespace FEXCore::Core {
namespace FEXCore::CPU {
class CPUBackend;
struct DispatcherConfig;
[[nodiscard]] std::unique_ptr<CPUBackend> CreateInterpreterCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread);
void InitializeInterpreterSignalHandlers(FEXCore::Context::Context *CTX);
void GetInterpreterDispatcherConfig(DispatcherConfig &config);
} // namespace FEXCore::CPU
@@ -12,7 +12,7 @@ namespace FEXCore::CPU {
uint64_t Arm64JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
switch (Op) {
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
return Dispatcher->ExitFunctionLinkerAddress;
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
break;
default:
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
@@ -73,7 +73,7 @@ DEF_OP(ExitFunction) {
uint64_t NewRIP;
if (IsInlineConstant(Op->NewRIP, &NewRIP) || IsInlineEntrypointOffset(Op->NewRIP, &NewRIP)) {
Literal l_BranchHost{Dispatcher->ExitFunctionLinkerAddress};
Literal l_BranchHost{ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker};
Literal l_BranchGuest{NewRIP};
ldr(x0, &l_BranchHost);
+16 -24
View File
@@ -415,17 +415,6 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
RegisterVectorHandlers();
RegisterEncryptionHandlers();
{
DispatcherConfig config;
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
config.StaticRegisterAssignment = ctx->Config.StaticRegisterAllocation;
Dispatcher = std::make_unique<Arm64Dispatcher>(CTX, ThreadState, config);
DispatchPtr = Dispatcher->DispatchPtr;
CallbackPtr = Dispatcher->CallbackPtr;
}
{
// Set up pointers that the JIT needs to load
@@ -464,29 +453,24 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::Intern
void Arm64JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
}, true);
CTX->SignalDelegation->RegisterHostSignalHandler(SIGBUS, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
if (!Core->Dispatcher->IsAddressInJITCode(ArchHelpers::Context::GetPc(ucontext))) {
if (!Thread->CPUBackend->IsAddressInCodeBuffer(ArchHelpers::Context::GetPc(ucontext))) {
// Wasn't a sigbus in JIT code
return false;
}
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Core->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
return FEXCore::ArchHelpers::Arm64::HandleSIGBUS(Thread->CTX->Config.ParanoidTSO(), Signal, info, ucontext);
}, true);
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
return Thread->CTX->Dispatcher->HandleSignalPause(Thread, Signal, info, ucontext);
}, true);
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
Arm64JITCore *Core = reinterpret_cast<Arm64JITCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
return Thread->CTX->Dispatcher->HandleGuestSignal(Thread, Signal, info, ucontext, GuestAction, GuestStack);
};
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
@@ -780,7 +764,7 @@ void *Arm64JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IR
return reinterpret_cast<void*>(GuestEntry);
}
uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
auto GuestRip = record[1];
@@ -789,11 +773,11 @@ uint64_t Arm64JITCore::ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuSt
if (!HostCode) {
//fmt::print("ExitFunctionLink: Aborting, {:X} not in cache\n", GuestRip);
Frame->State.rip = GuestRip;
return core->Dispatcher->AbsoluteLoopTopAddress;
return Frame->Pointers.Common.DispatcherLoopTop;
}
uintptr_t branch = (uintptr_t)(record) - 8;
auto LinkerAddress = core->Dispatcher->ExitFunctionLinkerAddress;
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
auto offset = HostCode/4 - branch/4;
if (IsInt26(offset)) {
@@ -849,4 +833,12 @@ std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx, F
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX) {
Arm64JITCore::InitializeSignalHandlers(CTX);
}
void GetArm64JITDispatcherConfig(DispatcherConfig &config) {
config = DispatcherConfig {
.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Arm64JITCore_ExitFunctionLink),
.SupportsStaticRegisterAllocation = true
};
}
}
@@ -59,10 +59,6 @@ public:
void ClearCache() override;
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const override {
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher);
}
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
void ClearRelocations() override { Relocations.clear(); }
@@ -146,8 +142,6 @@ private:
vixl::aarch64::Disassembler Disasm;
#endif
static uint64_t ExitFunctionLink(Arm64JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
void EmitDetectionString();
IR::RegisterAllocationPass *RAPass;
+2
View File
@@ -16,9 +16,11 @@ class CPUBackend;
[[nodiscard]] std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread);
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX);
void GetX86JITDispatcherConfig(DispatcherConfig &config);
[[nodiscard]] std::unique_ptr<CPUBackend> CreateArm64JITCore(FEXCore::Context::Context *ctx,
FEXCore::Core::InternalThreadState *Thread);
void InitializeArm64JITSignalHandlers(FEXCore::Context::Context *CTX);
void GetArm64JITDispatcherConfig(DispatcherConfig &config);
} // namespace FEXCore::CPU
@@ -91,7 +91,8 @@ DEF_OP(ExitFunction) {
jmp(qword[rax]);
L(l_BranchHost);
dq(Dispatcher->ExitFunctionLinkerAddress);
//FEX_TODO(this is not per thread)
dq(ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker);
L(l_BranchGuest);
dq(NewRIP);
} else {
+13 -19
View File
@@ -335,15 +335,6 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
RegisterVectorHandlers();
RegisterEncryptionHandlers();
DispatcherConfig config;
config.ExitFunctionLink = reinterpret_cast<uintptr_t>(&ExitFunctionLink);
config.ExitFunctionLinkThis = reinterpret_cast<uintptr_t>(this);
config.StaticRegisterAssignment = ctx->Config.StaticRegisterAllocation;
Dispatcher = std::make_unique<X86Dispatcher>(CTX, ThreadState, config);
DispatchPtr = Dispatcher->DispatchPtr;
CallbackPtr = Dispatcher->CallbackPtr;
{
auto &Common = ThreadState->CurrentFrame->Pointers.Common;
@@ -370,18 +361,15 @@ X86JITCore::X86JITCore(FEXCore::Context::Context *ctx, FEXCore::Core::InternalTh
void X86JITCore::InitializeSignalHandlers(FEXCore::Context::Context *CTX) {
CTX->SignalDelegation->RegisterHostSignalHandler(SIGILL, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleSIGILL(Signal, info, ucontext);
return Thread->CTX->Dispatcher->HandleSIGILL(Thread, Signal, info, ucontext);
}, true);
CTX->SignalDelegation->RegisterHostSignalHandler(SignalDelegator::SIGNAL_FOR_PAUSE, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool {
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleSignalPause(Signal, info, ucontext);
return Thread->CTX->Dispatcher->HandleSignalPause(Thread, Signal, info, ucontext);
}, true);
auto GuestSignalHandler = [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext, GuestSigAction *GuestAction, stack_t *GuestStack) -> bool {
X86JITCore *Core = reinterpret_cast<X86JITCore*>(Thread->CPUBackend.get());
return Core->Dispatcher->HandleGuestSignal(Signal, info, ucontext, GuestAction, GuestStack);
return Thread->CTX->Dispatcher->HandleGuestSignal(Thread, Signal, info, ucontext, GuestAction, GuestStack);
};
for (uint32_t Signal = 0; Signal <= SignalDelegator::MAX_SIGNALS; ++Signal) {
@@ -600,7 +588,7 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
cmp(dword [rax + (offsetof(FEXCore::Context::Context, Config.RunningMode))], 0);
je(RunBlock);
// Else we need to pause now
mov(rax, Dispatcher->ThreadPauseHandlerAddress);
mov(rax, offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.ThreadPauseHandlerSpillSRA));
jmp(rax);
ud2();
@@ -744,7 +732,7 @@ void *X86JITCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRLi
return GuestEntry;
}
uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
static uint64_t X86JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record) {
auto Thread = Frame->Thread;
auto GuestRip = record[1];
@@ -752,10 +740,10 @@ uint64_t X86JITCore::ExitFunctionLink(X86JITCore *core, FEXCore::Core::CpuStateF
if (!HostCode) {
Thread->CurrentFrame->State.rip = GuestRip;
return core->Dispatcher->AbsoluteLoopTopAddress;
return Frame->Pointers.Common.DispatcherLoopTop;
}
auto LinkerAddress = core->Dispatcher->ExitFunctionLinkerAddress;
auto LinkerAddress = Frame->Pointers.Common.ExitFunctionLinker;
Thread->LookupCache->AddBlockLink(GuestRip, (uintptr_t)record, [record, LinkerAddress]{
// undo the link
record[0] = LinkerAddress;
@@ -769,6 +757,12 @@ std::unique_ptr<CPUBackend> CreateX86JITCore(FEXCore::Context::Context *ctx, FEX
return std::make_unique<X86JITCore>(ctx, Thread);
}
void GetX86JITDispatcherConfig(DispatcherConfig &config) {
config = DispatcherConfig {
.ExitFunctionLink = reinterpret_cast<uintptr_t>(&X86JITCore_ExitFunctionLink)
};
}
void InitializeX86JITSignalHandlers(FEXCore::Context::Context *CTX) {
X86JITCore::InitializeSignalHandlers(CTX);
}
@@ -67,10 +67,6 @@ public:
void ClearCache() override;
bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const override {
return Dispatcher->IsAddressInJITCode(Address, IncludeDispatcher);
}
static void InitializeSignalHandlers(FEXCore::Context::Context *CTX);
void ClearRelocations() override { Relocations.clear(); }
@@ -197,7 +193,7 @@ private:
bool GetSamplingData {true};
#endif
static uint64_t ExitFunctionLink(X86JITCore* code, FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
static uint64_t ExitFunctionLink(FEXCore::Core::CpuStateFrame *Frame, uint64_t *record);
// This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory
void EmitDetectionString();
@@ -11,7 +11,7 @@ namespace FEXCore::CPU {
uint64_t X86JITCore::GetNamedSymbolLiteral(FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol Op) {
switch (Op) {
case FEXCore::CPU::RelocNamedSymbolLiteral::NamedSymbol::SYMBOL_LITERAL_EXITFUNCTION_LINKER:
return Dispatcher->ExitFunctionLinkerAddress;
return ThreadState->CurrentFrame->Pointers.Common.ExitFunctionLinker;
break;
default:
ERROR_AND_DIE_FMT("Unknown named symbol literal: {}", static_cast<uint32_t>(Op));
+1 -15
View File
@@ -33,8 +33,6 @@ namespace CodeSerialize {
}
namespace CPU {
class Dispatcher;
class CPUBackend {
public:
struct CodeBuffer {
@@ -110,23 +108,15 @@ class Dispatcher;
*/
[[nodiscard]] virtual bool NeedsOpDispatch() = 0;
void ExecuteDispatch(FEXCore::Core::CpuStateFrame *Frame) {
DispatchPtr(Frame);
}
virtual void ClearCache() {}
virtual bool IsAddressInJITCode(uint64_t Address, bool IncludeDispatcher = true) const { return false; }
/**
* @brief Clear any relocations after JIT compiling
*/
virtual void ClearRelocations() {}
using AsmDispatch = FEX_NAKED void(*)(FEXCore::Core::CpuStateFrame *Frame);
using JITCallback = FEX_NAKED void(*)(FEXCore::Core::CpuStateFrame *Frame, uint64_t RIP);
using IntCallbackReturn = FEX_NAKED void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
bool IsAddressInCodeBuffer(uintptr_t Address) const;
JITCallback CallbackPtr{};
protected:
FEXCore::Core::InternalThreadState *ThreadState;
@@ -136,10 +126,6 @@ class Dispatcher;
// This is the current code buffer that we are tracking
CodeBuffer *CurrentCodeBuffer{};
AsmDispatch DispatchPtr{};
std::unique_ptr<FEXCore::CPU::Dispatcher> Dispatcher;
private:
CodeBuffer AllocateNewCodeBuffer(size_t Size);
void FreeCodeBuffer(CodeBuffer Buffer);
+11 -1
View File
@@ -114,12 +114,14 @@ namespace FEXCore::Core {
* @{ */
uint64_t DispatcherLoopTop{};
uint64_t DispatcherLoopTopFillSRA{};
uint64_t ExitFunctionLinker{};
uint64_t ThreadStopHandlerSpillSRA{};
uint64_t ThreadPauseHandlerSpillSRA{};
uint64_t UnimplementedInstructionHandler{};
uint64_t OverflowExceptionHandler{};
uint64_t SignalReturnHandler{};
uint64_t L1Pointer{};
uint64_t L2Pointer{};
/** @} */
} Common;
@@ -149,7 +151,8 @@ namespace FEXCore::Core {
struct {
uint64_t FragmentExecuter;
CPU::CPUBackend::IntCallbackReturn CallbackReturn;
using IntCallbackReturn = void(*)(FEXCore::Core::InternalThreadState *Thread, volatile void *Host_RSP);
IntCallbackReturn CallbackReturn;
} Interpreter;
};
};
@@ -176,6 +179,13 @@ namespace FEXCore::Core {
uint32_t SignalHandlerRefCounter{};
struct SynchronousFaultDataStruct {
bool FaultToTopAndGeneratedException{};
uint32_t TrapNo;
uint32_t err_code;
uint32_t si_code;
} SynchronousFaultData;
InternalThreadState* Thread;
// Pointers that the JIT needs to load to remove relocations
@@ -108,7 +108,7 @@ namespace FEXCore::Core {
std::shared_mutex ObjectCacheRefCounter{};
bool DestroyedByParent{false}; // Should the parent destroy this thread, or it destory itself
alignas(16) FEXCore::Core::CpuStateFrame BaseFrameState{};
};
+2 -1
View File
@@ -214,7 +214,8 @@ namespace HostFactory {
}
void HostCore::Initialize() {
DispatchPtr = getCurr<CPUBackend::AsmDispatch>();
//FEX_TODO(FIXME)
//DispatchPtr = getCurr<AsmDispatch>();
// x86-64 ABI has the stack aligned when /call/ happens
// Which means the destination has a misaligned stack at that point
push(rbx);