#include #include #include #include #include #include #include #ifdef _M_X86_64 #include #include #endif #include #include #include #include #include #include #include #include #include using namespace Xbyak; namespace HostFactory { #ifdef _M_X86_64 static inline int modify_ldt(int func, void *ldt) { return ::syscall(SYS_modify_ldt, func, ldt, sizeof(struct user_desc)); } class HostCore final : public FEXCore::CPU::CPUBackend, public Xbyak::CodeGenerator { public: explicit HostCore(FEXCore::Context::Context* CTX, FEXCore::Core::InternalThreadState *Thread, bool Fallback); ~HostCore() override; std::string GetName() override { return "Host Core"; } void* CompileCode(uint64_t Entry, FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) override; void *MapRegion(void *HostPtr, uint64_t VirtualGuestPtr, uint64_t Size) override { return HostPtr; } void Initialize() override; bool NeedsOpDispatch() override { return false; } bool HandleSIGSEGV(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext); private: FEX_CONFIG_OPT(Is64BitMode, IS64BIT_MODE); int CodeSegmentEntry{}; int GlobalCodeSegmentEntry{}; uint64_t GetCodeSegmentEntryLocation; uint64_t ReturningStackLocation; uint64_t ThreadStopHandlerAddress; void Setup32BitCodeSegment(); uint32_t MakeSelector(int Segment, bool LDT) const { // Selector Index, Table Indicator (1 = LDT, 0 = GDT), CPL (3 = userland) return (Segment << 3) | ((uint32_t)LDT << 2) | 3; }; }; HostCore::~HostCore() { } HostCore::HostCore(FEXCore::Context::Context* CTX, FEXCore::Core::InternalThreadState *Thread, bool Fallback) : CPUBackend(Thread, 0, 0) , CodeGenerator(4096) { FEXCore::Context::RegisterHostSignalHandler(CTX, SIGSEGV, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool { auto InternalThread = Thread; HostCore *Core = reinterpret_cast(InternalThread->CPUBackend.get()); return Core->HandleSIGSEGV(Thread, Signal, info, ucontext); }, true ); FEXCore::Context::RegisterHostSignalHandler(CTX, 63, [](FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) -> bool { return true; }, true); Setup32BitCodeSegment(); } bool HostCore::HandleSIGSEGV(FEXCore::Core::InternalThreadState *Thread, int Signal, void *info, void *ucontext) { ucontext_t* _context = (ucontext_t*)ucontext; mcontext_t* _mcontext = &_context->uc_mcontext; // Check our current instruction that we just executed to ensure it was an HLT uint8_t *Inst{}; Inst = reinterpret_cast(_mcontext->gregs[REG_RIP]); if (!Is64BitMode()) { if (_mcontext->gregs[REG_RIP] == GetCodeSegmentEntryLocation) { // Backup the CSGSFS register GlobalCodeSegmentEntry = _mcontext->gregs[REG_CSGSFS]; // Skip past this hlt and keep running _mcontext->gregs[REG_RIP] += 1; return true; } } constexpr uint8_t HLT = 0xF4; if (Inst[0] != HLT) { return false; } // Store our host state in to the guest for testing against Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RAX] = _mcontext->gregs[REG_RAX]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RBX] = _mcontext->gregs[REG_RBX]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RCX] = _mcontext->gregs[REG_RCX]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDX] = _mcontext->gregs[REG_RDX]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RBP] = _mcontext->gregs[REG_RBP]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSI] = _mcontext->gregs[REG_RSI]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RDI] = _mcontext->gregs[REG_RDI]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_RSP] = _mcontext->gregs[REG_RSP]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_R8] = _mcontext->gregs[REG_R8]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_R9] = _mcontext->gregs[REG_R9]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_R10] = _mcontext->gregs[REG_R10]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_R11] = _mcontext->gregs[REG_R11]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_R12] = _mcontext->gregs[REG_R12]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_R13] = _mcontext->gregs[REG_R13]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_R14] = _mcontext->gregs[REG_R14]; Thread->CurrentFrame->State.gregs[FEXCore::X86State::REG_R15] = _mcontext->gregs[REG_R15]; Thread->CurrentFrame->State.rip = _mcontext->gregs[REG_RIP]; for (size_t i = 0; i < 16; ++i) { memcpy(&Thread->CurrentFrame->State.xmm[i], &_mcontext->fpregs->_xmm[i], sizeof(_mcontext->fpregs->_xmm[0])); } uint16_t CurrentOffset = (_mcontext->fpregs->swd >> 11) & 7; for (size_t i = 0; i < 8; ++i) { memcpy(&Thread->CurrentFrame->State.mm[(i + CurrentOffset) % 8], &_mcontext->fpregs->_st[i], sizeof(_mcontext->fpregs->_st[0])); } // Our thread is stopping // We don't care about anything at this point // Set the stack to our starting location when we entered the JIT and get out safely _mcontext->gregs[REG_RSP] = ReturningStackLocation; // Set the new PC _mcontext->gregs[REG_RIP] = ThreadStopHandlerAddress; if (!Is64BitMode()) { // Unset code segment so we can jump back in to 64-bit mode _mcontext->gregs[REG_CSGSFS] = GlobalCodeSegmentEntry; } return true; } void HostCore::Setup32BitCodeSegment() { if (Is64BitMode()) { return; } struct user_desc ldt{}; ldt.entry_number = 1; // This is where HarnessCodeLoader loads code to ldt.base_addr = 0; ldt.limit = ~0U; // No limit ldt.seg_32bit = 1; // 32-bit ldt.contents = MODIFY_LDT_CONTENTS_CODE; ldt.read_exec_only = 0; ldt.limit_in_pages = 1; ldt.seg_not_present = 0; ldt.useable = 1; ldt.lm = 0; // Not-64-bit int Res = modify_ldt(0x11, &ldt); if (Res == -1) { LogMan::Msg::EFmt("Couldn't load 32-bit LDT"); return; } CodeSegmentEntry = MakeSelector(ldt.entry_number, 1); // Make the data segment follow directly after the code segment // Overlapping region makes it read/write ldt.entry_number = 2; // This is where HarnessCodeLoader loads code to ldt.base_addr = 0; ldt.limit = ~0U; // No limit ldt.seg_32bit = 1; // 32-bit ldt.contents = MODIFY_LDT_CONTENTS_DATA; ldt.read_exec_only = 0; ldt.limit_in_pages = 1; ldt.seg_not_present = 0; ldt.useable = 1; ldt.lm = 0; // Not-64-bit Res = modify_ldt(0x11, &ldt); if (Res == -1) { LogMan::Msg::EFmt("Couldn't load 32-bit LDT"); return; } // Stack entry overlapping data ldt.entry_number = 3; // This is where HarnessCodeLoader loads code to ldt.base_addr = 0; ldt.limit = ~0U; // No limit ldt.seg_32bit = 1; // 32-bit ldt.contents = MODIFY_LDT_CONTENTS_STACK; ldt.read_exec_only = 0; ldt.limit_in_pages = 1; ldt.seg_not_present = 0; ldt.useable = 1; ldt.lm = 0; // Not-64-bit Res = modify_ldt(0x11, &ldt); if (Res == -1) { LogMan::Msg::EFmt("Couldn't load 32-bit LDT"); return; } } void HostCore::Initialize() { //FEX_TODO(FIXME) //DispatchPtr = getCurr(); // x86-64 ABI has the stack aligned when /call/ happens // Which means the destination has a misaligned stack at that point push(rbx); push(rbp); push(r12); push(r13); push(r14); push(r15); sub(rsp, 8); // Save this stack pointer so we can cleanly shutdown the emulation with a long jump // regardless of where we were in the stack mov(rax, (uint64_t)&ReturningStackLocation); mov(qword [rax], rsp); mov (rax, 0); mov (rbx, 0); mov (rcx, 0); mov (rdx, 0); mov (rbp, 0); mov (rsi, 0); mov (r8, 0); mov (r9, 0); mov (r10, 0); mov (r11, 0); mov (r12, 0); mov (r13, 0); mov (r14, 0); mov (r15, 0); finit(); if (Is64BitMode()) { push(qword [rdi + offsetof(FEXCore::Core::CPUState, rip)]); // Load our RIP // RSP won't be set to zero here but should be fine ret(); } else { // Far call needs to go through a gate // This is setup just like the following packing // { // uint32_t RIP; // uint16_t CodeSegment; // } GetCodeSegmentEntryLocation = getCurr(); hlt(); Label Gate{}; jmpf(ptr [rip + Gate]); L(Gate); dd(0x1'0000); // This is a 32-bit offset from the start of the gate. We start at 0x1'0000 + 0 dw(CodeSegmentEntry); } ThreadStopHandlerAddress = getCurr(); add(rsp, 8); pop(r15); pop(r14); pop(r13); pop(r12); pop(rbp); pop(rbx); ret(); ready(); } void* HostCore::CompileCode(uint64_t Entry, [[maybe_unused]] FEXCore::IR::IRListView const *IR, FEXCore::Core::DebugData *DebugData, FEXCore::IR::RegisterAllocationData *RAData) { return nullptr; } std::unique_ptr CPUCreationFactory(FEXCore::Context::Context* CTX, FEXCore::Core::InternalThreadState *Thread) { return std::make_unique(CTX, Thread, false); } #else std::unique_ptr CPUCreationFactory(FEXCore::Context::Context* CTX, FEXCore::Core::InternalThreadState *Thread) { LOGMAN_MSG_A_FMT("HostCPU factory doesn't exist for this host"); return nullptr; } #endif }