diff --git a/External/FEXCore/Source/Interface/Core/Core.cpp b/External/FEXCore/Source/Interface/Core/Core.cpp index b336f15ce..a5ee3411f 100644 --- a/External/FEXCore/Source/Interface/Core/Core.cpp +++ b/External/FEXCore/Source/Interface/Core/Core.cpp @@ -192,29 +192,6 @@ namespace FEXCore::Context { } } - static FEXCore::Core::CPUState CreateDefaultCPUState() { - FEXCore::Core::CPUState NewThreadState{}; - - // Initialize default CPU state - NewThreadState.rip = ~0ULL; - for (auto& greg : NewThreadState.gregs) { - greg = 0; - } - - for (auto& xmm : NewThreadState.xmm.avx.data) { - xmm[0] = 0xDEADBEEFULL; - xmm[1] = 0xBAD0DAD1ULL; - xmm[2] = 0xDEADCAFEULL; - xmm[3] = 0xBAD2CAD3ULL; - } - memset(NewThreadState.flags, 0, Core::CPUState::NUM_EFLAG_BITS); - NewThreadState.flags[1] = 1; - NewThreadState.flags[9] = 1; - NewThreadState.FCW = 0x37F; - NewThreadState.FTW = 0xFFFF; - return NewThreadState; - } - uint64_t ContextImpl::RestoreRIPFromHostPC(FEXCore::Core::InternalThreadState *Thread, uint64_t HostPC) { const auto Frame = Thread->CurrentFrame; const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader; @@ -315,8 +292,7 @@ namespace FEXCore::Context { using namespace FEXCore::Core; - FEXCore::Core::CPUState NewThreadState = CreateDefaultCPUState(); - FEXCore::Core::InternalThreadState *Thread = CreateThread(&NewThreadState, 0); + FEXCore::Core::InternalThreadState *Thread = CreateThread(nullptr, 0); // We are the parent thread ParentThread = Thread; @@ -625,7 +601,9 @@ namespace FEXCore::Context { FEXCore::Core::InternalThreadState *Thread = new FEXCore::Core::InternalThreadState{}; // Copy over the new thread state to the new object - memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState)); + if (NewThreadState) { + memcpy(Thread->CurrentFrame, NewThreadState, sizeof(FEXCore::Core::CPUState)); + } Thread->CurrentFrame->Thread = Thread; // Set up the thread manager state @@ -635,7 +613,7 @@ namespace FEXCore::Context { InitializeThreadData(Thread); Thread->CurrentFrame->State.DeferredSignalRefCount.Store(0); - Thread->CurrentFrame->State.DeferredSignalFaultAddress = reinterpret_cast*>(FEXCore::Allocator::VirtualAlloc(4096)); + Thread->CurrentFrame->State.DeferredSignalFaultAddress = reinterpret_cast*>(FEXCore::Allocator::VirtualAlloc(4096)); // Insert after the Thread object has been fully initialized { @@ -1230,6 +1208,8 @@ namespace FEXCore::Context { void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length) { // Potential deferred since Thread might not be valid. + // Thread object isn't valid very early in frontend's initialization. + // To be more optimal the frontend should provide this code with a valid Thread object earlier. ScopedPotentialDeferredSignalWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex, Thread); InvalidateGuestCodeRangeInternal(this, Start, Length); @@ -1237,6 +1217,8 @@ namespace FEXCore::Context { void ContextImpl::InvalidateGuestCodeRange(FEXCore::Core::InternalThreadState *Thread, uint64_t Start, uint64_t Length, std::function CallAfter) { // Potential deferred since Thread might not be valid. + // Thread object isn't valid very early in frontend's initialization. + // To be more optimal the frontend should provide this code with a valid Thread object earlier. ScopedPotentialDeferredSignalWithUniqueLock CodeInvalidationLock(CodeInvalidationMutex, Thread); InvalidateGuestCodeRangeInternal(this, Start, Length); diff --git a/External/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp b/External/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp index 5d34f0f8b..db326ccd4 100644 --- a/External/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp +++ b/External/FEXCore/Source/Interface/Core/Dispatcher/Arm64Dispatcher.cpp @@ -213,6 +213,7 @@ void Arm64Dispatcher::EmitDispatcher() { subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x1, ARMEmitter::XReg::x1, 1); str(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)); + // Trigger segfault if any deferred signals are pending ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress)); str(ARMEmitter::XReg::zr, ARMEmitter::XReg::x1, 0); @@ -248,6 +249,7 @@ void Arm64Dispatcher::EmitDispatcher() { subs(ARMEmitter::Size::i64Bit, ARMEmitter::XReg::x0, ARMEmitter::XReg::x0, 1); str(ARMEmitter::XReg::x0, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount)); + // Trigger segfault if any deferred signals are pending ldr(TMP1, STATE, offsetof(FEXCore::Core::CPUState, DeferredSignalFaultAddress)); str(ARMEmitter::XReg::zr, TMP1, 0); diff --git a/External/FEXCore/Source/Utils/Allocator/64BitAllocator.cpp b/External/FEXCore/Source/Utils/Allocator/64BitAllocator.cpp index a13d85298..50e7117dd 100644 --- a/External/FEXCore/Source/Utils/Allocator/64BitAllocator.cpp +++ b/External/FEXCore/Source/Utils/Allocator/64BitAllocator.cpp @@ -29,17 +29,14 @@ namespace Alloc::OSAllocator { - struct TLSData { - FEXCore::Core::InternalThreadState *Thread; - }; - thread_local TLSData TLS{}; + thread_local FEXCore::Core::InternalThreadState *TLSThread{}; void RegisterTLSData(FEXCore::Core::InternalThreadState *Thread) { - TLS.Thread = Thread; + TLSThread = Thread; } void UninstallTLSData(FEXCore::Core::InternalThreadState *Thread) { - TLS.Thread = nullptr; + TLSThread = nullptr; } class OSAllocator_64Bit final : public Alloc::HostAllocator { @@ -261,7 +258,7 @@ void *OSAllocator_64Bit::Mmap(void *addr, size_t length, int prot, int flags, in size_t NumberOfPages = length / FHU::FEX_PAGE_SIZE; // This needs a mutex to be thread safe - FEXCore::ScopedPotentialDeferredSignalWithMutex lk(AllocationMutex, TLS.Thread); + FEXCore::ScopedPotentialDeferredSignalWithMutex lk(AllocationMutex, TLSThread); uint64_t AllocatedOffset{}; LiveVMARegion *LiveRegion{}; @@ -449,7 +446,7 @@ int OSAllocator_64Bit::Munmap(void *addr, size_t length) { } // This needs a mutex to be thread safe - FEXCore::ScopedPotentialDeferredSignalWithMutex lk(AllocationMutex, TLS.Thread); + FEXCore::ScopedPotentialDeferredSignalWithMutex lk(AllocationMutex, TLSThread); length = FEXCore::AlignUp(length, FHU::FEX_PAGE_SIZE); @@ -574,7 +571,7 @@ OSAllocator_64Bit::OSAllocator_64Bit() { OSAllocator_64Bit::~OSAllocator_64Bit() { // This needs a mutex to be thread safe - FEXCore::ScopedPotentialDeferredSignalWithMutex lk(AllocationMutex, TLS.Thread); + FEXCore::ScopedPotentialDeferredSignalWithMutex lk(AllocationMutex, TLSThread); // Walk the pages and deallocate // First walk the live regions diff --git a/External/FEXCore/include/FEXCore/Core/CoreState.h b/External/FEXCore/include/FEXCore/Core/CoreState.h index 9ad024d0e..c292dc6a4 100644 --- a/External/FEXCore/include/FEXCore/Core/CoreState.h +++ b/External/FEXCore/include/FEXCore/Core/CoreState.h @@ -6,23 +6,20 @@ #include #include +#include #include #include #include namespace FEXCore::Core { - // This is defined here instead of a helper because it is a massive footgun in behaviour. - // Don't use this instead of std::atomic unless you really know what you're doing. - // Increment and decrement can visibily tear and only uses relaxed atomic behaviour internally. - // In particular decrement and increment can tear if a signal is received half-way through - // and you need to be aware of that when using this. + // Wrapper around std::atomic using std::memory_order_relaxed. + // This allows compilers to emit more performant code at the expense of visibly tearing. + // In particular, increments/decrements may visibly tear if a signal is received half-way through. // + // Prefer std::atomic with default memory ordering unless you really know what you're doing. // Primarily this ensure program ordering when signals are concerned. - // - // This casts its internal object to std::atomic internally so that compilers can't reorder operations around it. - // This is necessary with how it is used with deferring signals. template - class MoveableNonatomicRefCounter { + class NonAtomicRefCounter { public: void Increment(T Value) { // Specifically avoiding fetch_add here because that will turn in to ldxr+stxr or lock xadd. @@ -35,7 +32,6 @@ namespace FEXCore::Core { // // x86-64 ex: // inc qword [rax]; - auto AtomicVariable = std::atomic_ref(Variable); auto Current = AtomicVariable.load(std::memory_order_relaxed); AtomicVariable.store(Current + Value, std::memory_order_relaxed); } @@ -43,7 +39,7 @@ namespace FEXCore::Core { // Returns original value. // x86-64 needs to know the result on decrement. T Decrement(T Value) { - // Specifically avoiding fetch_sub here because that will turn in to ldxr+stxr or lock xadd. + // Specifically avoiding fetch_sub here because that will turn into ldxr+stxr or lock xadd. // FEX very specifically wants to use simple loadstore instructions for this // // ARM64 ex: @@ -53,29 +49,25 @@ namespace FEXCore::Core { // // x86-64 ex: // dec qword [rax]; - auto AtomicVariable = std::atomic_ref(Variable); auto Current = AtomicVariable.load(std::memory_order_relaxed); AtomicVariable.store(Current - Value, std::memory_order_relaxed); return Current; } T Load() const { - auto AtomicVariable = std::atomic_ref(Variable); return AtomicVariable.load(std::memory_order_relaxed); } void Store(T Value) { - auto AtomicVariable = std::atomic_ref(Variable); AtomicVariable.store(Value, std::memory_order_relaxed); } private: - // Internal variable can't be std::atomic because that doesn't have a move constructor and is non-pod. - T Variable; + std::atomic AtomicVariable; }; - static_assert(std::is_standard_layout_v>, "Needs to be standard layout"); - static_assert(std::is_trivially_copyable_v>, "needs to be trivially copyable"); - static_assert(sizeof(MoveableNonatomicRefCounter) == sizeof(uint64_t), "Needs to be correct size"); + static_assert(std::is_standard_layout_v>, "Needs to be standard layout"); + static_assert(std::is_trivially_copyable_v>, "needs to be trivially copyable"); + static_assert(sizeof(NonAtomicRefCounter) == sizeof(uint64_t), "Needs to be correct size"); struct FEX_PACKED CPUState { // Allows more efficient handling of the register @@ -118,9 +110,10 @@ namespace FEXCore::Core { uint32_t _pad2[1]; // Reference counter for FEX's per-thread deferred signals. - MoveableNonatomicRefCounter DeferredSignalRefCount; - // Since this memory region is thread local, it should only be accessed with relaxed atomics. - MoveableNonatomicRefCounter *DeferredSignalFaultAddress; + // Counts the nesting depth of program sections that cause signals to be deferred. + NonAtomicRefCounter DeferredSignalRefCount; + // Since this memory region is thread local, we use NonAtomicRefCounter for fast atomic access. + NonAtomicRefCounter *DeferredSignalFaultAddress; static constexpr size_t FLAG_SIZE = sizeof(flags[0]); static constexpr size_t GDT_SIZE = sizeof(gdt[0]); @@ -136,7 +129,26 @@ namespace FEXCore::Core { static constexpr size_t NUM_GPRS = sizeof(gregs) / GPR_REG_SIZE; static constexpr size_t NUM_XMMS = sizeof(xmm) / XMM_AVX_REG_SIZE; static constexpr size_t NUM_MMS = sizeof(mm) / MM_REG_SIZE; + CPUState() { + // Initialize default CPU state + rip = ~0ULL; + memset(gregs, 0, sizeof(gregs)); + + for (auto& xmm : xmm.avx.data) { + xmm[0] = 0xDEADBEEFULL; + xmm[1] = 0xBAD0DAD1ULL; + xmm[2] = 0xDEADCAFEULL; + xmm[3] = 0xBAD2CAD3ULL; + } + memset(&flags, 0, Core::CPUState::NUM_EFLAG_BITS); + flags[1] = 1; ///< Reserved - Always 1. + flags[9] = 1; ///< Interrupt flag - Always 1. + FCW = 0x37F; + FTW = 0xFFFF; + } }; + static_assert(std::is_trivially_copyable_v, "Needs to be trivial"); + static_assert(std::is_standard_layout_v, "This needs to be standard layout"); static_assert(offsetof(CPUState, xmm) % 32 == 0, "xmm needs to be 256-bit aligned!"); static_assert(offsetof(CPUState, mm) % 16 == 0, "mm needs to be 128-bit aligned!"); static_assert(offsetof(CPUState, DeferredSignalRefCount) % 8 == 0, "Needs to be 8-byte aligned"); diff --git a/External/FEXCore/include/FEXCore/Debug/InternalThreadState.h b/External/FEXCore/include/FEXCore/Debug/InternalThreadState.h index 84441ea90..aa1f63152 100644 --- a/External/FEXCore/include/FEXCore/Debug/InternalThreadState.h +++ b/External/FEXCore/include/FEXCore/Debug/InternalThreadState.h @@ -105,10 +105,14 @@ namespace FEXCore::Core { bool DestroyedByParent{false}; // Should the parent destroy this thread, or it destory itself struct DeferredSignalState { +#ifndef _WIN32 siginfo_t Info; +#endif int Signal; }; + // Queue of thread local signal frames that have been deferred. + // Async signals aren't guaranteed to be delivered in any particular order, but FEX treats them as FILO. fextl::vector DeferredSignalFrames; // BaseFrameState should always be at the end. diff --git a/External/FEXCore/include/FEXCore/Utils/DeferredSignalMutex.h b/External/FEXCore/include/FEXCore/Utils/DeferredSignalMutex.h index e0293bd03..7fb2c6fbf 100644 --- a/External/FEXCore/include/FEXCore/Utils/DeferredSignalMutex.h +++ b/External/FEXCore/include/FEXCore/Utils/DeferredSignalMutex.h @@ -103,7 +103,7 @@ namespace FEXCore { if (Thread) { #ifdef _M_X86_64 // Needs to be atomic so that operations can't end up getting reordered around this. - // Without this, the recount and the signal access could get reordered. + // Without this, the refcount and the signal access could get reordered. auto Result = Thread->CurrentFrame->State.DeferredSignalRefCount.Decrement(1); // X86-64 must do an additional check around the store. diff --git a/Source/Tools/FEXLoader/LinuxSyscalls/SignalDelegator.cpp b/Source/Tools/FEXLoader/LinuxSyscalls/SignalDelegator.cpp index 3053f687f..f461184f2 100644 --- a/Source/Tools/FEXLoader/LinuxSyscalls/SignalDelegator.cpp +++ b/Source/Tools/FEXLoader/LinuxSyscalls/SignalDelegator.cpp @@ -1383,7 +1383,7 @@ namespace FEX::HLE { /** @} */ - static bool IsAsyncSignal(siginfo_t* Info, int Signal) { + static bool IsAsyncSignal(const siginfo_t* Info, int Signal) { if (Info->si_code <= SI_USER) { // If the signal is not from the kernel then it is always async. // This is because synchronous signals can be sent through tgkill,sigqueue and other methods. @@ -1414,44 +1414,42 @@ namespace FEX::HLE { constexpr bool SupportDeferredSignals = true; if (SupportDeferredSignals) { - auto RefCount = Thread->CurrentFrame->State.DeferredSignalRefCount.Load(); + auto MustDeferSignal = (Thread->CurrentFrame->State.DeferredSignalRefCount.Load() != 0); if (Signal == SIGSEGV && SigInfo.si_code == SEGV_ACCERR && SigInfo.si_addr == reinterpret_cast(Thread->CurrentFrame->State.DeferredSignalFaultAddress)) { - if (RefCount == 0) { - // We are faulting with an attempt to check deferred signals. + if (!MustDeferSignal) { + // We just reached the end of the outermost signal-deferring section and faulted to check for pending signals. // Pull a signal frame off the stack. mprotect(reinterpret_cast(Thread->CurrentFrame->State.DeferredSignalFaultAddress), 4096, PROT_READ | PROT_WRITE); - if (Thread->DeferredSignalFrames.size() == 0) { + if (Thread->DeferredSignalFrames.empty()) { // No signals to defer. Just set the fault page back to RW and continue execution. // This occurs as a minor race condition between the refcount decrement and the access to the fault page. return; } - else { - auto Top = Thread->DeferredSignalFrames.back(); - Signal = Top.Signal; - SigInfo = Top.Info; - Thread->DeferredSignalFrames.pop_back(); - // Now that we are handling a deferred signal state. mprotect the fault page with RW permissions. - // This puts FEX in a state in-which we are *permanently* deferring signals and /not/ checking them. - // - // In order to return /back/ to a sane state, we wait for the rt_sigreturn to happen. - // rt_sigreturn will check if there are any more deferred signals to handle - // - If there are deferred signals - // - mprotect back to PROT_NONE - // - sigreturn will trampoline out to the previous fault address check, SIGSEGV and restart - // - If there are *no* deferred signals - // - No need to mprotect, it is already RW - } + auto Top = Thread->DeferredSignalFrames.back(); + Signal = Top.Signal; + SigInfo = Top.Info; + Thread->DeferredSignalFrames.pop_back(); + + // Until we re-protect the page to PROT_NONE, FEX will now *permanently* defer signals and /not/ check them. + // + // In order to return /back/ to a sane state, we wait for the rt_sigreturn to happen. + // rt_sigreturn will check if there are any more deferred signals to handle + // - If there are deferred signals + // - mprotect back to PROT_NONE + // - sigreturn will trampoline out to the previous fault address check, SIGSEGV and restart + // - If there are *no* deferred signals + // - No need to mprotect, it is already RW } else { #ifdef _M_ARM_64 - // If RefCount != 0 then that means we hit an access with stacked deferred signals. - // Increment the PC past the `str zr, [x1]` to continue code execution. + // If RefCount != 0 then that means we hit an access with nested signal-deferring sections. + // Increment the PC past the `str zr, [x1]` to continue code execution until we reach the outermost section. ArchHelpers::Context::SetPc(UContext, ArchHelpers::Context::GetPc(UContext) + 4); return; #else @@ -1462,19 +1460,20 @@ namespace FEX::HLE { } } else { - if (IsAsyncSignal(&SigInfo, Signal) && RefCount) { - // If the signal is asynchronous signal (as determined by si_code) and FEX is in a state of needing - // to defer the signal, then make sure to add the signal to the thread's signal queue. + if (IsAsyncSignal(&SigInfo, Signal) && MustDeferSignal) { + // If the signal is asynchronous (as determined by si_code) and FEX is in a state of needing + // to defer the signal, then add the signal to the thread's signal queue. + LOGMAN_THROW_A_FMT(Thread->DeferredSignalFrames.size() != Thread->DeferredSignalFrames.capacity(), + "Deferred signals vector hit capacity size. This will likely crash! Asserting now!"); Thread->DeferredSignalFrames.emplace_back(FEXCore::Core::InternalThreadState::DeferredSignalState { .Info = SigInfo, .Signal = Signal, }); // Now update the faulting page permissions so it will fault on write. - // Just in case of multiple deferred signals, keep track of protection state. mprotect(reinterpret_cast(Thread->CurrentFrame->State.DeferredSignalFaultAddress), 4096, PROT_NONE); - // Early exit, we are going to return to this later. + // Postpone the remainder of signal handling logic until we process the SIGSEGV triggered by writing to DeferredSignalFaultAddress. return; } } @@ -1553,7 +1552,6 @@ namespace FEX::HLE { // If the signal wasn't sent by the kernel then we need to reraise it. // This is necessary since returning from this signal handler now might just continue executing. // eg: If sent from tgkill then the signal gets dropped and returns. - Thread->CurrentFrame->State.DeferredSignalRefCount.Store(0); FHU::Syscalls::tgkill(::getpid(), FHU::Syscalls::gettid(), Signal); } } diff --git a/Source/Tools/FEXLoader/LinuxSyscalls/SyscallsSMCTracking.cpp b/Source/Tools/FEXLoader/LinuxSyscalls/SyscallsSMCTracking.cpp index da15359ad..d8c8b3489 100644 --- a/Source/Tools/FEXLoader/LinuxSyscalls/SyscallsSMCTracking.cpp +++ b/Source/Tools/FEXLoader/LinuxSyscalls/SyscallsSMCTracking.cpp @@ -191,7 +191,9 @@ void SyscallHandler::TrackMmap(FEXCore::Core::InternalThreadState *Thread, uintp } { - // Frontend calls this with nullptr Thread. + // NOTE: Frontend calls this with a nullptr Thread during initialization, but + // providing this code with a valid Thread object earlier would allow + // us to be more optimal by using ScopedDeferredSignalWithUniqueLock instead FEXCore::ScopedPotentialDeferredSignalWithUniqueLock lk(VMATracking.Mutex, Thread); static uint64_t AnonSharedId = 1; @@ -239,7 +241,9 @@ void SyscallHandler::TrackMunmap(FEXCore::Core::InternalThreadState *Thread, uin Size = FEXCore::AlignUp(Size, FHU::FEX_PAGE_SIZE); { - // Frontend calls this with nullptr Thread. + // Frontend calls this with nullptr Thread during initialization. + // This is why `ScopedPotentialDeferredSignalWithUniqueLock` is used here. + // To be more optimal the frontend should provide this code with a valid Thread object earlier. FEXCore::ScopedPotentialDeferredSignalWithUniqueLock lk(VMATracking.Mutex, Thread); VMATracking.ClearUnsafe(CTX, Base, Size); diff --git a/docs/DeferredSignals.md b/docs/DeferredSignals.md index ac6daf841..ee10419ce 100644 --- a/docs/DeferredSignals.md +++ b/docs/DeferredSignals.md @@ -4,29 +4,26 @@ FEX-Emu has locations in its code which are effectively "uninterruptible". In th "uninterruptible" code section, then FEX is likely to hang or crash in spurious and terrible ways. ## Example -FEX-Emu is in the middle of emitting code. This is a vulnerable state in which FEX can be in the middle of allocating memory or reading guest state. -If a signal is received in the middle of this, FEX-Emu might need to compile new code in a re-entrant state. Which could involve allocating memory and -other vulnerable things. In this case a mutex could very easily be held, interrupted, and then try to get held again while being reentrant. - -**This will result in a hang.** +When FEX is in the process of emitting code, it often needs to acquire mutexes to safeguard operations like memory allocations or reading guest state. +This puts FEX in a vulnerable state: If a signal is received in the middle of this, FEX may need to initiate compilation of new code. In this case a +mutex could already be held, so attempting to acquire it again would trigger a deadlock. ## How do we solve this? ### Classical signal masking One solution to this problem is to mask **all** signals going in to an uninterruptible section and then unmask when leaving. This is the classical -approach that is viable if performance isn't a significant concern. A major problem with this approach is that we must do two system calls per -"uninterruptible" code section, which if the section is fairly small then it might cost more to ensure state integrity than doing the work at all. +approach that is viable if performance isn't a significant concern. A major problem is that it requires two system calls per "uninterruptible" code +section, which adds overhead that may exceed the runtime of the section itself. ### Cooperative signal deferring -A new solution is to defer asynchronous signals if they are caught inside an uninterruptible section. At the basic level, we increment a reference -counter going in to the "uninterruptible" section, and then decrement the reference counter once we leave. This way when the signal handler receives a -signal, it can check that thread's reference counter, store the `siginfo_t` to an array/stack object, and return to the same code segment to be handled -later. +A new solution is to defer asynchronous signals caught inside an uninterruptible section and handle them at the end of that section. -The trick to how this works is the cost it takes to check if a signal occured, and then handling the signal safely. Ideally no signal has happened in -the "uninterruptible" code section, so the check needs to be as cheap as possible to ensure no cost in the case that no signal has occured. +At the basic level, we increment a reference counter going in to the "uninterruptible" section, and then decrement the reference counter once we leave. +This way when the signal handler receives a signal, it can check that thread's reference counter, store the `siginfo_t` to an array/stack object, and +return to the same code segment to be handled later. -The trick is that FEX maintains two memory regions for tracking deferred signals **per thread**. +By making this check as cheap as possible, overhead is minimized for the general case that no signal occurs during "uninterruptible" sections. FEX +achieves this by maintaining two memory regions for tracking deferred signals **per thread**. #### 1st memory region @@ -38,41 +35,37 @@ This reference counter is thread local and won't be read by any other threads, s Meaning it is usually three instructions (on ARM64) to increment and decrement. ```cpp -MoveableNonatomicRefCounter DeferredReferenceCounter; -MoveableNonatomicRefCounter *DeferredSignalHandlerPagePtr; +NonAtomicRefCounter DeferredSignalRefCount; ``` #### 2nd memory region -This memory region is a single page of memory that is allocated per thread. A pointer to this region exists inside the InternalThreadState object just -after the reference counter. This allows the JIT code to quickly load the pointer to this region after modifying the reference counter. +This memory region is a single page of memory that is allocated per thread. Its purpose is to trigger a SIGSEGV when FEX leaves an "uninterruptible" +section if a signal has been deferred. FEX's signal handler will check if the faulting address is in this special page and subsequently starts the +deferred signal mechanisms. -This is the tricky memory page where we determine if we have any deferred signals to process. I say it's tricky because after FEX leaves an -"uninterruptible" code section, it will decrement the reference counter, and then try to store in to the first byte of this page. In the case that no -signal has been deferred, this memory store will progress without issue, consuming only two instructions in the process. - -In the case that a signal **HAS** been deferred, then the permissions on this page will be set to `PROT_NONE` and this memory store will cause a -SIGSEGV. Then FEX's signal handler will check to see if the access was this special page and start the deferred signal mechanisms. This means that a -deferred signal check is only five instructions in total. +```cpp +NonAtomicRefCounter *DeferredSignalFaultAddress; +``` #### Example ARM64 JIT code for uninterruptible region ```asm ; Increment the reference counter. - ldr x0, [x28, #(offsetof(CPUState, DeferredReferenceCounter))] + ldr x0, [x28, #(offsetof(CPUState, DeferredSignalRefCount))] add x0, x0, #1 - str x0, [x28, #(offsetof(CPUState, DeferredReferenceCounter))] + str x0, [x28, #(offsetof(CPUState, DeferredSignalRefCount))] ; Do the uninterruptible code section here. <...> ; Now decrement the reference counter. - ldr x0, [x28, #(offsetof(CPUState, DeferredReferenceCounter))] + ldr x0, [x28, #(offsetof(CPUState, DeferredSignalRefCount))] sub x0, x0, #1 - str x0, [x28, #(offsetof(CPUState, DeferredReferenceCounter))] + str x0, [x28, #(offsetof(CPUState, DeferredSignalRefCount))] ; Now the magic memory access to check for any deferred signals. ; Load the page pointer from the CPUState - ldr x0, [x28, #(offsetof(CPUState, DeferredSignalHandlerPagePtr))] + ldr x0, [x28, #(offsetof(CPUState, DeferredSignalFaultAddress))] ; Just store zero. (1 cycle plus no dependencies on a register. Super fast!) ; Will store fine with no deferred signal, or SIGSEGV if there was one! str xzr, [x0] @@ -90,48 +83,40 @@ The signal handler now knows that FEX is in an uninterruptible code section. We - If it is an async signal (from tgkill, sigqueue, or something else) then we will start the deferring process. The deferring process starts with storing the kernel `siginfo_t` to a thread local array so we can restore it later. -We then modify the permissions on the thread local `DeferredSignalHandlerPagePtr` to be `PROT_NONE`. +We then modify the permissions on the thread local `DeferredSignalFaultAddress` to be `PROT_NONE`. We then immediately return from the signal handler so that FEX can resume its "uninterruptible" code section without breaking anything. Once the "uninterruptible" code section finishes, FEX will intentionally trigger a SIGSEGV by storing to the page. -This will trigger a jump to FEX's SIGSEGV handler, where FEX will process the signal as if it was the previously deferred signal. -the previous signal that was deferred. -- Replacing the SIGSEGV signal number with the previous captured signal number -- Replacing the `siginfo_t` with the previous captured `siginfo_t` -- mprotect the thread local `DeferredSignalHandlerPagePtr` to be RW. - - This is so that future signals get deferred, but we don't block forward code progress. -- TODO: Overwrite mcontext_t things? I don't think this matters but there will be some private state that might leak SIGSEGV information? - -Once we are now handling the guest signal, FEX-Emu is in a vulnerable state where any signals received will be deferred /and/ not handled at the end -of "uninterruptible" code sections. This is because FEX is now currently in a guest signal frame and we need to handle code compiling and other -potentially awkward interactions without checking for additional signals. +Once FEX-Emu is in its SIGSEGV handler, it will determine that it is handling a deferred signal. This will pull the previously saved `siginfo_t` and +start processing the signal. Once a guest signal handler has finished what it was working on, it will call `rt_sigreturn` or `sigreturn` which triggers FEX's SIGILL signal handler. - - This SIGILL behaviour is how FEX-Emu emulates sigreturn. In order to safely long-jump on AArch64, it must come from a signal context. - - The sigreturn syscall handlers intentionally trigger a SIGILL to do this. Inside of this SIGILL signal handler FEX will restore the state of FEX /back/ to where the deferred signal handler started (The str xzr, [x0]). -Inside of this signal handler FEX will check to see if all deferred signals are handled. -- Checks the reference counter to see if it is zero or not. +Then, FEX will check if any further deferred signals need to be handled. +- Checks if the reference counter is zero or not - If further asynchronous signals have been triggered that need handling, mprotect the fault page to `PROT_NONE` - This trampolining is repeated once per asynchronous signal queued during processing. - - This will cause further signal handling immediately once the JIT returns to its original location (Where it'll cause a SIGSEGV again). + - This will cause further signal handling immediately once the JIT returns to its original location (where it'll cause a SIGSEGV again). Once FEX gets back to the page store, it will trampoline back to the SIGSEGV handler if it has more signals to handle. - - This is an edge case where we aren't expecting multiple signals in almost all cases - - Slightly more expensive is fine in this case. ## Disadvantages of cooperative signal deferring -Still thinking about this, come back to me. I'm concerned about signal queueing. -- How do we handle the guest doing a long-jump out of a signal frame and still receiving signals? - - This will block FEX from handling /any/ more deferred signals. +- How do we handle the guest doing a longjmp out of a signal frame and still receiving signals? + - FEX relies on guest signal handlers returning via `sigreturn` to handle stacked deferred signals, so a longjmp would interfere with this - Do we need to store guest stack as well to see if it has reset its own stack frame? - moon-buggy does this as an example - We currently just leak stack for every guest signal handler that long jumps out of the signal frame. - Long term this would exhaust our stack and then crash. - Test with a second guest thread where our host will only have an 8MB stack instead of the 128MB primary stack. - See issue #2487 +- Deeply recursive signal deferring sections can have excessive SIGSEGV faults. + - In the case of ARM64 it will do a SIGSEGV at the end of each deferred signal section if a signal is queued. + - This can result in a bunch of trampolining. + - Just make sure to not do excessive nesting of deferred signal sections. + - Typically not a problem since deferred signals aren't common. + ## Expectations and considerations ### What happens with a race condition with the refcounter? There are two edges to this problem. The incrementing edge and the decrementing edge that must be considered. @@ -162,57 +147,52 @@ permissions and continue execution safely. ## Execution examples ### No signal This is a simple example because nothing happens. -``` diff -- -! Compiling JIT Code -+ -``` + +- **Enter Deferred region - 3 instructions** +- Compiling JIT Code +- **Exit deferred region - 5 instructions** ### Signal outside of region This is simple because the JIT just handles it. -``` diff -! In JIT code -# Signal received -# Guest Signal handler called -# JIT jumps to guest signal handler -# Hopefully guest calls rt_sigreturn instead of long jumping out. -``` + +- In JIT code +- Signal received +- Guest Signal handler called +- JIT jumps to guest signal handler +- Hopefully guest calls rt_sigreturn instead of long jumping out. ### Synchronous signal in JIT Deferred signals don't affect anything here because only asynchronous signals get affected. -* State reconstruction problems aren't discussed here. -``` diff -! In JIT code -! JIT code causes a synchronous signal (SIGSEGV or other) -# Guest Signal Handler called -# JIT jumps to guest signal handler -# Hopefully guest calls rt_sigreturn instead of long jumping out. -``` + +- In JIT code +- JIT code causes a synchronous signal (SIGSEGV or other) +- Guest Signal Handler called +- JIT jumps to guest signal handler +- Hopefully guest calls rt_sigreturn instead of long jumping out. ### Asynchronous signal in code emitter This is the first interesting example since deferred signals affects it. -``` diff -- -! Compiling JIT Code -# Asynchronous Signal received -# - Host signal handler determines the thread is in a deferred signal section. -# - Signal information is stored in a queue -# - mprotect signal page to NONE. -# - Signal handler returns without giving the signal to the guest -! Compiling JIT Code continues. -+ #1 - -+ Deferred region section causes SIGSEGV -+ - Host signal handler determines deferred region is done, Still has signal in queue. -+ - Pull signal information off of queue -# JIT jumps to guest signal handler -# Hopefully guest calls rt_sigreturn instead of long jumping out. -# Host PC is back at deferred signal section. -+ Deferred region section causes SIGSEGV #2 -+ - Host signal handler determines deferred region is done, No signals in the queue. -# - mprotect signal page to RW. -# - Continue execution. -+ #1 -``` + +- **Enter Deferred region - 3 instructions** +- Compiling JIT Code +- Asynchronous Signal received + - Host signal handler determines the thread is in a deferred signal section. + - Signal information is stored in a queue + - mprotect signal page to NONE. + - Signal handler returns without giving the signal to the guest +- Compiling JIT Code continues. +- **Exit deferred region - 5 instructions** +- Deferred region section causes SIGSEGV + - Host signal handler determines deferred region is done, Still has signal in queue. + - Pull signal information off of queue +- JIT jumps to guest signal handler +- Hopefully guest calls rt_sigreturn instead of long jumping out. +- Host PC is back at deferred signal section. +- Deferred region section causes SIGSEGV #2 + - Host signal handler determines deferred region is done, No signals in the queue. + - mprotect signal page to RW. + - Continue execution. +- **Exit deferred region continues** ### Recursive regions with signal in code emitter. This one mostly matches the previous example except the behaviour of deferred signal regions leaving. @@ -226,39 +206,34 @@ In this case, if the thread-local refcount is still >0 on ` -! Compiling JIT Code -- #2 -! Memory allocation -# -+ #2 -+ #2 Exit deferred region causes SIGSEGV -+ - TLS refcount is still 1 (from #1) -+ - PC is incremented by one instruction, signal still unhandled. -+ #1 -+ #1 Exit deferred region causes SIGSEGV -+ -``` +- **Enter Deferred region - 3 instructions** +- Compiling JIT Code + - Enter Deferred region - 3 instructions + - Memory allocation + - Async signal received logic from above + - Exit deferred region - 5 instructions + - Exit deferred region causes SIGSEGV + - TLS refcount is still 1 + - PC is incremented by one instruction, signal still unhandled. +- **Exit deferred region - 5 instructions** +- Exit deferred region causes SIGSEGV +- **Regular deferred region handling from above called** -### Multiple signals in deferred region +### Multiple signals in signal-deferring region This is slightly different from the previous iterations since multiple signals in the stack result in odd behaviour. -``` diff -- #1 -! Compiling JIT Code -# #1 Asynchronous Signal received -# - Signal queued logic -# #2 Asynchronous Signal received -# - Signal queued logic -+ #1 -+ Exit deferred region causes SIGSEGV -+ -+ Guest calls rt_sigreturn -+ - rt_sigreturn handler checks for number of queued signals -# - mprotect signal page to NONE because signals is > 0 -# - JIT is back to #1 -# - Exit deferred region causes SIGSEGV again. -# - Regular handler loop occurs -``` - +- **Enter Deferred region - 3 instructions** +- Compiling JIT Code +- Asynchronous Signal received + - Signal queued logic +- Asynchronous Signal received + - Signal queued logic +- **Exit deferred region - 5 instructions** +- Exit deferred region causes SIGSEGV +- **Regular deferred region handling from above called** +- Guest calls rt_sigreturn + - rt_sigreturn handler checks for number of queued signals + - mprotect signal page to NONE because signals is > 0 + - JIT is back to **Exit deferred region** + - Exit deferred region causes SIGSEGV again. + - Regular handler loop occurs