mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 17:00:19 +02:00
FEXCore: Moves SpinWaitLock and WritePriorityMutex to frontend visible includes
This will be used in a moment.
This commit is contained in:
8 files changed
+14
-9
No files matched your search
@@ -2,7 +2,7 @@
|
||||
|
||||
#include "Interface/Core/CPUBackend.h"
|
||||
#include "Interface/Context/Context.h"
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include <FEXCore/Utils/SpinWaitLock.h>
|
||||
|
||||
#include <FEXCore/Debug/InternalThreadState.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
@@ -1,5 +1,5 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
#include <FEXCore/Utils/SpinWaitLock.h>
|
||||
|
||||
namespace FEXCore::Utils::SpinWaitLock {
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
|
||||
@@ -1,315 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include <atomic>
|
||||
#include <chrono>
|
||||
#include <mutex>
|
||||
#include <type_traits>
|
||||
|
||||
#include <FEXCore/fextl/functional.h>
|
||||
#include <FEXCore/Utils/EnumUtils.h>
|
||||
|
||||
namespace FEXCore::Utils::SpinWaitLock {
|
||||
/**
|
||||
* @brief This provides routines to implement implement an "efficient spin-loop" using ARM's WFE and exclusive monitor interfaces.
|
||||
*
|
||||
* Spin-loops on mobile devices with a battery can be a bad idea as they burn a bunch of power. This attempts to mitigate some of the impact
|
||||
* by putting the CPU in to a lower-power state using WFE.
|
||||
* On platforms tested, WFE will put the CPU in to a lower power state for upwards of 0.11ms(!) per WFE. Which isn't a significant amount of
|
||||
* time but should still have power savings. Ideally WFE would be able to keep the CPU in a lower power state for longer. This also has the
|
||||
* added benefit that atomics aren't abusing the caches when spinning on a cacheline, which has knock-on powersaving benefits.
|
||||
*
|
||||
* This short timeout is because the Linux kernel has a 100 microsecond architecture timer which wakes up WFE and WFI. Nothing can be
|
||||
* improved beyond that period.
|
||||
*
|
||||
* FEAT_WFxT adds a new instruction with a timeout, but since the spurious wake-up is so aggressive it isn't worth using.
|
||||
*
|
||||
* It should be noted that this implementation has a few dozen cycles of start-up time. Which means the overhead for invoking this
|
||||
* implementation is slightly higher than a true spin-loop. The hot loop body itself is only three instructions so it is quite efficient.
|
||||
*
|
||||
* On non-ARM platforms it is truly a spin-loop, which is okay for debugging only.
|
||||
*/
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
|
||||
#define LOADEXCLUSIVE(LoadExclusiveOp, RegSize) \
|
||||
/* Prime the exclusive monitor with the passed in address. */ \
|
||||
#LoadExclusiveOp " %" #RegSize "[Result], [%[Futex]];\n"
|
||||
|
||||
#define SPINLOOP_BODY(LoadAtomicOp, RegSize) \
|
||||
/* WFE will wait for either the memory to change or spurious wake-up. */ \
|
||||
"wfe;\n" /* Load with acquire to get the result of memory. */ \
|
||||
#LoadAtomicOp " %" #RegSize "[Result], [%[Futex]];\n"
|
||||
|
||||
#define SPINLOOP_WFE_LDX_8BIT LOADEXCLUSIVE(ldaxrb, w)
|
||||
#define SPINLOOP_WFE_LDX_16BIT LOADEXCLUSIVE(ldaxrh, w)
|
||||
#define SPINLOOP_WFE_LDX_32BIT LOADEXCLUSIVE(ldaxr, w)
|
||||
#define SPINLOOP_WFE_LDX_64BIT LOADEXCLUSIVE(ldaxr, x)
|
||||
|
||||
#define SPINLOOP_8BIT SPINLOOP_BODY(ldarb, w)
|
||||
#define SPINLOOP_16BIT SPINLOOP_BODY(ldarh, w)
|
||||
#define SPINLOOP_32BIT SPINLOOP_BODY(ldar, w)
|
||||
#define SPINLOOP_64BIT SPINLOOP_BODY(ldar, x)
|
||||
|
||||
extern uint64_t CycleCounterFrequency;
|
||||
extern uint64_t CyclesPerNanosecond;
|
||||
|
||||
///< Get the raw cycle counter which is synchronizing.
|
||||
/// `CNTVCTSS_EL0` also does the same thing, but requires the FEAT_ECV feature.
|
||||
static inline uint64_t GetCycleCounter() {
|
||||
uint64_t Result {};
|
||||
__asm volatile(R"(
|
||||
isb;
|
||||
mrs %[Res], CNTVCT_EL0;
|
||||
)"
|
||||
: [Res] "=r"(Result));
|
||||
return Result;
|
||||
}
|
||||
|
||||
///< Converts nanoseconds to number of cycles.
|
||||
/// If the cycle counter is 1Ghz then this is a direct 1:1 map.
|
||||
static inline uint64_t ConvertNanosecondsToCycles(const std::chrono::nanoseconds& Nanoseconds) {
|
||||
const auto NanosecondCount = Nanoseconds.count();
|
||||
return NanosecondCount / CyclesPerNanosecond;
|
||||
}
|
||||
|
||||
static inline uint8_t LoadExclusive(uint8_t* Futex) {
|
||||
uint8_t Result {};
|
||||
__asm volatile(SPINLOOP_WFE_LDX_8BIT : [Result] "=r"(Result), [Futex] "+r"(Futex)::"memory");
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
static inline uint16_t LoadExclusive(uint16_t* Futex) {
|
||||
uint16_t Result {};
|
||||
__asm volatile(SPINLOOP_WFE_LDX_16BIT : [Result] "=r"(Result), [Futex] "+r"(Futex)::"memory");
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
static inline uint32_t LoadExclusive(uint32_t* Futex) {
|
||||
uint32_t Result {};
|
||||
__asm volatile(SPINLOOP_WFE_LDX_32BIT : [Result] "=r"(Result), [Futex] "+r"(Futex)::"memory");
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
static inline uint64_t LoadExclusive(uint64_t* Futex) {
|
||||
uint64_t Result {};
|
||||
__asm volatile(SPINLOOP_WFE_LDX_64BIT : [Result] "=r"(Result), [Futex] "+r"(Futex)::"memory");
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
static inline uint8_t WFELoadAtomic(uint8_t* Futex) {
|
||||
uint8_t Result {};
|
||||
__asm volatile(SPINLOOP_8BIT : [Result] "=r"(Result), [Futex] "+r"(Futex)::"memory");
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
static inline uint16_t WFELoadAtomic(uint16_t* Futex) {
|
||||
uint16_t Result {};
|
||||
__asm volatile(SPINLOOP_16BIT : [Result] "=r"(Result), [Futex] "+r"(Futex)::"memory");
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
static inline uint32_t WFELoadAtomic(uint32_t* Futex) {
|
||||
uint32_t Result {};
|
||||
__asm volatile(SPINLOOP_32BIT : [Result] "=r"(Result), [Futex] "+r"(Futex)::"memory");
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
static inline uint64_t WFELoadAtomic(uint64_t* Futex) {
|
||||
uint64_t Result {};
|
||||
__asm volatile(SPINLOOP_64BIT : [Result] "=r"(Result), [Futex] "+r"(Futex)::"memory");
|
||||
|
||||
return Result;
|
||||
}
|
||||
|
||||
template<typename Pred, typename T>
|
||||
static inline void WaitPred(T* Futex, T ComparisonValue) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
while (!Pred {}(Result, ComparisonValue)) {
|
||||
Result = LoadExclusive(Futex);
|
||||
if (Pred {}(Result, ComparisonValue)) {
|
||||
return;
|
||||
}
|
||||
|
||||
Result = WFELoadAtomic(Futex);
|
||||
}
|
||||
}
|
||||
|
||||
template<typename T, typename TT>
|
||||
static inline bool Wait(T* Futex, TT ExpectedValue, const std::chrono::nanoseconds& Timeout) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
// Early exit if possible.
|
||||
if (Result == ExpectedValue) {
|
||||
return true;
|
||||
}
|
||||
|
||||
const auto TimeoutCycles = ConvertNanosecondsToCycles(Timeout);
|
||||
const auto Begin = GetCycleCounter();
|
||||
|
||||
do {
|
||||
Result = LoadExclusive(Futex);
|
||||
if (Result == ExpectedValue) {
|
||||
return true;
|
||||
}
|
||||
Result = WFELoadAtomic(Futex);
|
||||
|
||||
const auto CurrentCycleCounter = GetCycleCounter();
|
||||
if ((CurrentCycleCounter - Begin) >= TimeoutCycles) {
|
||||
// Couldn't get value before timeout.
|
||||
return false;
|
||||
}
|
||||
} while (Result != ExpectedValue);
|
||||
|
||||
// We got our result.
|
||||
return true;
|
||||
}
|
||||
|
||||
template bool Wait<uint8_t>(uint8_t*, uint8_t, const std::chrono::nanoseconds&);
|
||||
template bool Wait<uint16_t>(uint16_t*, uint16_t, const std::chrono::nanoseconds&);
|
||||
template bool Wait<uint32_t>(uint32_t*, uint32_t, const std::chrono::nanoseconds&);
|
||||
template bool Wait<uint64_t>(uint64_t*, uint64_t, const std::chrono::nanoseconds&);
|
||||
|
||||
template<typename T>
|
||||
static inline T OneShotWFEBitComparison(T* Futex, T Mask, T Comp) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
// Early exit if possible.
|
||||
if ((Result & Mask) == Comp) {
|
||||
return Result;
|
||||
}
|
||||
|
||||
Result = LoadExclusive(Futex);
|
||||
if ((Result & Mask) == Comp) {
|
||||
return Result;
|
||||
}
|
||||
|
||||
// Waits for write and returns result.
|
||||
Result = WFELoadAtomic(Futex);
|
||||
return Result;
|
||||
}
|
||||
|
||||
#else
|
||||
|
||||
template<typename Pred, typename T>
|
||||
static inline void WaitPred(T* Futex, T ComparisonValue) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
while (!Pred {}(Result, ComparisonValue)) {
|
||||
Result = AtomicFutex.load();
|
||||
}
|
||||
}
|
||||
|
||||
template<typename T, typename TT>
|
||||
static inline bool Wait(T* Futex, TT ExpectedValue, const std::chrono::nanoseconds& Timeout) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
|
||||
T Result = AtomicFutex.load();
|
||||
|
||||
// Early exit if possible.
|
||||
if (Result == ExpectedValue) {
|
||||
return true;
|
||||
}
|
||||
|
||||
const auto Begin = std::chrono::high_resolution_clock::now();
|
||||
|
||||
do {
|
||||
Result = AtomicFutex.load();
|
||||
|
||||
const auto CurrentCycleCounter = std::chrono::high_resolution_clock::now();
|
||||
if ((CurrentCycleCounter - Begin) >= Timeout) {
|
||||
// Couldn't get value before timeout.
|
||||
return false;
|
||||
}
|
||||
} while (Result != ExpectedValue);
|
||||
|
||||
// We got our result.
|
||||
return true;
|
||||
}
|
||||
#endif
|
||||
|
||||
template<typename T, typename TT = T>
|
||||
static inline void Wait(T* Futex, TT ExpectedValue) {
|
||||
WaitPred<std::equal_to<>, T>(Futex, ExpectedValue);
|
||||
}
|
||||
|
||||
template void Wait<uint8_t>(uint8_t*, uint8_t);
|
||||
template void Wait<uint16_t>(uint16_t*, uint16_t);
|
||||
template void Wait<uint32_t>(uint32_t*, uint32_t);
|
||||
template void Wait<uint64_t>(uint64_t*, uint64_t);
|
||||
|
||||
template<typename T>
|
||||
static inline void lock(T* Futex) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Expected {};
|
||||
T Desired {1};
|
||||
|
||||
// Try to CAS immediately.
|
||||
if (AtomicFutex.compare_exchange_strong(Expected, Desired)) {
|
||||
return;
|
||||
}
|
||||
|
||||
do {
|
||||
// Wait until the futex is unlocked.
|
||||
Wait(Futex, 0);
|
||||
Expected = 0;
|
||||
} while (!AtomicFutex.compare_exchange_strong(Expected, Desired));
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
static inline bool try_lock(T* Futex) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
T Expected {};
|
||||
T Desired {1};
|
||||
|
||||
// Try to CAS immediately.
|
||||
if (AtomicFutex.compare_exchange_strong(Expected, Desired)) {
|
||||
return true;
|
||||
}
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
template<typename T>
|
||||
static inline void unlock(T* Futex) {
|
||||
auto AtomicFutex = std::atomic_ref<T>(*Futex);
|
||||
AtomicFutex.store(0);
|
||||
}
|
||||
|
||||
#undef SPINLOOP_8BIT
|
||||
#undef SPINLOOP_16BIT
|
||||
#undef SPINLOOP_32BIT
|
||||
#undef SPINLOOP_64BIT
|
||||
template<typename T>
|
||||
class UniqueSpinMutex final {
|
||||
public:
|
||||
// Move-only type
|
||||
UniqueSpinMutex(const UniqueSpinMutex&) = delete;
|
||||
UniqueSpinMutex& operator=(const UniqueSpinMutex&) = delete;
|
||||
UniqueSpinMutex(UniqueSpinMutex&& rhs) = default;
|
||||
UniqueSpinMutex& operator=(UniqueSpinMutex&&) = default;
|
||||
|
||||
UniqueSpinMutex(T* Futex)
|
||||
: Futex {Futex} {
|
||||
FEXCore::Utils::SpinWaitLock::lock(Futex);
|
||||
}
|
||||
|
||||
~UniqueSpinMutex() {
|
||||
FEXCore::Utils::SpinWaitLock::unlock(Futex);
|
||||
}
|
||||
private:
|
||||
T* Futex;
|
||||
};
|
||||
} // namespace FEXCore::Utils::SpinWaitLock
|
||||
@@ -1,407 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
#include <atomic>
|
||||
#include <cstdint>
|
||||
|
||||
#if !defined(_WIN32)
|
||||
#include <linux/futex.h> /* Definition of FUTEX_* constants */
|
||||
#include <sys/syscall.h> /* Definition of SYS_* constants */
|
||||
#include <unistd.h>
|
||||
#else
|
||||
#include <synchapi.h>
|
||||
#endif
|
||||
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
|
||||
#include "Utils/SpinWaitLock.h"
|
||||
|
||||
namespace FEXCore::Utils::WritePriorityMutex {
|
||||
|
||||
// A custom mutex that prioritizes exclusive locks.
|
||||
// In highly contested scenarios, this can help minimize overall contention time.
|
||||
//
|
||||
// Features:
|
||||
// - Up to 32767 pending exclusive locks ("writers")
|
||||
// - Up to 32767 pending shared_locks ("readers")
|
||||
// - Low-overhead waiting via WFE with a fallback to futex on timeout
|
||||
// - Direct writer->reader hand-off and vice-versa to further reduce overhead
|
||||
//
|
||||
// Trade-offs:
|
||||
// - No guaranteed order of wake-ups besides prioritizing writers
|
||||
// - No support for recursive locking
|
||||
// - We can't use FUTEX_LOCK_PI to enable priority inheritance
|
||||
class Mutex final {
|
||||
public:
|
||||
Mutex() = default;
|
||||
|
||||
// Move-only type
|
||||
Mutex(const Mutex&) = delete;
|
||||
Mutex& operator=(const Mutex&) = delete;
|
||||
Mutex(Mutex&& rhs) = delete;
|
||||
Mutex& operator=(Mutex&&) = delete;
|
||||
|
||||
void lock() {
|
||||
// Try a non-blocking lock first.
|
||||
if (try_lock()) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Try a quick WFE write-lock.
|
||||
if (Attempt_WFE_WriteLock()) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Still couldn't get it. Start waiting.
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Expected {};
|
||||
uint32_t Desired {};
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
do {
|
||||
// Increment the number of write waiters.
|
||||
Desired = Expected + WRITE_WAITER_INCREMENT;
|
||||
|
||||
LOGMAN_THROW_A_FMT((Desired & WRITE_WAITER_COUNT_MASK) != 0, "Overflow in write-waiters!");
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
#else
|
||||
// Increment the number of writers waiting. The following loop will attempt to acquire the write-lock while decrementing the waiter count.
|
||||
Expected = AtomicFutex.fetch_add(WRITE_WAITER_INCREMENT);
|
||||
Desired = Expected + WRITE_WAITER_INCREMENT;
|
||||
#endif
|
||||
|
||||
// Thread added to waiter list.
|
||||
Expected = Desired;
|
||||
|
||||
while (true) {
|
||||
bool Sleep = false;
|
||||
|
||||
do {
|
||||
if ((Expected & WRITE_OWNED_BIT) == 0 && (Expected & READ_OWNER_COUNT_MASK) == 0) {
|
||||
// If not write-owned, and no read-owners, try to acquire.
|
||||
LOGMAN_THROW_A_FMT((Expected & WRITE_WAITER_COUNT_MASK) != 0, "Underflow in write-waiters!");
|
||||
|
||||
// Add write-owned bit.
|
||||
Desired = Expected | WRITE_OWNED_BIT;
|
||||
|
||||
// Remove ourselves from the wait list.
|
||||
Desired -= WRITE_WAITER_INCREMENT;
|
||||
|
||||
Sleep = false;
|
||||
} else {
|
||||
// Already write-owned or read-locked. Go to sleep.
|
||||
Desired = Expected;
|
||||
Sleep = true;
|
||||
break;
|
||||
}
|
||||
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
|
||||
if (!Sleep) {
|
||||
// Acquired early.
|
||||
LOGMAN_THROW_A_FMT((Desired & WRITE_OWNED_BIT) == WRITE_OWNED_BIT, "Somehow acquired a write-lock without it being set!");
|
||||
return;
|
||||
}
|
||||
|
||||
// Two paths to get here.
|
||||
// Desired[31] = 1 (WRITE_OWNED_BIT)
|
||||
// OR
|
||||
// Desired[15:0] != 0 (READ_OWNER_COUNT_MASK)
|
||||
// Meaning that there was already a writer that owned the lock, or reads were owning it.
|
||||
// This thread already incremented `WRITE_WAITER_INCREMENT` before this loop.
|
||||
// - Linux waits for the full 32-bits to change (With bitset wakeup).
|
||||
// - Win32 also waits for the full 32-bits to change (with offset addr on the reader side to reduce stampeding).
|
||||
FutexWaitForWriteAvailable(Desired);
|
||||
|
||||
Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
void lock_shared() {
|
||||
// Try an uncontended lock first.
|
||||
if (try_lock_shared()) {
|
||||
return;
|
||||
}
|
||||
|
||||
// Try a quick WFE read-lock.
|
||||
if (Attempt_WFE_ReadLock()) {
|
||||
return;
|
||||
}
|
||||
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
uint32_t Desired {};
|
||||
|
||||
while (true) {
|
||||
bool Sleep = false;
|
||||
do {
|
||||
if ((Expected & WRITE_OWNED_BIT) == 0 && (Expected & WRITE_WAITER_COUNT_MASK) == 0) {
|
||||
// If no write-owner and no write-waiting, try and acquire.
|
||||
|
||||
Desired = Expected + READ_OWNER_INCREMENT;
|
||||
LOGMAN_THROW_A_FMT((Desired & READ_OWNER_COUNT_MASK) != 0, "Overflow in read-owners!");
|
||||
Sleep = false;
|
||||
} else {
|
||||
// Waiting for lock to become available. Add to waiters.
|
||||
Desired = Expected | READ_WAITER_BIT;
|
||||
Sleep = true;
|
||||
}
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
|
||||
if (!Sleep) {
|
||||
// Acquired early.
|
||||
LOGMAN_THROW_A_FMT((Desired & WRITE_OWNED_BIT) != WRITE_OWNED_BIT, "Somehow read-locked and got a write lock!");
|
||||
return;
|
||||
}
|
||||
|
||||
// Only one path to get here.
|
||||
// Desired[31][29:16] != 0 (Either writer-owned, or writer-waiting)
|
||||
// Desired[30][15:0] == READ_WAIT_BIT and number of read-owners (draining to zero as write-side is set)
|
||||
// - Linux waits for full 32-bit futex.
|
||||
// - Win32 waits for upper 16-bits to not match (Either zero writer owned, writer-wait is draining, and `READ_WAITER_BIT` changed).
|
||||
// Can get some spurious wake-ups which will `or` the `READ_WAITER_BIT` again, which does nothing.
|
||||
FutexWaitForReadAvailable(Desired);
|
||||
|
||||
Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
}
|
||||
}
|
||||
|
||||
void unlock() {
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
uint32_t Desired {};
|
||||
do {
|
||||
LOGMAN_THROW_A_FMT((Expected & WRITE_OWNED_BIT) == WRITE_OWNED_BIT, "Trying to write-unlock something not write-locked!");
|
||||
// Remove the exclusive lock bit.
|
||||
Desired = Expected & ~WRITE_OWNED_BIT;
|
||||
|
||||
// If no more writers, then make sure to clear the read-waiters bit as well.
|
||||
if ((Desired & WRITE_WAITER_COUNT_MASK) == 0) {
|
||||
Desired &= ~READ_WAITER_BIT;
|
||||
}
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
|
||||
// `Expected` has old value. Containing `READ_WAITER_BIT` which was just masked off, and also `WRITE_WAITER_COUNT_MASK`.
|
||||
//
|
||||
// Two paths here to be careful about dead-locking other waiters:
|
||||
// - If there are any writers waiting, those get priority to wake.
|
||||
// - If there are zero writers waiting, and there are read waiters then make sure to wake them all.
|
||||
// Failure to send wake events can cause readers to "infinitely" hang! (ignoring spurious wake-up).
|
||||
if ((Expected & WRITE_WAITER_COUNT_MASK)) {
|
||||
// Handle write-write handoff.
|
||||
FutexWakeWriter();
|
||||
} else if ((Expected & READ_WAITER_BIT)) {
|
||||
// Handle write-reader handoff.
|
||||
FutexWakeReaders();
|
||||
}
|
||||
}
|
||||
|
||||
void unlock_shared() {
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Desired {};
|
||||
#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
do {
|
||||
LOGMAN_THROW_A_FMT((Expected & WRITE_OWNED_BIT) != WRITE_OWNED_BIT, "Trying to read-unlock something write-locked!");
|
||||
LOGMAN_THROW_A_FMT((Expected & READ_OWNER_COUNT_MASK) != 0, "Trying to read-unlock something not read-locked!");
|
||||
|
||||
// Decrement the shared counter.
|
||||
Desired = Expected - READ_OWNER_INCREMENT;
|
||||
} while (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire) == false);
|
||||
#else
|
||||
Desired = AtomicFutex.fetch_sub(READ_OWNER_INCREMENT) - READ_OWNER_INCREMENT;
|
||||
#endif
|
||||
|
||||
// Handle read->write handoff if there are any waiting writers, and no readers left.
|
||||
// Only one path here but still need to be careful to not dead-lock waiting writers.
|
||||
// - If there are waiters /but/ this is not the final unlock_shared, then don't wake writer.
|
||||
// - Writer would wake and immediately sleep again if we woke on every unlock_shared.
|
||||
// - If there are waiters and this is the final unlock_shared, then wake a /single/ writer.
|
||||
// - We ignore any reader-waiters here as they must wait their turn for writers that are waiting.
|
||||
if ((Desired & WRITE_WAITER_COUNT_MASK) && (Desired & READ_OWNER_COUNT_MASK) == 0) {
|
||||
FutexWakeWriter();
|
||||
}
|
||||
}
|
||||
|
||||
bool try_lock() {
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
|
||||
uint32_t Expected = 0;
|
||||
|
||||
// Try and grab the owned bit.
|
||||
uint32_t Desired = WRITE_OWNED_BIT;
|
||||
|
||||
// try to CAS immediately.
|
||||
return AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire);
|
||||
}
|
||||
|
||||
// Can race with other threads trying to lock shared!
|
||||
bool try_lock_shared() {
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
|
||||
// Exclusively owned or has a list of waiting owners. Can't pass.
|
||||
if ((Expected & WRITE_OWNED_BIT) || (Expected & WRITE_WAITER_COUNT_MASK)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
// Try to add reader.
|
||||
uint32_t Desired = Expected + READ_OWNER_INCREMENT;
|
||||
LOGMAN_THROW_A_FMT((Desired & READ_OWNER_COUNT_MASK) != 0, "Overflow in read-owners!");
|
||||
|
||||
// Uncontended mutex check
|
||||
return AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire);
|
||||
}
|
||||
|
||||
#if !defined(_WIN32)
|
||||
// Initialize the internal mutex object to its default initializer state.
|
||||
// Should only ever be used in the child process when a Linux fork() has occured.
|
||||
void StealAndDropActiveLocks() {
|
||||
Futex = 0;
|
||||
}
|
||||
#endif
|
||||
|
||||
private:
|
||||
|
||||
#if !defined(_WIN32)
|
||||
void FutexWaitForWriteAvailable(uint32_t Expected) {
|
||||
::syscall(SYS_futex, &Futex, FUTEX_PRIVATE_FLAG | FUTEX_WAIT_BITSET, Expected, nullptr, nullptr, FUTEX_BITSET_WAIT_WRITERS);
|
||||
}
|
||||
|
||||
// Read-lock waiting for writers to drain out.
|
||||
void FutexWaitForReadAvailable(uint32_t Expected) {
|
||||
::syscall(SYS_futex, &Futex, FUTEX_PRIVATE_FLAG | FUTEX_WAIT_BITSET, Expected, nullptr, nullptr, FUTEX_BITSET_WAIT_READERS);
|
||||
}
|
||||
|
||||
// Read-Lock or Write-lock unlocked, wake one writer.
|
||||
// - Read->Write handoff.
|
||||
// - Write->Write handoff.
|
||||
void FutexWakeWriter() {
|
||||
::syscall(SYS_futex, &Futex, FUTEX_PRIVATE_FLAG | FUTEX_WAKE_BITSET, 1, nullptr, nullptr, FUTEX_BITSET_WAIT_WRITERS);
|
||||
}
|
||||
|
||||
// Write-lock unlocked, wake read-locks waiting.
|
||||
void FutexWakeReaders() {
|
||||
// Wake all readers.
|
||||
::syscall(SYS_futex, &Futex, FUTEX_PRIVATE_FLAG | FUTEX_WAKE_BITSET, INT_MAX, nullptr, nullptr, FUTEX_BITSET_WAIT_READERS);
|
||||
}
|
||||
#else
|
||||
// Writers wait for the full 32-bit futex.
|
||||
void FutexWaitForWriteAvailable(uint32_t Expected) {
|
||||
WaitOnAddress(&Futex, &Expected, sizeof(Futex), INFINITE);
|
||||
}
|
||||
|
||||
// Readers wait for Futex bits [31:16] to be zero.
|
||||
void FutexWaitForReadAvailable(uint32_t Expected) {
|
||||
auto ReadWaiterAddress = reinterpret_cast<uint8_t*>(&Futex) + 2;
|
||||
uint16_t smol_Expected = Expected >> 16;
|
||||
WaitOnAddress(ReadWaiterAddress, &smol_Expected, sizeof(smol_Expected), INFINITE);
|
||||
}
|
||||
|
||||
void FutexWakeWriter() {
|
||||
WakeByAddressSingle(&Futex);
|
||||
}
|
||||
|
||||
void FutexWakeReaders() {
|
||||
auto ReadWaiterAddress = reinterpret_cast<uint8_t*>(&Futex) + 2;
|
||||
WakeByAddressAll(ReadWaiterAddress);
|
||||
}
|
||||
#endif
|
||||
|
||||
// Reuse the SpinWaitLock WFE implementations for read/write lock acquiring with WFE.
|
||||
// Can't reuse the spin-lock directly as some bit-representations are different.
|
||||
// WFE-write-lock is less likely to occur the more read-lock threads are participating. Can still occur so good to try.
|
||||
// WFE-read-lock is actually quite likely to succeed.
|
||||
// Return: true if the lock was acquired.
|
||||
bool Attempt_WFE_WriteLock() {
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
const auto Begin = FEXCore::Utils::SpinWaitLock::GetCycleCounter();
|
||||
auto Now = Begin;
|
||||
const auto Duration = FEXCore::Utils::SpinWaitLock::CycleCounterFrequency / CYCLECOUNT_DIVISOR;
|
||||
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
|
||||
while ((Now - Begin) < Duration) {
|
||||
if (Expected == 0) {
|
||||
// Try and grab the owned bit.
|
||||
uint32_t Desired = WRITE_OWNED_BIT;
|
||||
|
||||
if (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// One-shot attempt to wait for mask to be zero.
|
||||
Expected = FEXCore::Utils::SpinWaitLock::OneShotWFEBitComparison(&Futex, ~0U, 0U);
|
||||
Now = FEXCore::Utils::SpinWaitLock::GetCycleCounter();
|
||||
}
|
||||
#endif
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
// Return: true if the lock was acquired.
|
||||
bool Attempt_WFE_ReadLock() {
|
||||
#ifdef ARCHITECTURE_arm64
|
||||
// Spin on a WFE for a short-amount of time, waiting for write-owned and writer-count to be zero.
|
||||
// - Attempt to acquire read-lock at that point.
|
||||
// - Don't add read-waiters bit on failure, return false.
|
||||
const auto Begin = FEXCore::Utils::SpinWaitLock::GetCycleCounter();
|
||||
auto Now = Begin;
|
||||
const auto Duration = FEXCore::Utils::SpinWaitLock::CycleCounterFrequency / CYCLECOUNT_DIVISOR;
|
||||
|
||||
auto AtomicFutex = std::atomic_ref<uint32_t>(Futex);
|
||||
uint32_t Expected = AtomicFutex.load(std::memory_order_relaxed);
|
||||
uint32_t Desired {};
|
||||
|
||||
while ((Now - Begin) < Duration) {
|
||||
if ((Expected & WRITE_OWNED_BIT) == 0 && (Expected & WRITE_WAITER_COUNT_MASK) == 0) {
|
||||
// If no write-owner and no write-waiting, try and acquire.
|
||||
|
||||
Desired = Expected + READ_OWNER_INCREMENT;
|
||||
LOGMAN_THROW_A_FMT((Desired & READ_OWNER_COUNT_MASK) != 0, "Overflow in read-owners!");
|
||||
if (AtomicFutex.compare_exchange_strong(Expected, Desired, std::memory_order_acq_rel, std::memory_order_acquire)) {
|
||||
return true;
|
||||
}
|
||||
}
|
||||
|
||||
// One-shot attempt to wait for mask to be zero.
|
||||
Expected = FEXCore::Utils::SpinWaitLock::OneShotWFEBitComparison(&Futex, WRITE_OWNED_BIT | WRITE_WAITER_COUNT_MASK, 0U);
|
||||
Now = FEXCore::Utils::SpinWaitLock::GetCycleCounter();
|
||||
}
|
||||
#endif
|
||||
|
||||
return false;
|
||||
}
|
||||
|
||||
constexpr static uint32_t WRITE_OWNED_BIT = 1U << 31;
|
||||
constexpr static uint32_t READ_WAITER_BIT = 1U << 30;
|
||||
constexpr static uint32_t WRITE_WAITER_OFFSET = 16;
|
||||
constexpr static uint32_t WRITE_WAITER_INCREMENT = 1U << WRITE_WAITER_OFFSET;
|
||||
constexpr static uint32_t READ_OWNER_INCREMENT = 1;
|
||||
|
||||
// Count masks
|
||||
constexpr static uint32_t WRITE_WAITER_COUNT_MASK = 0x3FFFU << WRITE_WAITER_OFFSET;
|
||||
constexpr static uint32_t READ_OWNER_COUNT_MASK = 0xFFFFU;
|
||||
|
||||
// Independent futex bit-set masks.
|
||||
// Wait for readers to drain.
|
||||
constexpr static uint32_t FUTEX_BITSET_WAIT_READERS = 1U << 0;
|
||||
// Wait for writers to drain.
|
||||
constexpr static uint32_t FUTEX_BITSET_WAIT_WRITERS = 1U << 1;
|
||||
|
||||
// Only spin on WFE for 0.01ms (10k ns).
|
||||
constexpr static uint64_t CYCLECOUNT_DIVISOR = 1'000'000'000ULL / 10'000U;
|
||||
|
||||
// Layout:
|
||||
// Bits[31]: Write-lock bit.
|
||||
// Bits[30]: Read-waiter bit.
|
||||
// Bits[29:16]: Write-waiter count.
|
||||
// Bits[15:0]: Read-owner count.
|
||||
uint32_t Futex {};
|
||||
};
|
||||
} // namespace FEXCore::Utils::WritePriorityMutex
|
||||
Reference in new issue
Block a user