// SPDX-License-Identifier: MIT #include "Utils/SpinWaitLock.h" #include #include #include #include #include #include #include #include namespace FEXCore::ArchHelpers::Arm64 { FEXCORE_TELEMETRY_STATIC_INIT(SplitLock, TYPE_HAS_SPLIT_LOCKS); FEXCORE_TELEMETRY_STATIC_INIT(SplitLock16B, TYPE_16BYTE_SPLIT); FEXCORE_TELEMETRY_STATIC_INIT(Cas16Tear, TYPE_CAS_16BIT_TEAR); FEXCORE_TELEMETRY_STATIC_INIT(Cas32Tear, TYPE_CAS_32BIT_TEAR); FEXCORE_TELEMETRY_STATIC_INIT(Cas64Tear, TYPE_CAS_64BIT_TEAR); FEXCORE_TELEMETRY_STATIC_INIT(Cas128Tear, TYPE_CAS_128BIT_TEAR); static void ClearICache(void* Begin, std::size_t Length) { __builtin___clear_cache(static_cast(Begin), static_cast(Begin) + Length); } static __uint128_t LoadAcquire128(uint64_t Addr) { __uint128_t Result{}; uint64_t Lower; uint64_t Upper; // This specifically avoids using std::atomic<__uint128_t> // std::atomic helper does a ldaxp + stxp pair that crashes when the page is only mapped readable __asm volatile( R"( ldaxp %[ResultLower], %[ResultUpper], [%[Addr]]; clrex; )" : [ResultLower] "=r" (Lower) , [ResultUpper] "=r" (Upper) : [Addr] "r" (Addr) : "memory"); Result = Upper; Result <<= 64; Result |= Lower; return Result; } static uint64_t LoadAcquire64(uint64_t Addr) { std::atomic *Atom = reinterpret_cast*>(Addr); return Atom->load(std::memory_order_acquire); } static bool StoreCAS64(uint64_t &Expected, uint64_t Val, uint64_t Addr) { std::atomic *Atom = reinterpret_cast*>(Addr); return Atom->compare_exchange_strong(Expected, Val); } static uint32_t LoadAcquire32(uint64_t Addr) { std::atomic *Atom = reinterpret_cast*>(Addr); return Atom->load(std::memory_order_acquire); } static bool StoreCAS32(uint32_t &Expected, uint32_t Val, uint64_t Addr) { std::atomic *Atom = reinterpret_cast*>(Addr); return Atom->compare_exchange_strong(Expected, Val); } static uint8_t LoadAcquire8(uint64_t Addr) { std::atomic *Atom = reinterpret_cast*>(Addr); return Atom->load(std::memory_order_acquire); } static bool StoreCAS8(uint8_t &Expected, uint8_t Val, uint64_t Addr) { std::atomic *Atom = reinterpret_cast*>(Addr); return Atom->compare_exchange_strong(Expected, Val); } uint16_t DoLoad16(uint64_t Addr) { uint64_t AlignmentMask = 0b1111; if ((Addr & AlignmentMask) == 15) { // Address crosses over 16byte or 64byte threshold // Needs two loads uint64_t AddrUpper = Addr + 1; uint8_t ActualUpper{}; uint8_t ActualLower{}; // Careful ordering here ActualUpper = LoadAcquire8(AddrUpper); ActualLower = LoadAcquire8(Addr); uint16_t Result = ActualUpper; Result <<= 8; Result |= ActualLower; return Result; } else { AlignmentMask = 0b111; if ((Addr & AlignmentMask) == 7) { // Crosses 8byte boundary // Needs 128bit load // Fits within a 16byte region uint64_t Alignment = Addr & 0b1111; Addr &= ~0b1111ULL; __uint128_t TmpResult = LoadAcquire128(Addr); // Zexts the result uint16_t Result = TmpResult >> (Alignment * 8); return Result; } else { AlignmentMask = 0b11; if ((Addr & AlignmentMask) == 3) { // Crosses 4byte boundary // Needs 64bit Load uint64_t Alignment = Addr & AlignmentMask; Addr &= ~AlignmentMask; std::atomic *Atomic = reinterpret_cast*>(Addr); uint64_t TmpResult = Atomic->load(); // Zexts the result uint16_t Result = TmpResult >> (Alignment * 8); return Result; } else { // Fits within 4byte boundary // Only needs 32bit Load // Only alignment offset will be 1 here uint64_t Alignment = Addr & AlignmentMask; Addr &= ~AlignmentMask; std::atomic *Atomic = reinterpret_cast*>(Addr); uint32_t TmpResult = Atomic->load(); // Zexts the result uint16_t Result = TmpResult >> (Alignment * 8); return Result; } } } } uint32_t DoLoad32(uint64_t Addr) { uint64_t AlignmentMask = 0b1111; if ((Addr & AlignmentMask) > 12) { // Address crosses over 16byte threshold // Needs dual 32bit load uint64_t Alignment = Addr & 0b11; Addr &= ~0b11ULL; uint64_t AddrUpper = Addr + 4; // Careful ordering here uint32_t ActualUpper = LoadAcquire32(AddrUpper); uint32_t ActualLower = LoadAcquire32(Addr); uint64_t Result = ActualUpper; Result <<= 32; Result |= ActualLower; return Result >> (Alignment * 8); } else { AlignmentMask = 0b111; if ((Addr & AlignmentMask) >= 5) { // Crosses 8byte boundary // Needs 128bit load // Fits within a 16byte region uint64_t Alignment = Addr & 0b1111; Addr &= ~0b1111ULL; __uint128_t TmpResult = LoadAcquire128(Addr); return TmpResult >> (Alignment * 8); } else { // Fits within 8byte boundary // Only needs 64bit CAS // Alignments can be [1,5) uint64_t Alignment = Addr & AlignmentMask; Addr &= ~AlignmentMask; std::atomic *Atomic = reinterpret_cast*>(Addr); uint64_t TmpResult = Atomic->load(); return TmpResult >> (Alignment * 8); } } } uint64_t DoLoad64(uint64_t Addr) { uint64_t AlignmentMask = 0b1111; if ((Addr & AlignmentMask) > 8) { uint64_t Alignment = Addr & 0b111; Addr &= ~0b111ULL; uint64_t AddrUpper = Addr + 8; // Crosses a 16byte boundary // Needs two 8 byte loads uint64_t ActualUpper{}; uint64_t ActualLower{}; // Careful ordering here ActualUpper = LoadAcquire64(AddrUpper); ActualLower = LoadAcquire64(Addr); __uint128_t Result = ActualUpper; Result <<= 64; Result |= ActualLower; return Result >> (Alignment * 8); } else { // Fits within a 16byte region uint64_t Alignment = Addr & AlignmentMask; Addr &= ~AlignmentMask; __uint128_t TmpResult = LoadAcquire128(Addr); uint64_t Result = TmpResult >> (Alignment * 8); return Result; } } std::pair DoLoad128(uint64_t Addr) { // Any misalignment here means we cross a 16byte boundary // So we need two 128bit loads uint64_t Alignment = Addr & 0b1111; Addr &= ~0b1111ULL; uint64_t AddrUpper = Addr + 16; union AlignedData { struct { __uint128_t Lower; __uint128_t Upper; } Large; struct { uint8_t Data[32]; } Bytes; }; AlignedData *Data = reinterpret_cast(alloca(sizeof(AlignedData))); Data->Large.Upper = LoadAcquire128(AddrUpper); Data->Large.Lower = LoadAcquire128(Addr); uint64_t ResultLower{}, ResultUpper{}; memcpy(&ResultLower, &Data->Bytes.Data[Alignment], sizeof(uint64_t)); memcpy(&ResultUpper, &Data->Bytes.Data[Alignment + sizeof(uint64_t)], sizeof(uint64_t)); return {ResultLower, ResultUpper}; } static bool RunCASPAL(uint64_t *GPRs, uint32_t Size, uint32_t DesiredReg1, uint32_t DesiredReg2, uint32_t ExpectedReg1, uint32_t ExpectedReg2, uint32_t AddressReg) { if (Size == 0) { // 32bit uint64_t Addr = GPRs[AddressReg]; uint32_t DesiredLower = GPRs[DesiredReg1]; uint32_t DesiredUpper = GPRs[DesiredReg2]; uint32_t ExpectedLower = GPRs[ExpectedReg1]; uint32_t ExpectedUpper = GPRs[ExpectedReg2]; // Cross-cacheline CAS doesn't work on ARM // It isn't even guaranteed to work on x86 // Intel will do a "split lock" which locks the full bus // AMD will tear instead // Both cross-cacheline and cross 16byte both need dual CAS loops that can tear // ARMv8.4 LSE2 solves all atomic issues except cross-cacheline // Check for Split lock across a cacheline if ((Addr & 63) > 56) { FEXCORE_TELEMETRY_SET(SplitLock, 1); } uint64_t AlignmentMask = 0b1111; if ((Addr & AlignmentMask) > 8) { FEXCORE_TELEMETRY_SET(SplitLock16B, 1); uint64_t Alignment = Addr & 0b111; Addr &= ~0b111ULL; uint64_t AddrUpper = Addr + 8; // Crosses a 16byte boundary // Need to do 256bit atomic, but since that doesn't exist we need to do a dual CAS loop __uint128_t Mask = ~0ULL; Mask <<= Alignment * 8; __uint128_t NegMask = ~Mask; __uint128_t TmpExpected{}; __uint128_t TmpDesired{}; __uint128_t Desired = DesiredUpper; Desired <<= 32; Desired |= DesiredLower; Desired <<= Alignment * 8; __uint128_t Expected = ExpectedUpper; Expected <<= 32; Expected |= ExpectedLower; Expected <<= Alignment * 8; while (1) { __uint128_t LoadOrderUpper = LoadAcquire64(AddrUpper); LoadOrderUpper <<= 64; __uint128_t TmpActual = LoadOrderUpper | LoadAcquire64(Addr); // Set up expected TmpExpected = TmpActual; TmpExpected &= NegMask; TmpExpected |= Expected; // Set up desired TmpDesired = TmpExpected; TmpDesired &= NegMask; TmpDesired |= Desired; uint64_t TmpExpectedLower = TmpExpected; uint64_t TmpExpectedUpper = TmpExpected >> 64; uint64_t TmpDesiredLower = TmpDesired; uint64_t TmpDesiredUpper = TmpDesired >> 64; if (TmpExpected == TmpActual) { if (StoreCAS64(TmpExpectedUpper, TmpDesiredUpper, AddrUpper)) { if (StoreCAS64(TmpExpectedLower, TmpDesiredLower, Addr)) { // Stored successfully return true; } else { // CAS managed to tear, we can't really solve this // Continue down the path to let the guest know values weren't expected FEXCORE_TELEMETRY_SET(Cas128Tear, 1); } } TmpExpected = TmpExpectedUpper; TmpExpected <<= 64; TmpExpected |= TmpExpectedLower; } else { // Mismatch up front TmpExpected = TmpActual; } // Not successful // Now we need to check the results to see if we need to try again __uint128_t FailedResultOurBits = TmpExpected & Mask; __uint128_t FailedResultNotOurBits = TmpExpected & NegMask; __uint128_t FailedDesiredOurBits = TmpDesired & Mask; __uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask; if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) { // If the bits changed that weren't part of our regular CAS then we need to try again continue; } if ((FailedResultOurBits ^ FailedDesiredOurBits) != 0) { // If the bits changed that we were wanting to change then we have failed and can return // We need to extract the bits and return them in EXPECTED uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8); GPRs[ExpectedReg1] = FailedResult & ~0U; GPRs[ExpectedReg2] = FailedResult >> 32; return true; } // This happens in the case that between Load and CAS that something has store our desired in to the memory location // This means our CAS fails because what we wanted to store was already stored uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8); GPRs[ExpectedReg1] = FailedResult & ~0U; GPRs[ExpectedReg2] = FailedResult >> 32; return true; } } else { // Fits within a 16byte region uint64_t Alignment = Addr & 0b1111; Addr &= ~0b1111ULL; std::atomic<__uint128_t> *Atomic128 = reinterpret_cast*>(Addr); __uint128_t Mask = ~0ULL; Mask <<= Alignment * 8; __uint128_t NegMask = ~Mask; __uint128_t TmpExpected{}; __uint128_t TmpDesired{}; __uint128_t Desired = (uint64_t)DesiredUpper << 32 | DesiredLower; Desired <<= Alignment * 8; __uint128_t Expected = (uint64_t)ExpectedUpper << 32 | ExpectedLower; Expected <<= Alignment * 8; while (1) { TmpExpected = Atomic128->load(); // Set up expected TmpExpected &= NegMask; TmpExpected |= Expected; // Set up desired TmpDesired = TmpExpected; TmpDesired &= NegMask; TmpDesired |= Desired; bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired); if (CASResult) { // Successful, so we are done return true; } else { // Not successful // Now we need to check the results to see if we need to try again __uint128_t FailedResultOurBits = TmpExpected & Mask; __uint128_t FailedResultNotOurBits = TmpExpected & NegMask; __uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask; if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) { // If the bits changed that weren't part of our regular CAS then we need to try again continue; } // This happens in the case that between Load and CAS that something has store our desired in to the memory location // This means our CAS fails because what we wanted to store was already stored uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8); GPRs[ExpectedReg1] = FailedResult & ~0U; GPRs[ExpectedReg2] = FailedResult >> 32; return true; } } } } return false; } bool HandleCASPAL(uint32_t Instr, uint64_t *GPRs) { uint32_t Size = (Instr >> 30) & 1; uint32_t DesiredReg1 = Instr & 0b11111; uint32_t DesiredReg2 = DesiredReg1 + 1; uint32_t ExpectedReg1 = (Instr >> 16) & 0b11111; uint32_t ExpectedReg2 = ExpectedReg1 + 1; uint32_t AddressReg = (Instr >> 5) & 0b11111; return RunCASPAL(GPRs, Size, DesiredReg1, DesiredReg2, ExpectedReg1, ExpectedReg2, AddressReg); } uint64_t HandleCASPAL_ARMv8(uint32_t Instr, uintptr_t ProgramCounter, uint64_t *GPRs) { // caspair // [1] ldaxp(TMP2.W(), TMP3.W(), MemOperand(MemSrc)); <-- DataReg & AddrReg // [2] cmp(TMP2.W(), Expected.first.W()); <-- ExpectedReg1 // [3] ccmp(TMP3.W(), Expected.second.W(), NoFlag, Condition::eq); <-- ExpectedREg2 // [4] b(&LoopNotExpected, Condition::ne); // [5] stlxp(TMP2.W(), Desired.first.W(), Desired.second.W(), MemOperand(MemSrc)); <-- DesiredReg // [6] cbnz(TMP2.W(), &LoopTop); // [7] mov(Dst.first.W(), Expected.first.W()); // [8] mov(Dst.second.W(), Expected.second.W()); // [9] b(&LoopExpected); // [10] mov(Dst.first.W(), TMP2.W()); // [11] mov(Dst.second.W(), TMP3.W()); // [12] clrex(); uint32_t *PC = (uint32_t*)ProgramCounter; uint32_t Size = (Instr >> 30) & 1; uint32_t AddrReg = (Instr >> 5) & 0x1F; uint32_t DataReg = Instr & 0x1F; uint32_t DataReg2 = (Instr >> 10) & 0x1F; uint32_t ExpectedReg1{}; uint32_t ExpectedReg2{}; uint32_t DesiredReg1{}; uint32_t DesiredReg2{}; if(Size == 1) { // 64-bit pair happens on paranoid vector loads // [1] ldaxp(TMP1, TMP2, MemSrc); // [2] clrex(); // // 64-bit pair happens on paranoid vector stores // [1] ldaxp(xzr, TMP3, MemSrc); // <- Can hit SIGBUS // [2] stlxp(TMP3, TMP1, TMP2, MemSrc); // <- Can also hit SIGBUS // [3] cbnz(TMP3, &B); // < Overwritten with DMB if (DataReg == 31) { } else { uint32_t NextInstr = PC[1]; if ((NextInstr & ArchHelpers::Arm64::CLREX_MASK) == ArchHelpers::Arm64::CLREX_INST) { uint64_t Addr = GPRs[AddrReg]; auto Res = DoLoad128(Addr); // We set the result register if it isn't a zero register if (DataReg != 31) { GPRs[DataReg] = std::get<0>(Res); } if (DataReg2 != 31) { GPRs[DataReg2] = std::get<1>(Res); } // Skip ldaxp and clrex return 2 * sizeof(uint32_t); } } return 0; } //Only 32-bit pairs for(int i = 1; i < 10; i++) { uint32_t NextInstr = PC[i]; if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_INST || (NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_SHIFT_INST) { ExpectedReg1 = GetRmReg(NextInstr); } else if ((NextInstr & ArchHelpers::Arm64::CCMP_MASK) == ArchHelpers::Arm64::CCMP_INST) { ExpectedReg2 = GetRmReg(NextInstr); } else if ((NextInstr & ArchHelpers::Arm64::STLXP_MASK) == ArchHelpers::Arm64::STLXP_INST) { DesiredReg1 = (NextInstr & 0x1F); DesiredReg2 = (NextInstr >> 10) & 0x1F; } } //mov expected into the temp registers used by JIT GPRs[DataReg] = GPRs[ExpectedReg1]; GPRs[DataReg2] = GPRs[ExpectedReg2]; if(RunCASPAL(GPRs, Size, DesiredReg1, DesiredReg2, DataReg, DataReg2, AddrReg)) { return 9 * sizeof(uint32_t); // skip to mov + clrex } else { return 0; } } static bool HandleAtomicVectorStore(uint32_t Instr, uintptr_t ProgramCounter) { uint32_t *PC = (uint32_t*)ProgramCounter; uint32_t Size = (Instr >> 30) & 1; uint32_t DataReg = Instr & 0x1F; if(Size == 1) { // 64-bit pair happens on paranoid vector stores // [0] ldaxp(xzr, TMP3, MemSrc); // <- Can hit SIGBUS. Overwritten with DMB // [1] stlxp(TMP3, TMP1, TMP2, MemSrc); // <- Can also hit SIGBUS // [2] cbnz(TMP3, &B); // < Overwritten with DMB if (DataReg == 31) { uint32_t NextInstr = PC[1]; uint32_t AddrReg = (NextInstr >> 5) & 0x1F; DataReg = NextInstr & 0x1F; uint32_t DataReg2 = (NextInstr >> 10) & 0x1F; uint32_t STP = (0b10 << 30) | (0b101001000000000 << 15) | (DataReg2 << 10) | (AddrReg << 5) | DataReg; PC[0] = DMB; PC[1] = STP; PC[2] = DMB; // Back up one instruction and have another go ClearICache(&PC[0], 16); return true; } } return false; } template using CASExpectedFn = T (*)(T Src, T Expected); template using CASDesiredFn = T (*)(T Src, T Desired); template static uint16_t DoCAS16( uint16_t DesiredSrc, uint16_t ExpectedSrc, uint64_t Addr, CASExpectedFn ExpectedFunction, CASDesiredFn DesiredFunction) { if ((Addr & 63) == 63) { FEXCORE_TELEMETRY_SET(SplitLock, 1); } // 16 bit uint64_t AlignmentMask = 0b1111; if ((Addr & AlignmentMask) == 15) { FEXCORE_TELEMETRY_SET(SplitLock16B, 1); // Address crosses over 16byte or 64byte threshold // Need a dual 8bit CAS loop uint64_t AddrUpper = Addr + 1; while (1) { uint8_t ActualUpper{}; uint8_t ActualLower{}; // Careful ordering here ActualUpper = LoadAcquire8(AddrUpper); ActualLower = LoadAcquire8(Addr); uint16_t Actual = ActualUpper; Actual <<= 8; Actual |= ActualLower; uint16_t Desired = DesiredFunction(Actual, DesiredSrc); uint8_t DesiredLower = Desired; uint8_t DesiredUpper = Desired >> 8; uint16_t Expected = ExpectedFunction(Actual, ExpectedSrc); uint8_t ExpectedLower = Expected; uint8_t ExpectedUpper = Expected >> 8; bool Tear = false; if (ActualUpper == ExpectedUpper && ActualLower == ExpectedLower) { if (StoreCAS8(ExpectedUpper, DesiredUpper, AddrUpper)) { if (StoreCAS8(ExpectedLower, DesiredLower, Addr)) { // Stored successfully return Expected; } else { // CAS managed to tear, we can't really solve this // Continue down the path to let the guest know values weren't expected Tear = true; FEXCORE_TELEMETRY_SET(Cas16Tear, 1); } } ActualLower = ExpectedLower; ActualUpper = ExpectedUpper; } // If the bits changed that we were wanting to change then we have failed and can return // We need to extract the bits and return them in EXPECTED uint16_t FailedResult = ActualUpper; FailedResult <<= 8; FailedResult |= ActualLower; if constexpr (Retry) { if (Tear) { // If we are retrying and tearing then we can't do anything here // XXX: Resolve with TME return FailedResult; } else { // We can retry safely } } else { // Without Retry (CAS) then we have failed regardless of tear // CAS failed but handled successfully return FailedResult; } } } else { AlignmentMask = 0b111; if ((Addr & AlignmentMask) == 7) { // Crosses 8byte boundary // Needs 128bit CAS // Fits within a 16byte region uint64_t Alignment = Addr & 0b1111; Addr &= ~0b1111ULL; std::atomic<__uint128_t> *Atomic128 = reinterpret_cast*>(Addr); __uint128_t Mask = 0xFFFF; Mask <<= Alignment * 8; __uint128_t NegMask = ~Mask; __uint128_t TmpExpected{}; __uint128_t TmpDesired{}; while (1) { TmpExpected = Atomic128->load(); __uint128_t Desired = DesiredFunction(TmpExpected >> (Alignment * 8), DesiredSrc); Desired <<= Alignment * 8; __uint128_t Expected = ExpectedFunction(TmpExpected >> (Alignment * 8), ExpectedSrc); Expected <<= Alignment * 8; // Set up expected TmpExpected &= NegMask; TmpExpected |= Expected; // Set up desired TmpDesired = TmpExpected; TmpDesired &= NegMask; TmpDesired |= Desired; bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired); if (CASResult) { // Successful, so we are done return Expected >> (Alignment * 8); } else { if constexpr (Retry) { // If we failed but we have enabled retry then just retry without checking results // CAS can't retry but atomic memory ops need to retry until passing continue; } // Not successful // Now we need to check the results to see if we need to try again __uint128_t FailedResultOurBits = TmpExpected & Mask; __uint128_t FailedResultNotOurBits = TmpExpected & NegMask; __uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask; if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) { // If the bits changed that weren't part of our regular CAS then we need to try again continue; } // This happens in the case that between Load and CAS that something has store our desired in to the memory location // This means our CAS fails because what we wanted to store was already stored uint16_t FailedResult = FailedResultOurBits >> (Alignment * 8); // CAS failed but handled successfully return FailedResult; } } } else { AlignmentMask = 0b11; if ((Addr & AlignmentMask) == 3) { // Crosses 4byte boundary // Needs 64bit CAS uint64_t Alignment = Addr & AlignmentMask; Addr &= ~AlignmentMask; uint64_t Mask = 0xFFFF; Mask <<= Alignment * 8; uint64_t NegMask = ~Mask; uint64_t TmpExpected{}; uint64_t TmpDesired{}; std::atomic *Atomic = reinterpret_cast*>(Addr); while (1) { TmpExpected = Atomic->load(); uint64_t Desired = DesiredFunction(TmpExpected >> (Alignment * 8), DesiredSrc); Desired <<= Alignment * 8; uint64_t Expected = ExpectedFunction(TmpExpected >> (Alignment * 8), ExpectedSrc); Expected <<= Alignment * 8; // Set up expected TmpExpected &= NegMask; TmpExpected |= Expected; // Set up desired TmpDesired = TmpExpected; TmpDesired &= NegMask; TmpDesired |= Desired; bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired); if (CASResult) { // Successful, so we are done return Expected >> (Alignment * 8); } else { if constexpr (Retry) { // If we failed but we have enabled retry then just retry without checking results // CAS can't retry but atomic memory ops need to retry until passing continue; } // Not successful // Now we need to check the results to see if we can try again uint64_t FailedResultOurBits = TmpExpected & Mask; uint64_t FailedResultNotOurBits = TmpExpected & NegMask; uint64_t FailedDesiredNotOurBits = TmpDesired & NegMask; if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) { // If the bits changed that weren't part of our regular CAS then we need to try again continue; } // This happens in the case that between Load and CAS that something has store our desired in to the memory location // This means our CAS fails because what we wanted to store was already stored uint16_t FailedResult = FailedResultOurBits >> (Alignment * 8); // CAS failed but handled successfully return FailedResult; } } } else { // Fits within 4byte boundary // Only needs 32bit CAS // Only alignment offset will be 1 here uint64_t Alignment = Addr & AlignmentMask; Addr &= ~AlignmentMask; uint32_t Mask = 0xFFFF; Mask <<= Alignment * 8; uint32_t NegMask = ~Mask; uint32_t TmpExpected{}; uint32_t TmpDesired{}; std::atomic *Atomic = reinterpret_cast*>(Addr); while (1) { TmpExpected = Atomic->load(); uint32_t Desired = DesiredFunction(TmpExpected >> (Alignment * 8), DesiredSrc); Desired <<= Alignment * 8; uint32_t Expected = ExpectedFunction(TmpExpected >> (Alignment * 8), ExpectedSrc); Expected <<= Alignment * 8; // Set up expected TmpExpected &= NegMask; TmpExpected |= Expected; // Set up desired TmpDesired = TmpExpected; TmpDesired &= NegMask; TmpDesired |= Desired; bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired); if (CASResult) { // Successful, so we are done return Expected >> (Alignment * 8); } else { if constexpr (Retry) { // If we failed but we have enabled retry then just retry without checking results // CAS can't retry but atomic memory ops need to retry until passing continue; } // Not successful // Now we need to check the results to see if we can try again uint32_t FailedResultOurBits = TmpExpected & Mask; uint32_t FailedResultNotOurBits = TmpExpected & NegMask; uint32_t FailedDesiredNotOurBits = TmpDesired & NegMask; if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) { // If the bits changed that weren't part of our regular CAS then we need to try again continue; } // This happens in the case that between Load and CAS that something has store our desired in to the memory location // This means our CAS fails because what we wanted to store was already stored uint16_t FailedResult = FailedResultOurBits >> (Alignment * 8); // CAS failed but handled successfully return FailedResult; } } } } } } template static uint32_t DoCAS32( uint32_t DesiredSrc, uint32_t ExpectedSrc, uint64_t Addr, CASExpectedFn ExpectedFunction, CASDesiredFn DesiredFunction) { if ((Addr & 63) > 60) { FEXCORE_TELEMETRY_SET(SplitLock, 1); } // 32 bit uint64_t AlignmentMask = 0b1111; if ((Addr & AlignmentMask) > 12) { FEXCORE_TELEMETRY_SET(SplitLock16B, 1); // Address crosses over 16byte threshold // Needs dual 4 byte CAS loop uint64_t Alignment = Addr & 0b11; Addr &= ~0b11; uint64_t AddrUpper = Addr + 4; uint64_t Mask = ~0U; Mask <<= Alignment * 8; uint64_t NegMask = ~Mask; // Careful ordering here while (1) { uint64_t LoadOrderUpper = LoadAcquire32(AddrUpper); LoadOrderUpper <<= 32; uint64_t TmpActual = LoadOrderUpper | LoadAcquire32(Addr); uint64_t Desired = DesiredFunction(TmpActual >> (Alignment * 8), DesiredSrc); uint64_t Expected = ExpectedFunction(TmpActual >> (Alignment * 8), ExpectedSrc); uint64_t TmpExpected = TmpActual; TmpExpected &= NegMask; TmpExpected |= Expected << (Alignment * 8); uint64_t TmpDesired = TmpExpected; TmpDesired &= NegMask; TmpDesired |= Desired << (Alignment * 8); bool Tear = false; if (TmpExpected == TmpActual) { uint32_t TmpExpectedLower = TmpExpected; uint32_t TmpExpectedUpper = TmpExpected >> 32; uint32_t TmpDesiredLower = TmpDesired; uint32_t TmpDesiredUpper = TmpDesired >> 32; if (StoreCAS32(TmpExpectedUpper, TmpDesiredUpper, AddrUpper)) { if (StoreCAS32(TmpExpectedLower, TmpDesiredLower, Addr)) { // Stored successfully return Expected; } else { // CAS managed to tear, we can't really solve this // Continue down the path to let the guest know values weren't expected Tear = true; FEXCORE_TELEMETRY_SET(Cas32Tear, 1); } } TmpExpected = TmpExpectedUpper; TmpExpected <<= 32; TmpExpected |= TmpExpectedLower; } else { // Mismatch up front TmpExpected = TmpActual; } // Not successful // Now we need to check the results to see if we need to try again uint64_t FailedResultOurBits = TmpExpected & Mask; uint64_t FailedResultNotOurBits = TmpExpected & NegMask; uint64_t FailedDesiredNotOurBits = TmpDesired & NegMask; if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) { // If the bits changed that weren't part of our regular CAS then we need to try again continue; } // This happens in the case that between Load and CAS that something has store our desired in to the memory location // This means our CAS fails because what we wanted to store was already stored uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8); if constexpr (Retry) { if (Tear) { // If we are retrying and tearing then we can't do anything here // XXX: Resolve with TME return FailedResult; } else { // We can retry safely } } else { // Without Retry (CAS) then we have failed regardless of tear // CAS failed but handled successfully return FailedResult; } } } else { AlignmentMask = 0b111; if ((Addr & AlignmentMask) >= 5) { // Crosses 8byte boundary // Needs 128bit CAS // Fits within a 16byte region uint64_t Alignment = Addr & 0b1111; Addr &= ~0b1111ULL; std::atomic<__uint128_t> *Atomic128 = reinterpret_cast*>(Addr); __uint128_t Mask = ~0U; Mask <<= Alignment * 8; __uint128_t NegMask = ~Mask; __uint128_t TmpExpected{}; __uint128_t TmpDesired{}; while (1) { __uint128_t TmpActual = Atomic128->load(); __uint128_t Desired = DesiredFunction(TmpActual >> (Alignment * 8), DesiredSrc); __uint128_t Expected = ExpectedFunction(TmpActual >> (Alignment * 8), ExpectedSrc); // Set up expected TmpExpected = TmpActual; TmpExpected &= NegMask; TmpExpected |= Expected << (Alignment * 8); // Set up desired TmpDesired = TmpExpected; TmpDesired &= NegMask; TmpDesired |= Desired << (Alignment * 8); bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired); if (CASResult) { // Stored successfully return Expected; } else { if constexpr (Retry) { // If we failed but we have enabled retry then just retry without checking results // CAS can't retry but atomic memory ops need to retry until passing continue; } // Not successful // Now we need to check the results to see if we need to try again __uint128_t FailedResultOurBits = TmpExpected & Mask; __uint128_t FailedResultNotOurBits = TmpExpected & NegMask; __uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask; if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) { // If the bits changed that weren't part of our regular CAS then we need to try again continue; } // This happens in the case that between Load and CAS that something has store our desired in to the memory location // This means our CAS fails because what we wanted to store was already stored uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8); // CAS failed but handled successfully return FailedResult; } } } else { // Fits within 8byte boundary // Only needs 64bit CAS // Alignments can be [1,5) uint64_t Alignment = Addr & AlignmentMask; Addr &= ~AlignmentMask; uint64_t Mask = ~0U; Mask <<= Alignment * 8; uint64_t NegMask = ~Mask; uint64_t TmpExpected{}; uint64_t TmpDesired{}; std::atomic *Atomic = reinterpret_cast*>(Addr); while (1) { uint64_t TmpActual = Atomic->load(); uint64_t Desired = DesiredFunction(TmpActual >> (Alignment * 8), DesiredSrc); uint64_t Expected = ExpectedFunction(TmpActual >> (Alignment * 8), ExpectedSrc); // Set up expected TmpExpected = TmpActual; TmpExpected &= NegMask; TmpExpected |= Expected << (Alignment * 8); // Set up desired TmpDesired = TmpExpected; TmpDesired &= NegMask; TmpDesired |= Desired << (Alignment * 8); bool CASResult = Atomic->compare_exchange_strong(TmpExpected, TmpDesired); if (CASResult) { // Stored successfully return Expected; } else { if constexpr (Retry) { // If we failed but we have enabled retry then just retry without checking results // CAS can't retry but atomic memory ops need to retry until passing continue; } // Not successful // Now we need to check the results to see if we can try again uint64_t FailedResultOurBits = TmpExpected & Mask; uint64_t FailedResultNotOurBits = TmpExpected & NegMask; uint64_t FailedDesiredNotOurBits = TmpDesired & NegMask; if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) { // If the bits changed that weren't part of our regular CAS then we need to try again continue; } // This happens in the case that between Load and CAS that something has store our desired in to the memory location // This means our CAS fails because what we wanted to store was already stored uint32_t FailedResult = FailedResultOurBits >> (Alignment * 8); // CAS failed but handled successfully return FailedResult; } } } } } template static uint64_t DoCAS64( uint64_t DesiredSrc, uint64_t ExpectedSrc, uint64_t Addr, CASExpectedFn ExpectedFunction, CASDesiredFn DesiredFunction) { if ((Addr & 63) > 56) { FEXCORE_TELEMETRY_SET(SplitLock, 1); } // 64bit uint64_t AlignmentMask = 0b1111; if ((Addr & AlignmentMask) > 8) { FEXCORE_TELEMETRY_SET(SplitLock16B, 1); uint64_t Alignment = Addr & 0b111; Addr &= ~0b111ULL; uint64_t AddrUpper = Addr + 8; // Crosses a 16byte boundary // Need to do 256bit atomic, but since that doesn't exist we need to do a dual CAS loop __uint128_t Mask = ~0ULL; Mask <<= Alignment * 8; __uint128_t NegMask = ~Mask; __uint128_t TmpExpected{}; __uint128_t TmpDesired{}; while (1) { __uint128_t LoadOrderUpper = LoadAcquire64(AddrUpper); LoadOrderUpper <<= 64; __uint128_t TmpActual = LoadOrderUpper | LoadAcquire64(Addr); __uint128_t Desired = DesiredFunction(TmpActual >> (Alignment * 8), DesiredSrc); __uint128_t Expected = ExpectedFunction(TmpActual >> (Alignment * 8), ExpectedSrc); // Set up expected TmpExpected = TmpActual; TmpExpected &= NegMask; TmpExpected |= Expected << (Alignment * 8); // Set up desired TmpDesired = TmpExpected; TmpDesired &= NegMask; TmpDesired |= Desired << (Alignment * 8); uint64_t TmpExpectedLower = TmpExpected; uint64_t TmpExpectedUpper = TmpExpected >> 64; uint64_t TmpDesiredLower = TmpDesired; uint64_t TmpDesiredUpper = TmpDesired >> 64; bool Tear = false; if (TmpExpected == TmpActual) { if (StoreCAS64(TmpExpectedUpper, TmpDesiredUpper, AddrUpper)) { if (StoreCAS64(TmpExpectedLower, TmpDesiredLower, Addr)) { // Stored successfully return Expected; } else { // CAS managed to tear, we can't really solve this // Continue down the path to let the guest know values weren't expected Tear = true; FEXCORE_TELEMETRY_SET(Cas64Tear, 1); } } TmpExpected = TmpExpectedUpper; TmpExpected <<= 64; TmpExpected |= TmpExpectedLower; } else { // Mismatch up front TmpExpected = TmpActual; } // Not successful // Now we need to check the results to see if we need to try again __uint128_t FailedResultOurBits = TmpExpected & Mask; __uint128_t FailedResultNotOurBits = TmpExpected & NegMask; __uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask; if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) { // If the bits changed that weren't part of our regular CAS then we need to try again continue; } // This happens in the case that between Load and CAS that something has store our desired in to the memory location // This means our CAS fails because what we wanted to store was already stored uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8); if constexpr (Retry) { if (Tear) { // If we are retrying and tearing then we can't do anything here // XXX: Resolve with TME return FailedResult; } else { // We can retry safely } } else { // Without Retry (CAS) then we have failed regardless of tear // CAS failed but handled successfully return FailedResult; } } } else { // Fits within a 16byte region uint64_t Alignment = Addr & AlignmentMask; Addr &= ~AlignmentMask; std::atomic<__uint128_t> *Atomic128 = reinterpret_cast*>(Addr); __uint128_t Mask = ~0ULL; Mask <<= Alignment * 8; __uint128_t NegMask = ~Mask; __uint128_t TmpExpected{}; __uint128_t TmpDesired{}; while (1) { __uint128_t TmpActual = Atomic128->load(); __uint128_t Desired = DesiredFunction(TmpActual >> (Alignment * 8), DesiredSrc); __uint128_t Expected = ExpectedFunction(TmpActual >> (Alignment * 8), ExpectedSrc); // Set up expected TmpExpected = TmpActual; TmpExpected &= NegMask; TmpExpected |= Expected << (Alignment * 8); // Set up desired TmpDesired = TmpExpected; TmpDesired &= NegMask; TmpDesired |= Desired << (Alignment * 8); bool CASResult = Atomic128->compare_exchange_strong(TmpExpected, TmpDesired); if (CASResult) { // Stored successfully return Expected; } else { if constexpr (Retry) { // If we failed but we have enabled retry then just retry without checking results // CAS can't retry but atomic memory ops need to retry until passing continue; } // Not successful // Now we need to check the results to see if we need to try again __uint128_t FailedResultOurBits = TmpExpected & Mask; __uint128_t FailedResultNotOurBits = TmpExpected & NegMask; __uint128_t FailedDesiredNotOurBits = TmpDesired & NegMask; if ((FailedResultNotOurBits ^ FailedDesiredNotOurBits) != 0) { // If the bits changed that weren't part of our regular CAS then we need to try again continue; } // This happens in the case that between Load and CAS that something has store our desired in to the memory location // This means our CAS fails because what we wanted to store was already stored uint64_t FailedResult = FailedResultOurBits >> (Alignment * 8); // CAS failed but handled successfully return FailedResult; } } } } static bool RunCASAL(uint64_t *GPRs, uint32_t Size, uint32_t DesiredReg, uint32_t ExpectedReg, uint32_t AddressReg) { uint64_t Addr = GPRs[AddressReg]; // Cross-cacheline CAS doesn't work on ARM // It isn't even guaranteed to work on x86 // Intel will do a "split lock" which locks the full bus // AMD will tear instead // Both cross-cacheline and cross 16byte both need dual CAS loops that can tear // ARMv8.4 LSE2 solves all atomic issues except cross-cacheline // ARM's TME extension solves the cross-cacheline problem // 8bit can't be unaligned // Only need to handle 16, 32, 64 if (Size == 2) { auto Res = DoCAS16( GPRs[DesiredReg], GPRs[ExpectedReg], Addr, [](uint16_t, uint16_t Expected) -> uint16_t { // Expected is just Expected return Expected; }, [](uint16_t, uint16_t Desired) -> uint16_t { // Desired is just Desired return Desired; }); // Regardless of pass or fail // We set the result register if it isn't a zero register if (ExpectedReg != 31) { GPRs[ExpectedReg] = Res; } return true; } else if (Size == 4) { auto Res = DoCAS32( GPRs[DesiredReg], GPRs[ExpectedReg], Addr, [](uint32_t, uint32_t Expected) -> uint32_t { // Expected is just Expected return Expected; }, [](uint32_t, uint32_t Desired) -> uint32_t { // Desired is just Desired return Desired; }); // Regardless of pass or fail // We set the result register if it isn't a zero register if (ExpectedReg != 31) { GPRs[ExpectedReg] = Res; } return true; } else if (Size == 8) { auto Res = DoCAS64( GPRs[DesiredReg], GPRs[ExpectedReg], Addr, [](uint64_t, uint64_t Expected) -> uint64_t { // Expected is just Expected return Expected; }, [](uint64_t, uint64_t Desired) -> uint64_t { // Desired is just Desired return Desired; }); // Regardless of pass or fail // We set the result register if it isn't a zero register if (ExpectedReg != 31) { GPRs[ExpectedReg] = Res; } return true; } return false; } static bool HandleCASAL(uint64_t *GPRs, uint32_t Instr) { uint32_t Size = 1 << (Instr >> 30); uint32_t DesiredReg = Instr & 0b11111; uint32_t ExpectedReg = (Instr >> 16) & 0b11111; uint32_t AddressReg = (Instr >> 5) & 0b11111; return RunCASAL(GPRs, Size, DesiredReg, ExpectedReg, AddressReg); } static bool HandleAtomicMemOp(uint32_t Instr, uint64_t *GPRs) { uint32_t Size = 1 << (Instr >> 30); uint32_t ResultReg = Instr & 0b11111; uint32_t SourceReg = (Instr >> 16) & 0b11111; uint32_t AddressReg = (Instr >> 5) & 0b11111; uint64_t Addr = GPRs[AddressReg]; uint8_t Op = (Instr >> 12) & 0xF; if (Size == 2) { auto NOPExpected = [](uint16_t SrcVal, uint16_t) -> uint16_t { return SrcVal; }; auto ADDDesired = [](uint16_t SrcVal, uint16_t Desired) -> uint16_t { return SrcVal + Desired; }; auto CLRDesired = [](uint16_t SrcVal, uint16_t Desired) -> uint16_t { return SrcVal & ~Desired; }; auto EORDesired = [](uint16_t SrcVal, uint16_t Desired) -> uint16_t { return SrcVal ^ Desired; }; auto SETDesired = [](uint16_t SrcVal, uint16_t Desired) -> uint16_t { return SrcVal | Desired; }; auto SWAPDesired = [](uint16_t SrcVal, uint16_t Desired) -> uint16_t { return Desired; }; CASDesiredFn DesiredFunction{}; switch (Op) { case ATOMIC_ADD_OP: DesiredFunction = ADDDesired; break; case ATOMIC_CLR_OP: DesiredFunction = CLRDesired; break; case ATOMIC_EOR_OP: DesiredFunction = EORDesired; break; case ATOMIC_SET_OP: DesiredFunction = SETDesired; break; case ATOMIC_SWAP_OP: DesiredFunction = SWAPDesired; break; default: LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", Op); return false; } auto Res = DoCAS16( GPRs[SourceReg], 0, // Unused Addr, NOPExpected, DesiredFunction); // If we passed and our destination register is not zero // Then we need to update the result register with what was in memory if (ResultReg != 31) { GPRs[ResultReg] = Res; } return true; } else if (Size == 4) { auto NOPExpected = [](uint32_t SrcVal, uint32_t) -> uint32_t { return SrcVal; }; auto ADDDesired = [](uint32_t SrcVal, uint32_t Desired) -> uint32_t { return SrcVal + Desired; }; auto CLRDesired = [](uint32_t SrcVal, uint32_t Desired) -> uint32_t { return SrcVal & ~Desired; }; auto EORDesired = [](uint32_t SrcVal, uint32_t Desired) -> uint32_t { return SrcVal ^ Desired; }; auto SETDesired = [](uint32_t SrcVal, uint32_t Desired) -> uint32_t { return SrcVal | Desired; }; auto SWAPDesired = [](uint32_t SrcVal, uint32_t Desired) -> uint32_t { return Desired; }; CASDesiredFn DesiredFunction{}; switch (Op) { case ATOMIC_ADD_OP: DesiredFunction = ADDDesired; break; case ATOMIC_CLR_OP: DesiredFunction = CLRDesired; break; case ATOMIC_EOR_OP: DesiredFunction = EORDesired; break; case ATOMIC_SET_OP: DesiredFunction = SETDesired; break; case ATOMIC_SWAP_OP: DesiredFunction = SWAPDesired; break; default: LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", Op); return false; } auto Res = DoCAS32( GPRs[SourceReg], 0, // Unused Addr, NOPExpected, DesiredFunction); // If we passed and our destination register is not zero // Then we need to update the result register with what was in memory if (ResultReg != 31) { GPRs[ResultReg] = Res; } return true; } else if (Size == 8) { auto NOPExpected = [](uint64_t SrcVal, uint64_t) -> uint64_t { return SrcVal; }; auto ADDDesired = [](uint64_t SrcVal, uint64_t Desired) -> uint64_t { return SrcVal + Desired; }; auto CLRDesired = [](uint64_t SrcVal, uint64_t Desired) -> uint64_t { return SrcVal & ~Desired; }; auto EORDesired = [](uint64_t SrcVal, uint64_t Desired) -> uint64_t { return SrcVal ^ Desired; }; auto SETDesired = [](uint64_t SrcVal, uint64_t Desired) -> uint64_t { return SrcVal | Desired; }; auto SWAPDesired = [](uint64_t SrcVal, uint64_t Desired) -> uint64_t { return Desired; }; CASDesiredFn DesiredFunction{}; switch (Op) { case ATOMIC_ADD_OP: DesiredFunction = ADDDesired; break; case ATOMIC_CLR_OP: DesiredFunction = CLRDesired; break; case ATOMIC_EOR_OP: DesiredFunction = EORDesired; break; case ATOMIC_SET_OP: DesiredFunction = SETDesired; break; case ATOMIC_SWAP_OP: DesiredFunction = SWAPDesired; break; default: LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", Op); return false; } auto Res = DoCAS64( GPRs[SourceReg], 0, // Unused Addr, NOPExpected, DesiredFunction); // If we passed and our destination register is not zero // Then we need to update the result register with what was in memory if (ResultReg != 31) { GPRs[ResultReg] = Res; } return true; } return false; } static bool HandleAtomicLoad(uint32_t Instr, uint64_t *GPRs, int64_t Offset) { uint32_t Size = 1 << (Instr >> 30); uint32_t ResultReg = Instr & 0b11111; uint32_t AddressReg = (Instr >> 5) & 0b11111; uint64_t Addr = GPRs[AddressReg] + Offset; if (Size == 2) { auto Res = DoLoad16(Addr); // We set the result register if it isn't a zero register if (ResultReg != 31) { GPRs[ResultReg] = Res; } return true; } else if (Size == 4) { auto Res = DoLoad32(Addr); // We set the result register if it isn't a zero register if (ResultReg != 31) { GPRs[ResultReg] = Res; } return true; } else if (Size == 8) { auto Res = DoLoad64(Addr); // We set the result register if it isn't a zero register if (ResultReg != 31) { GPRs[ResultReg] = Res; } return true; } return false; } static bool HandleAtomicStore(uint32_t Instr, uint64_t *GPRs, int64_t Offset) { uint32_t Size = 1 << (Instr >> 30); uint32_t DataReg = Instr & 0x1F; uint32_t AddressReg = (Instr >> 5) & 0b11111; uint64_t Addr = GPRs[AddressReg] + Offset; constexpr bool DoRetry = false; if (Size == 2) { DoCAS16( GPRs[DataReg], 0, // Unused Addr, [](uint16_t SrcVal, uint16_t) -> uint16_t { // Expected is just src return SrcVal; }, [](uint16_t, uint16_t Desired) -> uint16_t { // Desired is just Desired return Desired; }); return true; } else if (Size == 4) { DoCAS32( GPRs[DataReg], 0, // Unused Addr, [](uint32_t SrcVal, uint32_t) -> uint32_t { // Expected is just src return SrcVal; }, [](uint32_t, uint32_t Desired) -> uint32_t { // Desired is just Desired return Desired; }); return true; } else if (Size == 8) { DoCAS64( GPRs[DataReg], 0, // Unused Addr, [](uint64_t SrcVal, uint64_t) -> uint64_t { // Expected is just src return SrcVal; }, [](uint64_t, uint64_t Desired) -> uint64_t { // Desired is just Desired return Desired; }); return true; } return false; } static uint64_t HandleCAS_NoAtomics(uintptr_t ProgramCounter, uint64_t *GPRs) { // ARMv8.0 CAS // [1] ldaxrb(TMP2.W(), MemOperand(MemSrc)) // [2] cmp (TMP2.W(), Expected.W()) // [3] b // [4] stlxrb(TMP3.W(), Desired.W(), MemOperand(MemSrc) // [5] cbnz // [6] mov // [7] b // [8] mov (.., TMP2.W()); // [9] clrex uint32_t *PC = (uint32_t*)ProgramCounter; uint32_t Instr = PC[0]; uint32_t Size = 1 << (Instr >> 30); uint32_t AddressReg = GetRnReg(Instr); uint32_t ResultReg = GetRdReg(Instr); //TMP2 uint32_t DesiredReg = 0; uint32_t ExpectedReg = 0; for (size_t i = 1; i < 6; ++i) { uint32_t NextInstr = PC[i]; if ((NextInstr & ArchHelpers::Arm64::STLXR_MASK) == ArchHelpers::Arm64::STLXR_INST) { #if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED // Just double check that the memory destination matches const uint32_t StoreAddressReg = GetRnReg(NextInstr); LOGMAN_THROW_A_FMT(StoreAddressReg == AddressReg, "StoreExclusive memory register didn't match the store exclusive register"); #endif DesiredReg = GetRdReg(NextInstr); } else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_INST || (NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_SHIFT_INST) { ExpectedReg = GetRmReg(NextInstr); } } //set up CASAL by doing mov(TMP2, Expected) GPRs[ResultReg] = GPRs[ExpectedReg]; if(RunCASAL(GPRs, Size, DesiredReg, ResultReg, AddressReg)) { return 7 * sizeof(uint32_t); //jump to mov to allocated register } else { return 0; } } static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_t *GPRs) { uint32_t *PC = (uint32_t*)ProgramCounter; uint32_t Instr = PC[0]; // Atomic Add // [1] ldaxrb(TMP2.W(), MemOperand(MemSrc)); // [2] add(TMP2.W(), TMP2.W(), GetReg(Op->Header.Args[1].ID())); // [3] stlxrb(TMP2.W(), TMP2.W(), MemOperand(MemSrc)); // [4] cbnz(TMP2.W(), &LoopTop); // // Atomic Fetch Add // [1] ldaxrb(TMP2.W(), MemOperand(MemSrc)); // [2] add(TMP3.W(), TMP2.W(), GetReg(Op->Header.Args[1].ID())); // [3] stlxrb(TMP4.W(), TMP3.W(), MemOperand(MemSrc)); // [4] cbnz(TMP4.W(), &LoopTop); // [5] mov(GetReg(Node), TMP2.W()); // // Atomic Swap // // [1] ldaxrb(TMP2.W(), MemOperand(MemSrc)); // [2] stlxrb(TMP4.W(), GetReg(Op->Header.Args[1].ID()), MemOperand(MemSrc)); // [3] cbnz(TMP4.W(), &LoopTop); // [4] uxtb(GetReg(Node), TMP2.W()); // // ASSUMPTIONS: // - Both cases: // - The [2]ALU op: (Non NEG case) // - First source is from [1]ldaxr // - Second source is incoming value // - The [2]ALU op: (NEG case) // - First source is zero register // - The second source is the from [1]ldaxr // - No ALU op: (SWAP case) // - No DataSourceRegister // // - In Atomic case (non-fetch) // - The [3]stlxr instruction status + memory register are the SAME register // // - In Atomic FETCH case // - The [3]stlxr instruction's status + memory register are never the same register // - The [5]mov instruction source is always the destination register from [1] ldaxr* uint32_t ResultReg = GetRdReg(Instr); uint32_t AddressReg = GetRnReg(Instr); uint64_t Addr = GPRs[AddressReg]; size_t NumInstructionsToSkip = 0; // Are we an Atomic op or AtomicFetch? bool AtomicFetch = false; // This is the register that is the incoming source to the ALU operation // = // NEG case is special // = Zero // DataSourceRegister must always be the Rm register uint32_t DataSourceReg {}; ExclusiveAtomicPairType AtomicOp {ExclusiveAtomicPairType::TYPE_SWAP}; // Scan forward at most five instructions to find our instructions for (size_t i = 1; i < 6; ++i) { uint32_t NextInstr = PC[i]; if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::ADD_INST || (NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::ADD_SHIFT_INST) { AtomicOp = ExclusiveAtomicPairType::TYPE_ADD; DataSourceReg = GetRmReg(NextInstr); } else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::SUB_INST || (NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::SUB_SHIFT_INST) { uint32_t RnReg = GetRnReg(NextInstr); if (RnReg == REGISTER_MASK) { // Zero reg means neg AtomicOp = ExclusiveAtomicPairType::TYPE_NEG; } else { AtomicOp = ExclusiveAtomicPairType::TYPE_SUB; } DataSourceReg = GetRmReg(NextInstr); } else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_INST || (NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::CMP_SHIFT_INST ) { return HandleCAS_NoAtomics(ProgramCounter, GPRs); //ARMv8.0 CAS } else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::AND_INST) { AtomicOp = ExclusiveAtomicPairType::TYPE_AND; DataSourceReg = GetRmReg(NextInstr); } else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::BIC_INST) { AtomicOp = ExclusiveAtomicPairType::TYPE_BIC; DataSourceReg = GetRmReg(NextInstr); } else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::OR_INST) { AtomicOp = ExclusiveAtomicPairType::TYPE_OR; DataSourceReg = GetRmReg(NextInstr); } else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::ORN_INST) { AtomicOp = ExclusiveAtomicPairType::TYPE_ORN; DataSourceReg = GetRmReg(NextInstr); } else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::EOR_INST) { AtomicOp = ExclusiveAtomicPairType::TYPE_EOR; DataSourceReg = GetRmReg(NextInstr); } else if ((NextInstr & ArchHelpers::Arm64::ALU_OP_MASK) == ArchHelpers::Arm64::EON_INST) { AtomicOp = ExclusiveAtomicPairType::TYPE_EON; DataSourceReg = GetRmReg(NextInstr); } else if ((NextInstr & ArchHelpers::Arm64::STLXR_MASK) == ArchHelpers::Arm64::STLXR_INST) { #if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED // Just double check that the memory destination matches const uint32_t StoreAddressReg = GetRnReg(NextInstr); LOGMAN_THROW_A_FMT(StoreAddressReg == AddressReg, "StoreExclusive memory register didn't match the store exclusive register"); #endif uint32_t StatusReg = GetRmReg(NextInstr); uint32_t StoreResultReg = GetRdReg(NextInstr); // We are an atomic fetch instruction if the data register isn't the status register AtomicFetch = !(StatusReg == StoreResultReg); if (AtomicOp == ExclusiveAtomicPairType::TYPE_SWAP) { // In the case of swap we don't have an ALU op inbetween // Source is directly in STLXR DataSourceReg = StoreResultReg; } } else if ((NextInstr & ArchHelpers::Arm64::CBNZ_MASK) == ArchHelpers::Arm64::CBNZ_INST) { // Found the CBNZ, we want to skip to just after this instruction when done NumInstructionsToSkip = i + 1; // This is the last instruction we care about. Leave now break; } else { LogMan::Msg::AFmt("Unknown instruction 0x{:08x}", NextInstr); } } uint32_t Size = 1 << (Instr >> 30); constexpr bool DoRetry = true; auto NOPExpected = [](AtomicType SrcVal, AtomicType) -> AtomicType { return SrcVal; }; auto ADDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType { return SrcVal + Desired; }; auto SUBDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType { return SrcVal - Desired; }; auto ANDDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType { return SrcVal & Desired; }; auto BICDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType { return SrcVal & ~Desired; }; auto ORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType { return SrcVal | Desired; }; auto ORNDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType { return SrcVal | ~Desired; }; auto EORDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType { return SrcVal ^ Desired; }; auto EONDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType { return SrcVal ^ ~Desired; }; auto NEGDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType { return -SrcVal; }; auto SWAPDesired = [](AtomicType SrcVal, AtomicType Desired) -> AtomicType { return Desired; }; if (Size == 2) { using AtomicType = uint16_t; CASDesiredFn DesiredFunction{}; switch (AtomicOp) { case ExclusiveAtomicPairType::TYPE_SWAP: DesiredFunction = SWAPDesired; break; case ExclusiveAtomicPairType::TYPE_ADD: DesiredFunction = ADDDesired; break; case ExclusiveAtomicPairType::TYPE_SUB: DesiredFunction = SUBDesired; break; case ExclusiveAtomicPairType::TYPE_AND: DesiredFunction = ANDDesired; break; case ExclusiveAtomicPairType::TYPE_BIC: DesiredFunction = BICDesired; break; case ExclusiveAtomicPairType::TYPE_OR: DesiredFunction = ORDesired; break; case ExclusiveAtomicPairType::TYPE_ORN: DesiredFunction = ORNDesired; break; case ExclusiveAtomicPairType::TYPE_EOR: DesiredFunction = EORDesired; break; case ExclusiveAtomicPairType::TYPE_EON: DesiredFunction = EONDesired; break; case ExclusiveAtomicPairType::TYPE_NEG: DesiredFunction = NEGDesired; break; default: LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", FEXCore::ToUnderlying(AtomicOp)); return false; } auto Res = DoCAS16( GPRs[DataSourceReg], 0, // Unused Addr, NOPExpected, DesiredFunction); if (AtomicFetch && ResultReg != 31) { // On atomic fetch then we store the resulting value back in to the loadacquire destination register // We want the memory value BEFORE the ALU op GPRs[ResultReg] = Res; } } else if (Size == 4) { using AtomicType = uint32_t; CASDesiredFn DesiredFunction{}; switch (AtomicOp) { case ExclusiveAtomicPairType::TYPE_SWAP: DesiredFunction = SWAPDesired; break; case ExclusiveAtomicPairType::TYPE_ADD: DesiredFunction = ADDDesired; break; case ExclusiveAtomicPairType::TYPE_SUB: DesiredFunction = SUBDesired; break; case ExclusiveAtomicPairType::TYPE_AND: DesiredFunction = ANDDesired; break; case ExclusiveAtomicPairType::TYPE_BIC: DesiredFunction = BICDesired; break; case ExclusiveAtomicPairType::TYPE_OR: DesiredFunction = ORDesired; break; case ExclusiveAtomicPairType::TYPE_ORN: DesiredFunction = ORNDesired; break; case ExclusiveAtomicPairType::TYPE_EOR: DesiredFunction = EORDesired; break; case ExclusiveAtomicPairType::TYPE_EON: DesiredFunction = EONDesired; break; case ExclusiveAtomicPairType::TYPE_NEG: DesiredFunction = NEGDesired; break; default: LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", FEXCore::ToUnderlying(AtomicOp)); return false; } auto Res = DoCAS32( GPRs[DataSourceReg], 0, // Unused Addr, NOPExpected, DesiredFunction); if (AtomicFetch && ResultReg != 31) { // On atomic fetch then we store the resulting value back in to the loadacquire destination register // We want the memory value BEFORE the ALU op GPRs[ResultReg] = Res; } } else if (Size == 8) { using AtomicType = uint64_t; CASDesiredFn DesiredFunction{}; switch (AtomicOp) { case ExclusiveAtomicPairType::TYPE_SWAP: DesiredFunction = SWAPDesired; break; case ExclusiveAtomicPairType::TYPE_ADD: DesiredFunction = ADDDesired; break; case ExclusiveAtomicPairType::TYPE_SUB: DesiredFunction = SUBDesired; break; case ExclusiveAtomicPairType::TYPE_AND: DesiredFunction = ANDDesired; break; case ExclusiveAtomicPairType::TYPE_BIC: DesiredFunction = BICDesired; break; case ExclusiveAtomicPairType::TYPE_OR: DesiredFunction = ORDesired; break; case ExclusiveAtomicPairType::TYPE_ORN: DesiredFunction = ORNDesired; break; case ExclusiveAtomicPairType::TYPE_EOR: DesiredFunction = EORDesired; break; case ExclusiveAtomicPairType::TYPE_EON: DesiredFunction = EONDesired; break; case ExclusiveAtomicPairType::TYPE_NEG: DesiredFunction = NEGDesired; break; default: LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}", FEXCore::ToUnderlying(AtomicOp)); return false; } auto Res = DoCAS64( GPRs[DataSourceReg], 0, // Unused Addr, NOPExpected, DesiredFunction); if (AtomicFetch && ResultReg != 31) { // On atomic fetch then we store the resulting value back in to the loadacquire destination register // We want the memory value BEFORE the ALU op GPRs[ResultReg] = Res; } } // Multiply by 4 for number of bytes to skip return NumInstructionsToSkip * 4; } [[nodiscard]] std::pair HandleUnalignedAccess(FEXCore::Core::InternalThreadState *Thread, bool ParanoidTSO, uintptr_t ProgramCounter, uint64_t *GPRs) { #ifdef _M_ARM_64 constexpr bool is_arm64 = true; #else constexpr bool is_arm64 = false; #endif constexpr auto NotHandled = std::make_pair(false, 0); if constexpr (is_arm64) { uint32_t *PC = (uint32_t*)ProgramCounter; uint32_t Instr = PC[0]; // 1 = 16bit // 2 = 32bit // 3 = 64bit uint32_t Size = (Instr & 0xC000'0000) >> 30; uint32_t AddrReg = (Instr >> 5) & 0x1F; uint32_t DataReg = Instr & 0x1F; // ParanoidTSO path doesn't modify any code. if (ParanoidTSO) [[unlikely]] { if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR* (Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR* if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0)) { // Skip this instruction now return std::make_pair(true, 4); } else { LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]); return NotHandled; } } else if ( (Instr & LDAXR_MASK) == STLR_INST) { // STLR* if (ArchHelpers::Arm64::HandleAtomicStore(Instr, GPRs, 0)) { // Skip this instruction now return std::make_pair(true, 4); } else { LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]); return NotHandled; } } else if ((Instr & RCPC2_MASK) == LDAPUR_INST) { // LDAPUR* // Extract the 9-bit offset from the instruction int32_t Offset = static_cast(Instr) << 11 >> 23; if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, Offset)) { // Skip this instruction now return std::make_pair(true, 4); } else { LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAPUR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]); return NotHandled; } } else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR* // Extract the 9-bit offset from the instruction int32_t Offset = static_cast(Instr) << 11 >> 23; if (ArchHelpers::Arm64::HandleAtomicStore(Instr, GPRs, Offset)) { // Skip this instruction now return std::make_pair(true, 4); } else { LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDLUR*: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]); return NotHandled; } } } const auto Frame = Thread->CurrentFrame; const uint64_t BlockBegin = Frame->State.InlineJITBlockHeader; auto InlineHeader = reinterpret_cast(BlockBegin); auto InlineTail = reinterpret_cast(Frame->State.InlineJITBlockHeader + InlineHeader->OffsetToBlockTail); // Lock code mutex during any SIGBUS handling that potentially changes code. // Need to be careful to not read any code part-way through modification. FEXCore::Utils::SpinWaitLock::UniqueSpinMutex lk(&InlineTail->SpinLockFutex); if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR* (Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR* uint32_t LDR = 0b0011'1000'0111'1111'0110'1000'0000'0000; LDR |= Size << 30; LDR |= AddrReg << 5; LDR |= DataReg; PC[0] = LDR; PC[1] = DMB_LD; // Back-patch the half-barrier. ClearICache(&PC[-1], 16); // With the instruction modified, now execute again. return std::make_pair(true, 0); } else if ( (Instr & LDAXR_MASK) == STLR_INST) { // STLR* uint32_t STR = 0b0011'1000'0011'1111'0110'1000'0000'0000; STR |= Size << 30; STR |= AddrReg << 5; STR |= DataReg; PC[-1] = DMB; // Back-patch the half-barrier. PC[0] = STR; ClearICache(&PC[-1], 16); // Back up one instruction and have another go return std::make_pair(true, -4); } else if ((Instr & RCPC2_MASK) == LDAPUR_INST) { // LDAPUR* // Extract the 9-bit offset from the instruction uint32_t LDUR = 0b0011'1000'0100'0000'0000'0000'0000'0000; LDUR |= Size << 30; LDUR |= AddrReg << 5; LDUR |= DataReg; LDUR |= Instr & (0b1'1111'1111 << 9); PC[0] = LDUR; PC[1] = DMB_LD; // Back-patch the half-barrier. ClearICache(&PC[-1], 16); // With the instruction modified, now execute again. return std::make_pair(true, 0); } else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR* uint32_t STUR = 0b0011'1000'0000'0000'0000'0000'0000'0000; STUR |= Size << 30; STUR |= AddrReg << 5; STUR |= DataReg; STUR |= Instr & (0b1'1111'1111 << 9); PC[-1] = DMB; // Back-patch the half-barrier. PC[0] = STUR; ClearICache(&PC[-1], 16); // Back up one instruction and have another go return std::make_pair(true, -4); } else if ((Instr & ArchHelpers::Arm64::LDAXP_MASK) == ArchHelpers::Arm64::LDAXP_INST) { // LDAXP //Should be compare and swap pair only. LDAXP not used elsewhere uint64_t BytesToSkip = ArchHelpers::Arm64::HandleCASPAL_ARMv8(Instr, ProgramCounter, GPRs); if (BytesToSkip) { // Skip this instruction now return std::make_pair(true, BytesToSkip); } else { if (ArchHelpers::Arm64::HandleAtomicVectorStore(Instr, ProgramCounter)) { return std::make_pair(true, 0); } else { LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXP: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]); return NotHandled; } } } else if ((Instr & ArchHelpers::Arm64::STLXP_MASK) == ArchHelpers::Arm64::STLXP_INST) { // STLXP //Should not trigger - middle of an LDAXP/STAXP pair. LogMan::Msg::EFmt("Unhandled JIT SIGBUS STLXP: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]); return NotHandled; } else if ((Instr & ArchHelpers::Arm64::CASPAL_MASK) == ArchHelpers::Arm64::CASPAL_INST) { // CASPAL if (ArchHelpers::Arm64::HandleCASPAL(Instr, GPRs)) { // Skip this instruction now return std::make_pair(true, 4); } else { LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASPAL: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]); return NotHandled; } } else if ((Instr & ArchHelpers::Arm64::CASAL_MASK) == ArchHelpers::Arm64::CASAL_INST) { // CASAL if (ArchHelpers::Arm64::HandleCASAL(GPRs, Instr)) { // Skip this instruction now return std::make_pair(true, 4); } else { LogMan::Msg::EFmt("Unhandled JIT SIGBUS CASAL: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]); return NotHandled; } } else if ((Instr & ArchHelpers::Arm64::ATOMIC_MEM_MASK) == ArchHelpers::Arm64::ATOMIC_MEM_INST) { // Atomic memory op if (ArchHelpers::Arm64::HandleAtomicMemOp(Instr, GPRs)) { // Skip this instruction now return std::make_pair(true, 4); } else { uint8_t Op = (PC[0] >> 12) & 0xF; LogMan::Msg::EFmt("Unhandled JIT SIGBUS Atomic mem op 0x{:02x}: PC: 0x{:x} Instruction: 0x{:08x}\n", Op, ProgramCounter, PC[0]); return NotHandled; } } else if ((Instr & ArchHelpers::Arm64::LDAXR_MASK) == ArchHelpers::Arm64::LDAXR_INST) { // LDAXR* uint64_t BytesToSkip = ArchHelpers::Arm64::HandleAtomicLoadstoreExclusive(ProgramCounter, GPRs); if (BytesToSkip) { // Skip this instruction now return std::make_pair(true, BytesToSkip); } else { LogMan::Msg::EFmt("Unhandled JIT SIGBUS LDAXR: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]); return NotHandled; } } else { LogMan::Msg::EFmt("Unhandled JIT SIGBUS: PC: 0x{:x} Instruction: 0x{:08x}\n", ProgramCounter, PC[0]); return NotHandled; } } return NotHandled; } }