Arm64: Adds another TSO hack to disable half-barrier TSO

A feature of FEX's JIT is that when an unaligned atomic load/store
operation occurs, the instructions will be backpatched in to a barrier
plus a non-atomic memory instruction. This is the half-barrier technique
that still ensures correct visibility of loadstores in an unaligned
context.

The problem with this approach is that the dmb instructions are HEAVY,
because they effectively stop the world until all memory operations in
flight are visible. But it is a necessary evil since unaligned atomics
aren't a thing on ARM processors. FEAT_LSE only gives you unaligned
atomics inside of a 16-byte granularity, which doesn't match x86
behaviour of cacheline size (effectively always 64B).

This adds a new TSO option to disable the half-barrier on unaligned
atomic and instead only convert it to a regular loadstore instruction,
ommiting the half-barrier. This gives more insight in to how well a
CPU's LRCPC implementation is by not stalling on DMB instructions when
possible.

Originally implemented as a test to see if this makes Sonic Adventure 2
run full speed with TSO enabled (but all available TSO options disabled)
on NVIDIA Orin. Unfortunately this basically makes the code no longer
stall on dmb instructions and instead just showing how bad the LRCPC
implementation is, since the stalls show up on `ldapur` instructions
instead.

Tested Sonic Adventure 2 on X13s and it ran at 60FPS there without the
hack anyway.
This commit is contained in:
Ryan Houdek committed 2024-04-24 13:09:00 -07:00
1 parent 81a4206805
commit 6463054fa3
8 files changed
+92 -14

No files matched your search

@@ -414,6 +414,14 @@
"Only affects REP MOVS and REP STOS instructions"
]
},
"HalfBarrierTSOEnabled": {
"Type": "bool",
"Default": "true",
"Desc": [
"When TSO emulation is enabled, controls if unaligned loads and stores should be backpatched to half-barrier atomics.",
"Can be dangerous due to aligned loadstores through the same code now become non-atomic."
]
},
"TSOAutoMigration": {
"Type": "bool",
"Default": "true",
+16 -8
View File
@@ -1822,7 +1822,7 @@ static uint64_t HandleAtomicLoadstoreExclusive(uintptr_t ProgramCounter, uint64_
[[nodiscard]]
std::pair<bool, int32_t>
HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, bool ParanoidTSO, uintptr_t ProgramCounter, uint64_t* GPRs) {
HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandlerType HandleType, uintptr_t ProgramCounter, uint64_t* GPRs) {
#ifdef _M_ARM_64
constexpr bool is_arm64 = true;
#else
@@ -1843,7 +1843,7 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, bool ParanoidT
uint32_t DataReg = Instr & 0x1F;
// ParanoidTSO path doesn't modify any code.
if (ParanoidTSO) [[unlikely]] {
if (HandleType == UnalignedHandlerType::Paranoid) [[unlikely]] {
if ((Instr & LDAXR_MASK) == LDAR_INST || // LDAR*
(Instr & LDAXR_MASK) == LDAPR_INST) { // LDAPR*
if (ArchHelpers::Arm64::HandleAtomicLoad(Instr, GPRs, 0)) {
@@ -1900,8 +1900,10 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, bool ParanoidT
LDR |= AddrReg << 5;
LDR |= DataReg;
PC[0] = LDR;
PC[1] = DMB_LD; // Back-patch the half-barrier.
ClearICache(&PC[-1], 16);
if (HandleType != UnalignedHandlerType::NonAtomic) {
PC[1] = DMB_LD; // Back-patch the half-barrier.
}
ClearICache(&PC[0], 16);
// With the instruction modified, now execute again.
return std::make_pair(true, 0);
} else if ((Instr & LDAXR_MASK) == STLR_INST) { // STLR*
@@ -1909,7 +1911,9 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, bool ParanoidT
STR |= Size << 30;
STR |= AddrReg << 5;
STR |= DataReg;
PC[-1] = DMB; // Back-patch the half-barrier.
if (HandleType != UnalignedHandlerType::NonAtomic) {
PC[-1] = DMB; // Back-patch the half-barrier.
}
PC[0] = STR;
ClearICache(&PC[-1], 16);
// Back up one instruction and have another go
@@ -1922,8 +1926,10 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, bool ParanoidT
LDUR |= DataReg;
LDUR |= Instr & (0b1'1111'1111 << 9);
PC[0] = LDUR;
PC[1] = DMB_LD; // Back-patch the half-barrier.
ClearICache(&PC[-1], 16);
if (HandleType != UnalignedHandlerType::NonAtomic) {
PC[1] = DMB_LD; // Back-patch the half-barrier.
}
ClearICache(&PC[0], 16);
// With the instruction modified, now execute again.
return std::make_pair(true, 0);
} else if ((Instr & RCPC2_MASK) == STLUR_INST) { // STLUR*
@@ -1932,7 +1938,9 @@ HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, bool ParanoidT
STUR |= AddrReg << 5;
STUR |= DataReg;
STUR |= Instr & (0b1'1111'1111 << 9);
PC[-1] = DMB; // Back-patch the half-barrier.
if (HandleType != UnalignedHandlerType::NonAtomic) {
PC[-1] = DMB; // Back-patch the half-barrier.
}
PC[0] = STUR;
ClearICache(&PC[-1], 16);
// Back up one instruction and have another go
@@ -24,7 +24,7 @@ bool HandleAtomicMemOp(void* _ucontext, void* _info, uint32_t Instr) {
}
std::pair<bool, int32_t>
HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, bool ParanoidTSO, uintptr_t ProgramCounter, uint64_t* GPRs) {
HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandlerType HandleType, uintptr_t ProgramCounter, uint64_t* GPRs) {
ERROR_AND_DIE_FMT("HandleAtomicMemOp Not Implemented");
}
@@ -108,6 +108,15 @@ inline uint32_t GetRmReg(uint32_t Instr) {
return (Instr >> RM_OFFSET) & REGISTER_MASK;
}
enum class UnalignedHandlerType {
///< Don't backpatch code, instead handle inside SIGBUS handler.
Paranoid,
///< Backpatch unaligned access to half-barrier based atomic.
HalfBarrier,
///< Backpatch unaligned access to non-atomic.
NonAtomic,
};
/**
* @brief On ARM64 handles an unaligned memory access that the JIT has done.
*
@@ -123,5 +132,5 @@ inline uint32_t GetRmReg(uint32_t Instr) {
*/
[[nodiscard]]
FEX_DEFAULT_VISIBILITY std::pair<bool, int32_t>
HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, bool ParanoidTSO, uintptr_t ProgramCounter, uint64_t* GPRs);
HandleUnalignedAccess(FEXCore::Core::InternalThreadState* Thread, UnalignedHandlerType HandleType, uintptr_t ProgramCounter, uint64_t* GPRs);
} // namespace FEXCore::ArchHelpers::Arm64
+12
View File
@@ -556,10 +556,12 @@ void FillHackConfig() {
auto Value = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED);
auto VectorTSO = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_VECTORTSOENABLED);
auto MemcpyTSO = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_MEMCPYSETTSOENABLED);
auto HalfBarrierTSO = LoadedConfig->Get(FEXCore::Config::ConfigOption::CONFIG_HALFBARRIERTSOENABLED);
bool TSOEnabled = Value.has_value() && **Value == "1";
bool VectorTSOEnabled = VectorTSO.has_value() && **VectorTSO == "1";
bool MemcpyTSOEnabled = MemcpyTSO.has_value() && **MemcpyTSO == "1";
bool HalfBarrierTSOEnabled = HalfBarrierTSO.has_value() && **HalfBarrierTSO == "1";
if (ImGui::Checkbox("TSO Enabled", &TSOEnabled)) {
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_TSOENABLED, TSOEnabled ? "1" : "0");
@@ -588,6 +590,16 @@ void FillHackConfig() {
ImGui::EndTooltip();
}
if (ImGui::Checkbox("Unaligned Half-Barrier TSO Enabled", &HalfBarrierTSOEnabled)) {
LoadedConfig->EraseSet(FEXCore::Config::ConfigOption::CONFIG_HALFBARRIERTSOENABLED, HalfBarrierTSOEnabled ? "1" : "0");
ConfigChanged = true;
}
if (ImGui::IsItemHovered()) {
ImGui::BeginTooltip();
ImGui::Text("Disables half-barrier TSO emulation on unaligned load/store instructions");
ImGui::EndTooltip();
}
ImGui::TreePop();
}
}
@@ -1627,6 +1627,14 @@ SignalDelegator::SignalDelegator(FEXCore::Context::Context* _CTX, const std::str
HostHandlers[SIGKILL].Installed = true;
HostHandlers[SIGSTOP].Installed = true;
if (ParanoidTSO()) {
UnalignedHandlerType = FEXCore::ArchHelpers::Arm64::UnalignedHandlerType::Paranoid;
} else if (HalfBarrierTSOEnabled()) {
UnalignedHandlerType = FEXCore::ArchHelpers::Arm64::UnalignedHandlerType::HalfBarrier;
} else {
UnalignedHandlerType = FEXCore::ArchHelpers::Arm64::UnalignedHandlerType::NonAtomic;
}
// Most signals default to termination
// These ones are slightly different
static constexpr std::array<std::pair<int, SignalDelegator::DefaultBehaviour>, 14> SignalDefaultBehaviours = {{
@@ -1703,7 +1711,7 @@ SignalDelegator::SignalDelegator(FEXCore::Context::Context* _CTX, const std::str
return false;
}
const auto Result = FEXCore::ArchHelpers::Arm64::HandleUnalignedAccess(Thread, GlobalDelegator->ParanoidTSO(), PC,
const auto Result = FEXCore::ArchHelpers::Arm64::HandleUnalignedAccess(Thread, GlobalDelegator->GetUnalignedHandlerType(), PC,
ArchHelpers::Context::GetArmGPRs(ucontext));
ArchHelpers::Context::SetPc(ucontext, PC + Result.second);
return Result.first;
@@ -23,6 +23,7 @@ $end_info$
#include <mutex>
#include <FEXCore/Core/SignalDelegator.h>
#include <FEXCore/Utils/ArchHelpers/Arm64.h>
#include <FEXCore/Utils/Telemetry.h>
namespace FEXCore {
@@ -127,7 +128,9 @@ public:
void SignalThread(FEXCore::Core::InternalThreadState* Thread, FEXCore::Core::SignalEvent Event) override;
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEXCore::ArchHelpers::Arm64::UnalignedHandlerType GetUnalignedHandlerType() const {
return UnalignedHandlerType;
}
void SaveTelemetry();
private:
@@ -156,6 +159,10 @@ private:
fextl::string const ApplicationName;
FEXCORE_TELEMETRY_INIT(CrashMask, TYPE_CRASH_MASK);
FEXCORE_TELEMETRY_INIT(UnhandledNonCanonical, TYPE_UNHANDLED_NONCANONICAL_ADDRESS);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
FEXCore::ArchHelpers::Arm64::UnalignedHandlerType UnalignedHandlerType {FEXCore::ArchHelpers::Arm64::UnalignedHandlerType::HalfBarrier};
enum DefaultBehaviour {
DEFAULT_TERM,
+28 -2
View File
@@ -263,14 +263,39 @@ WOW64_CONTEXT ReconstructWowContext(CONTEXT* Context) {
return WowContext;
}
class TSOHandlerConfig final {
public:
TSOHandlerConfig() {
if (ParanoidTSO()) {
UnalignedHandlerType = FEXCore::ArchHelpers::Arm64::UnalignedHandlerType::Paranoid;
} else if (HalfBarrierTSOEnabled()) {
UnalignedHandlerType = FEXCore::ArchHelpers::Arm64::UnalignedHandlerType::HalfBarrier;
} else {
UnalignedHandlerType = FEXCore::ArchHelpers::Arm64::UnalignedHandlerType::NonAtomic;
}
}
FEXCore::ArchHelpers::Arm64::UnalignedHandlerType GetUnalignedHandlerType() const {
return UnalignedHandlerType;
}
private:
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(HalfBarrierTSOEnabled, HALFBARRIERTSOENABLED);
FEXCore::ArchHelpers::Arm64::UnalignedHandlerType UnalignedHandlerType {FEXCore::ArchHelpers::Arm64::UnalignedHandlerType::HalfBarrier};
};
static std::optional<TSOHandlerConfig> HandlerConfig;
bool HandleUnalignedAccess(CONTEXT* Context) {
auto Thread = GetTLS().ThreadState();
if (!Thread->CTX->IsAddressInCodeBuffer(Thread, Context->Pc)) {
return false;
}
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
const auto Result = FEXCore::ArchHelpers::Arm64::HandleUnalignedAccess(Thread, ParanoidTSO(), Context->Pc, &Context->X0);
const auto Result =
FEXCore::ArchHelpers::Arm64::HandleUnalignedAccess(Thread, HandlerConfig->GetUnalignedHandlerType(), Context->Pc, &Context->X0);
if (!Result.first) {
return false;
}
@@ -418,6 +443,7 @@ void BTCpuProcessInit() {
SignalDelegator = fextl::make_unique<FEX::DummyHandlers::DummySignalDelegator>();
SyscallHandler = fextl::make_unique<WowSyscallHandler>();
Context::HandlerConfig.emplace();
CTX = FEXCore::Context::Context::CreateNewContext();
CTX->SetSignalDelegator(SignalDelegator.get());