From c91c6efc87039914806db06b335e347035fcc468 Mon Sep 17 00:00:00 2001 From: Justin Becker Date: Wed, 16 Sep 2026 13:05:38 -0700 Subject: [PATCH] Software fallback for RDRAND/RDSEED --- .../Source/Interface/Config/Config.json.in | 8 +++ FEXCore/Source/Interface/Context/Context.h | 6 ++ FEXCore/Source/Interface/Core/CPUID.cpp | 70 ++++++++++--------- FEXCore/Source/Interface/Core/Core.cpp | 8 +++ FEXCore/Source/Interface/Core/JIT/JIT.cpp | 36 ++++++++++ FEXCore/Source/Interface/Core/JIT/MiscOps.cpp | 34 ++++++++- .../Interface/Core/OpcodeDispatcher.cpp | 2 +- FEXCore/Source/Interface/IR/IR.json | 3 +- FEXCore/include/FEXCore/Core/CoreState.h | 1 + FEXHeaderUtils/FEXHeaderUtils/Syscalls.h | 4 ++ unittests/FEXLinuxTests/CMakeLists.txt | 5 ++ unittests/FEXLinuxTests/tests/cpu/rdrand.cpp | 37 ++++++++++ 12 files changed, 178 insertions(+), 36 deletions(-) create mode 100644 unittests/FEXLinuxTests/tests/cpu/rdrand.cpp diff --git a/FEXCore/Source/Interface/Config/Config.json.in b/FEXCore/Source/Interface/Config/Config.json.in index 030c1e6d7..062f615f8 100644 --- a/FEXCore/Source/Interface/Config/Config.json.in +++ b/FEXCore/Source/Interface/Config/Config.json.in @@ -141,6 +141,14 @@ "Hides hybrid CPU core arrangement." ] }, + "SoftwareRNG": { + "Type": "bool", + "Default": "true", + "AffectsCodeGen": "true", + "Desc": [ + "Emulates RDRAND and RDSEED in software when the host does not implement FEAT_RNG." + ] + }, "CPUFeatureRegisters": { "Type": "str", "Default": "", diff --git a/FEXCore/Source/Interface/Context/Context.h b/FEXCore/Source/Interface/Context/Context.h index 1a7d2fd9e..ebceec145 100644 --- a/FEXCore/Source/Interface/Context/Context.h +++ b/FEXCore/Source/Interface/Context/Context.h @@ -369,10 +369,16 @@ public: FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY); FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS); FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE); + FEX_CONFIG_OPT(SoftwareRNG, SOFTWARERNG); FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS); FEX_CONFIG_OPT(MonoHacks, MONOHACKS); } Config; + bool SoftwareRNGEnabled() const { + return Config.SoftwareRNG() && HostRNGAvailable; + } + bool HostRNGAvailable {}; + FEXCore::Utils::WritePriorityMutex::Mutex CodeInvalidationMutex {}; uint32_t StrictSplitLockMutex {}; diff --git a/FEXCore/Source/Interface/Core/CPUID.cpp b/FEXCore/Source/Interface/Core/CPUID.cpp index fa8cd8ed1..ddce98d66 100644 --- a/FEXCore/Source/Interface/Core/CPUID.cpp +++ b/FEXCore/Source/Interface/Core/CPUID.cpp @@ -458,6 +458,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const { (Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU (GetCPUID() << 24); // Local APIC ID + const uint32_t SupportsRAND = CTX->HostFeatures.SupportsRAND || CTX->SoftwareRNGEnabled(); + Res.ecx = (1 << 0) | // SSE3 (CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ (1 << 2) | // DS area supports 64bit layout @@ -488,7 +490,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const { (SupportsAVX() << 27) | // OSXSAVE (SupportsAVX() << 28) | // AVX (SupportsAVX() << 29) | // F16C - (CTX->HostFeatures.SupportsRAND << 30) | // RDRAND + (SupportsRAND << 30) | // RDRAND (Hypervisor << 31); Res.edx = (1 << 0) | // FPU @@ -670,42 +672,44 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const { const uint32_t SupportsVPCLMULQDQ = CTX->HostFeatures.SupportsPMULL_128Bit && SupportsAVX(); const uint32_t SupportsWFXT = CTX->HostFeatures.SupportsWFXT; + const uint32_t SupportsRAND = CTX->HostFeatures.SupportsRAND || CTX->SoftwareRNGEnabled(); + // Number of subfunctions // TODO: For now, subfunction 1 only exposes AVX-VNNI so we make it conditional // on AVX-VNNI support. We should revisit this if/when we add more to this leaf. Res.eax = SupportsAVXVNNI; - Res.ebx = (1 << 0) | // FS/GS support - (0 << 1) | // TSC adjust MSR - (0 << 2) | // SGX - (SupportsAVX() << 3) | // BMI1 - (0 << 4) | // Intel Hardware Lock Elison - (SupportsAVX() << 5) | // AVX2 support - (1 << 6) | // FPU data pointer updated only on exception - (1 << 7) | // SMEP support - (SupportsAVX() << 8) | // BMI2 - (SupportsEnhancedREPMOVS << 9) | // Enhanced REP MOVSB/STOSB - (1 << 10) | // INVPCID for system software control of process-context - (0 << 11) | // Restricted transactional memory - (0 << 12) | // Intel resource directory technology Monitoring - (1 << 13) | // Deprecates FPU CS and DS - (0 << 14) | // Intel MPX - (0 << 15) | // Intel Resource Directory Technology Allocation - (0 << 16) | // AVX512-F - (0 << 17) | // AVX512-DQ - (CTX->HostFeatures.SupportsRAND << 18) | // RDSEED - (1 << 19) | // ADCX and ADOX instructions - (0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions - (0 << 21) | // AVX512-IFMA - (0 << 22) | // PCOMMIT (deprecated?) - (1 << 23) | // CLFLUSHOPT instruction - (1 << 24) | // CLWB instruction - (0 << 25) | // Intel processor trace - (0 << 26) | // AVX512-PF - (0 << 27) | // AVX512-ER - (0 << 28) | // AVX512-CD - (Features.SHA << 29) | // SHA instructions - (0 << 30) | // AVX512-BW - (0 << 31); // AVX512-VL + Res.ebx = (1 << 0) | // FS/GS support + (0 << 1) | // TSC adjust MSR + (0 << 2) | // SGX + (SupportsAVX() << 3) | // BMI1 + (0 << 4) | // Intel Hardware Lock Elison + (SupportsAVX() << 5) | // AVX2 support + (1 << 6) | // FPU data pointer updated only on exception + (1 << 7) | // SMEP support + (SupportsAVX() << 8) | // BMI2 + (SupportsEnhancedREPMOVS << 9) | // Enhanced REP MOVSB/STOSB + (1 << 10) | // INVPCID for system software control of process-context + (0 << 11) | // Restricted transactional memory + (0 << 12) | // Intel resource directory technology Monitoring + (1 << 13) | // Deprecates FPU CS and DS + (0 << 14) | // Intel MPX + (0 << 15) | // Intel Resource Directory Technology Allocation + (0 << 16) | // AVX512-F + (0 << 17) | // AVX512-DQ + (SupportsRAND << 18) | // RDSEED + (1 << 19) | // ADCX and ADOX instructions + (0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions + (0 << 21) | // AVX512-IFMA + (0 << 22) | // PCOMMIT (deprecated?) + (1 << 23) | // CLFLUSHOPT instruction + (1 << 24) | // CLWB instruction + (0 << 25) | // Intel processor trace + (0 << 26) | // AVX512-PF + (0 << 27) | // AVX512-ER + (0 << 28) | // AVX512-CD + (Features.SHA << 29) | // SHA instructions + (0 << 30) | // AVX512-BW + (0 << 31); // AVX512-VL Res.ecx = (1 << 0) | // PREFETCHWT1 (0 << 1) | // AVX512VBMI diff --git a/FEXCore/Source/Interface/Core/Core.cpp b/FEXCore/Source/Interface/Core/Core.cpp index c250cd3c4..e7047b8b6 100644 --- a/FEXCore/Source/Interface/Core/Core.cpp +++ b/FEXCore/Source/Interface/Core/Core.cpp @@ -105,6 +105,14 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features) // Track atomic TSO emulation configuration. UpdateAtomicTSOEmulationConfig(); +#ifndef _WIN32 + // Check if the kernel supports getrandom(). + uint64_t Probe {}; + HostRNGAvailable = FHU::Syscalls::getrandom(&Probe, sizeof(Probe), 0) == sizeof(Probe); +#else + HostRNGAvailable = true; +#endif + DiskCache.Init(this); } diff --git a/FEXCore/Source/Interface/Core/JIT/JIT.cpp b/FEXCore/Source/Interface/Core/JIT/JIT.cpp index 95fd79b45..2a29153a9 100644 --- a/FEXCore/Source/Interface/Core/JIT/JIT.cpp +++ b/FEXCore/Source/Interface/Core/JIT/JIT.cpp @@ -36,7 +36,14 @@ $end_info$ #include #include +#include +#include #include +#ifdef _WIN32 +#include +#else +#include +#endif namespace { struct DivRem { @@ -63,6 +70,34 @@ LDIV(uint64_t SrcHigh, uint64_t SrcLow, int64_t Divisor) { }; } +#ifndef _WIN32 + +static std::optional RDRANDFallback(uint64_t Reseed) { + uint64_t Value {}; + FHU::Syscalls::getrandom(&Value, sizeof(Value), 0); + return Value; +} + +#else + +// Windows does not have an equivalent to getrandom() without dynamically linking +// bcrypt et al. Since this fallback is just for compat it does not need to be +// cryptographic, so instead we vendor SplitMix64 as a naive fallback. + +// Reference implementation by Sebastiano Vigna, public domain (CC0) +// https://prng.di.unimi.it/splitmix64.c +static std::atomic + RNGState {static_cast(__builtin_readcyclecounter())}; + +static std::optional RDRANDFallback(uint64_t Reseed) { + const uint64_t State = RNGState.load(std::memory_order_relaxed) + 0x9E3779B97F4A7C15ULL; + RNGState.store(State, std::memory_order_relaxed); + uint64_t Value = (State ^ (State >> 30)) * 0xBF58476D1CE4E5B9ULL; + Value = (Value ^ (Value >> 27)) * 0x94D049BB133111EBULL; + return Value ^ (Value >> 31); +} +#endif + static void PrintValue(uint64_t Value) { LogMan::Msg::DFmt("Value: 0x{:x}", Value); @@ -664,6 +699,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In Ptrs.ExitFunctionLink = reinterpret_cast(&Arm64JITCore::ExitFunctionLink); Ptrs.LUDIV = reinterpret_cast(LUDIV); Ptrs.LDIV = reinterpret_cast(LDIV); + Ptrs.RDRANDFallback = reinterpret_cast(RDRANDFallback); } CurrentCodeBuffer = SharedCodeBuffers.GetLatest(); diff --git a/FEXCore/Source/Interface/Core/JIT/MiscOps.cpp b/FEXCore/Source/Interface/Core/JIT/MiscOps.cpp index 6c2b3499a..6e8df9553 100644 --- a/FEXCore/Source/Interface/Core/JIT/MiscOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/MiscOps.cpp @@ -309,7 +309,39 @@ DEF_OP(ProcessorID) { DEF_OP(RDRAND) { auto Op = IROp->C(); - mrs(GetReg(Node), Op->GetReseeded ? ARMEmitter::SystemRegister::RNDRRS : ARMEmitter::SystemRegister::RNDR); + if (CTX->HostFeatures.SupportsRAND) { + mrs(GetReg(Node), Op->GetReseeded ? ARMEmitter::SystemRegister::RNDRRS : ARMEmitter::SystemRegister::RNDR); + return; + } + + // Software fallback, call the host RNG generator. + PushDynamicRegs(TMP4); + SpillStaticRegs(TMP4); + + // x0 = Reseed + // x1 = Generator + LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, Op->GetReseeded ? 1 : 0); + ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.RDRANDFallback)); + + if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] { + GenerateIndirectRuntimeCall<__uint128_t, uint64_t>(ARMEmitter::Reg::r1); + } else { + blr(ARMEmitter::Reg::r1); + } + + if (!TMP_ABIARGS) { + mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0); + mov(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Reg::r1); + } + + FillStaticRegs(); + PopDynamicRegs(); + + // Results are in x0, x1 + // std::optional: value in x0, engaged flag in the low byte of x1. Match the hardware behaviour of setting Z when + // no number was produced. + mov(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1); + tst(ARMEmitter::Size::i64Bit, TMP2, 0xFF); } DEF_OP(Yield) { diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index bb5e04630..5ee618981 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -5125,7 +5125,7 @@ void OpDispatchBuilder::CRC32(OpcodeArgs) { } void OpDispatchBuilder::RDRANDOp(OpcodeArgs, bool Reseed) { - if (!CTX->HostFeatures.SupportsRAND) { + if (!CTX->HostFeatures.SupportsRAND && !CTX->SoftwareRNGEnabled()) { UnimplementedOp(Op); return; } diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index d201e63fb..2d850ddbb 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -275,7 +275,8 @@ "The boolean argument asks if we should be reading the reseeded number or not", "Reseeded RNG calculation is more expensive and will be heavier to use", "Returns the 64-bit number", - "Sets the Z flag if the number is valid.", + "Falls back to a host RNG call when the hardware doesn't support it", + "Sets the Z flag if the number is invalid.", "RNG hardware is allowed to fail early and return. Software must always check this" ], "HasSideEffects": true, diff --git a/FEXCore/include/FEXCore/Core/CoreState.h b/FEXCore/include/FEXCore/Core/CoreState.h index 3bdffa7ba..0ac202eed 100644 --- a/FEXCore/include/FEXCore/Core/CoreState.h +++ b/FEXCore/include/FEXCore/Core/CoreState.h @@ -350,6 +350,7 @@ struct JITPointers { uint64_t MonoBackpatcherWrite {}; uint64_t LUDIV {}; uint64_t LDIV {}; + uint64_t RDRANDFallback {}; uint64_t ThunkCallbackRet {}; // Handles returning/calling ARM64EC code from the JIT, expects the target PC in TMP3 diff --git a/FEXHeaderUtils/FEXHeaderUtils/Syscalls.h b/FEXHeaderUtils/FEXHeaderUtils/Syscalls.h index e5c2824f4..dae3d3bcc 100644 --- a/FEXHeaderUtils/FEXHeaderUtils/Syscalls.h +++ b/FEXHeaderUtils/FEXHeaderUtils/Syscalls.h @@ -96,6 +96,10 @@ inline int32_t renameat2(int olddirfd, const char* oldpath, int newdirfd, const inline int32_t pidfd_open(pid_t pid, unsigned int flags) { return ::syscall(SYS_pidfd_open, pid, flags); } + +inline ssize_t getrandom(void* buf, size_t buflen, unsigned int flags) { + return ::syscall(SYS_getrandom, buf, buflen, flags); +} #else inline int32_t getcpu(uint32_t* cpu, uint32_t* node) { diff --git a/unittests/FEXLinuxTests/CMakeLists.txt b/unittests/FEXLinuxTests/CMakeLists.txt index 39f1b3998..5851004d8 100644 --- a/unittests/FEXLinuxTests/CMakeLists.txt +++ b/unittests/FEXLinuxTests/CMakeLists.txt @@ -63,6 +63,11 @@ function(AddTests Tests BinDirectory Bitness) set_property(TEST "${TEST_CASE}.jit.flt" APPEND PROPERTY ENVIRONMENT "FEX_THUNKCONFIG=${CMAKE_SOURCE_DIR}/Data/CI/FEXLinuxTestsThunks.json") endif() + if(TEST_NAME STREQUAL "rdrand") + # Force use of the software falllback to excercise that path + set_property(TEST "${TEST_CASE}.jit.flt" APPEND PROPERTY ENVIRONMENT "FEX_HOSTFEATURES=disablerng") + endif() + if(TEST_NAME STREQUAL "vulkan_procaddr") set_property(TEST "${TEST_CASE}.jit.flt" APPEND PROPERTY ENVIRONMENT "FEX_THUNKCONFIG=${CMAKE_SOURCE_DIR}/Data/CI/VulkanThunks.json;FEX_THUNKHOSTLIBS=${HOSTLIBS_DATA_DIRECTORY}/HostThunks;FEX_THUNKGUESTLIBS=${CMAKE_INSTALL_PREFIX}/share/fex-emu/GuestThunks") diff --git a/unittests/FEXLinuxTests/tests/cpu/rdrand.cpp b/unittests/FEXLinuxTests/tests/cpu/rdrand.cpp new file mode 100644 index 000000000..fb08b4f9d --- /dev/null +++ b/unittests/FEXLinuxTests/tests/cpu/rdrand.cpp @@ -0,0 +1,37 @@ +#include + +#include + +// Retries until the instruction reports success. +// clang-format off +#define RNG(insn, Value) \ + __asm volatile(".Lretry%=:\n" \ + insn " %0\n" \ + "jnc .Lretry%=\n" \ + : "=r"(Value) \ + : \ + : "cc") +// clang-format on + +// This test is designed to test the software fallback implementation for RDRAND and RDSEED. +// To that end, it is executed with FEX_HOSTFEATURES=disablerng. + +TEST_CASE("rdrand") { + uint32_t eax, ebx, ecx, edx; + __asm volatile("cpuid" : "=a"(eax), "=b"(ebx), "=c"(ecx), "=d"(edx) : "a"(1), "c"(0)); + CHECK((ecx >> 30) & 1); // RDRAND + __asm volatile("cpuid" : "=a"(eax), "=b"(ebx), "=c"(ecx), "=d"(edx) : "a"(7), "c"(0)); + CHECK((ebx >> 18) & 1); // RDSEED + + uintptr_t A, B; + RNG("rdrand", A); + RNG("rdrand", B); + + // This assert is not technically correct, + // but at a 1/(2^32) chance of failure, I like our odds. + CHECK(A != B); + + RNG("rdseed", A); + RNG("rdseed", B); + CHECK(A != B); +}