Software fallback for RDRAND/RDSEED

This commit is contained in:
Justin Becker committed 2026-09-16 16:08:40 -07:00
1 parent 9bbdb35711
commit c91c6efc87
12 files changed
+178 -36

No files matched your search

@@ -141,6 +141,14 @@
"Hides hybrid CPU core arrangement."
]
},
"SoftwareRNG": {
"Type": "bool",
"Default": "true",
"AffectsCodeGen": "true",
"Desc": [
"Emulates RDRAND and RDSEED in software when the host does not implement FEAT_RNG."
]
},
"CPUFeatureRegisters": {
"Type": "str",
"Default": "",
@@ -369,10 +369,16 @@ public:
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
FEX_CONFIG_OPT(SoftwareRNG, SOFTWARERNG);
FEX_CONFIG_OPT(StrictInProcessSplitLocks, STRICTINPROCESSSPLITLOCKS);
FEX_CONFIG_OPT(MonoHacks, MONOHACKS);
} Config;
bool SoftwareRNGEnabled() const {
return Config.SoftwareRNG() && HostRNGAvailable;
}
bool HostRNGAvailable {};
FEXCore::Utils::WritePriorityMutex::Mutex CodeInvalidationMutex {};
uint32_t StrictSplitLockMutex {};
+37 -33
View File
@@ -458,6 +458,8 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
(Cores << 16) | // Number of addressable IDs for the logical cores in the physical CPU
(GetCPUID() << 24); // Local APIC ID
const uint32_t SupportsRAND = CTX->HostFeatures.SupportsRAND || CTX->SoftwareRNGEnabled();
Res.ecx = (1 << 0) | // SSE3
(CTX->HostFeatures.SupportsPMULL_128Bit << 1) | // PCLMULQDQ
(1 << 2) | // DS area supports 64bit layout
@@ -488,7 +490,7 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_01h(uint32_t Leaf) const {
(SupportsAVX() << 27) | // OSXSAVE
(SupportsAVX() << 28) | // AVX
(SupportsAVX() << 29) | // F16C
(CTX->HostFeatures.SupportsRAND << 30) | // RDRAND
(SupportsRAND << 30) | // RDRAND
(Hypervisor << 31);
Res.edx = (1 << 0) | // FPU
@@ -670,42 +672,44 @@ FEXCore::CPUID::FunctionResults CPUIDEmu::Function_07h(uint32_t Leaf) const {
const uint32_t SupportsVPCLMULQDQ = CTX->HostFeatures.SupportsPMULL_128Bit && SupportsAVX();
const uint32_t SupportsWFXT = CTX->HostFeatures.SupportsWFXT;
const uint32_t SupportsRAND = CTX->HostFeatures.SupportsRAND || CTX->SoftwareRNGEnabled();
// Number of subfunctions
// TODO: For now, subfunction 1 only exposes AVX-VNNI so we make it conditional
// on AVX-VNNI support. We should revisit this if/when we add more to this leaf.
Res.eax = SupportsAVXVNNI;
Res.ebx = (1 << 0) | // FS/GS support
(0 << 1) | // TSC adjust MSR
(0 << 2) | // SGX
(SupportsAVX() << 3) | // BMI1
(0 << 4) | // Intel Hardware Lock Elison
(SupportsAVX() << 5) | // AVX2 support
(1 << 6) | // FPU data pointer updated only on exception
(1 << 7) | // SMEP support
(SupportsAVX() << 8) | // BMI2
(SupportsEnhancedREPMOVS << 9) | // Enhanced REP MOVSB/STOSB
(1 << 10) | // INVPCID for system software control of process-context
(0 << 11) | // Restricted transactional memory
(0 << 12) | // Intel resource directory technology Monitoring
(1 << 13) | // Deprecates FPU CS and DS
(0 << 14) | // Intel MPX
(0 << 15) | // Intel Resource Directory Technology Allocation
(0 << 16) | // AVX512-F
(0 << 17) | // AVX512-DQ
(CTX->HostFeatures.SupportsRAND << 18) | // RDSEED
(1 << 19) | // ADCX and ADOX instructions
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
(0 << 21) | // AVX512-IFMA
(0 << 22) | // PCOMMIT (deprecated?)
(1 << 23) | // CLFLUSHOPT instruction
(1 << 24) | // CLWB instruction
(0 << 25) | // Intel processor trace
(0 << 26) | // AVX512-PF
(0 << 27) | // AVX512-ER
(0 << 28) | // AVX512-CD
(Features.SHA << 29) | // SHA instructions
(0 << 30) | // AVX512-BW
(0 << 31); // AVX512-VL
Res.ebx = (1 << 0) | // FS/GS support
(0 << 1) | // TSC adjust MSR
(0 << 2) | // SGX
(SupportsAVX() << 3) | // BMI1
(0 << 4) | // Intel Hardware Lock Elison
(SupportsAVX() << 5) | // AVX2 support
(1 << 6) | // FPU data pointer updated only on exception
(1 << 7) | // SMEP support
(SupportsAVX() << 8) | // BMI2
(SupportsEnhancedREPMOVS << 9) | // Enhanced REP MOVSB/STOSB
(1 << 10) | // INVPCID for system software control of process-context
(0 << 11) | // Restricted transactional memory
(0 << 12) | // Intel resource directory technology Monitoring
(1 << 13) | // Deprecates FPU CS and DS
(0 << 14) | // Intel MPX
(0 << 15) | // Intel Resource Directory Technology Allocation
(0 << 16) | // AVX512-F
(0 << 17) | // AVX512-DQ
(SupportsRAND << 18) | // RDSEED
(1 << 19) | // ADCX and ADOX instructions
(0 << 20) | // SMAP Supervisor mode access prevention and CLAC/STAC instructions
(0 << 21) | // AVX512-IFMA
(0 << 22) | // PCOMMIT (deprecated?)
(1 << 23) | // CLFLUSHOPT instruction
(1 << 24) | // CLWB instruction
(0 << 25) | // Intel processor trace
(0 << 26) | // AVX512-PF
(0 << 27) | // AVX512-ER
(0 << 28) | // AVX512-CD
(Features.SHA << 29) | // SHA instructions
(0 << 30) | // AVX512-BW
(0 << 31); // AVX512-VL
Res.ecx = (1 << 0) | // PREFETCHWT1
(0 << 1) | // AVX512VBMI
+8
View File
@@ -105,6 +105,14 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
// Track atomic TSO emulation configuration.
UpdateAtomicTSOEmulationConfig();
#ifndef _WIN32
// Check if the kernel supports getrandom().
uint64_t Probe {};
HostRNGAvailable = FHU::Syscalls::getrandom(&Probe, sizeof(Probe), 0) == sizeof(Probe);
#else
HostRNGAvailable = true;
#endif
DiskCache.Init(this);
}
+36
View File
@@ -36,7 +36,14 @@ $end_info$
#include <cstdio>
#include <cstring>
#include <optional>
#include <type_traits>
#include <unistd.h>
#ifdef _WIN32
#include <atomic>
#else
#include <FEXHeaderUtils/Syscalls.h>
#endif
namespace {
struct DivRem {
@@ -63,6 +70,34 @@ LDIV(uint64_t SrcHigh, uint64_t SrcLow, int64_t Divisor) {
};
}
#ifndef _WIN32
static std::optional<uint64_t> RDRANDFallback(uint64_t Reseed) {
uint64_t Value {};
FHU::Syscalls::getrandom(&Value, sizeof(Value), 0);
return Value;
}
#else
// Windows does not have an equivalent to getrandom() without dynamically linking
// bcrypt et al. Since this fallback is just for compat it does not need to be
// cryptographic, so instead we vendor SplitMix64 as a naive fallback.
// Reference implementation by Sebastiano Vigna, public domain (CC0)
// https://prng.di.unimi.it/splitmix64.c
static std::atomic<uint64_t>
RNGState {static_cast<uint64_t>(__builtin_readcyclecounter())};
static std::optional<uint64_t> RDRANDFallback(uint64_t Reseed) {
const uint64_t State = RNGState.load(std::memory_order_relaxed) + 0x9E3779B97F4A7C15ULL;
RNGState.store(State, std::memory_order_relaxed);
uint64_t Value = (State ^ (State >> 30)) * 0xBF58476D1CE4E5B9ULL;
Value = (Value ^ (Value >> 27)) * 0x94D049BB133111EBULL;
return Value ^ (Value >> 31);
}
#endif
static void
PrintValue(uint64_t Value) {
LogMan::Msg::DFmt("Value: 0x{:x}", Value);
@@ -664,6 +699,7 @@ Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::In
Ptrs.ExitFunctionLink = reinterpret_cast<uintptr_t>(&Arm64JITCore::ExitFunctionLink);
Ptrs.LUDIV = reinterpret_cast<uint64_t>(LUDIV);
Ptrs.LDIV = reinterpret_cast<uint64_t>(LDIV);
Ptrs.RDRANDFallback = reinterpret_cast<uint64_t>(RDRANDFallback);
}
CurrentCodeBuffer = SharedCodeBuffers.GetLatest();
+33 -1
View File
@@ -309,7 +309,39 @@ DEF_OP(ProcessorID) {
DEF_OP(RDRAND) {
auto Op = IROp->C<IR::IROp_RDRAND>();
mrs(GetReg(Node), Op->GetReseeded ? ARMEmitter::SystemRegister::RNDRRS : ARMEmitter::SystemRegister::RNDR);
if (CTX->HostFeatures.SupportsRAND) {
mrs(GetReg(Node), Op->GetReseeded ? ARMEmitter::SystemRegister::RNDRRS : ARMEmitter::SystemRegister::RNDR);
return;
}
// Software fallback, call the host RNG generator.
PushDynamicRegs(TMP4);
SpillStaticRegs(TMP4);
// x0 = Reseed
// x1 = Generator
LoadConstant(ARMEmitter::Size::i64Bit, ARMEmitter::Reg::r0, Op->GetReseeded ? 1 : 0);
ldr(ARMEmitter::XReg::x1, STATE, offsetof(FEXCore::Core::CpuStateFrame, Pointers.RDRANDFallback));
if (!CTX->Config.DisableVixlIndirectCalls) [[unlikely]] {
GenerateIndirectRuntimeCall<__uint128_t, uint64_t>(ARMEmitter::Reg::r1);
} else {
blr(ARMEmitter::Reg::r1);
}
if (!TMP_ABIARGS) {
mov(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Reg::r0);
mov(ARMEmitter::Size::i64Bit, TMP2, ARMEmitter::Reg::r1);
}
FillStaticRegs();
PopDynamicRegs();
// Results are in x0, x1
// std::optional<uint64_t>: value in x0, engaged flag in the low byte of x1. Match the hardware behaviour of setting Z when
// no number was produced.
mov(ARMEmitter::Size::i64Bit, GetReg(Node), TMP1);
tst(ARMEmitter::Size::i64Bit, TMP2, 0xFF);
}
DEF_OP(Yield) {
@@ -5125,7 +5125,7 @@ void OpDispatchBuilder::CRC32(OpcodeArgs) {
}
void OpDispatchBuilder::RDRANDOp(OpcodeArgs, bool Reseed) {
if (!CTX->HostFeatures.SupportsRAND) {
if (!CTX->HostFeatures.SupportsRAND && !CTX->SoftwareRNGEnabled()) {
UnimplementedOp(Op);
return;
}
+2 -1
View File
@@ -275,7 +275,8 @@
"The boolean argument asks if we should be reading the reseeded number or not",
"Reseeded RNG calculation is more expensive and will be heavier to use",
"Returns the 64-bit number",
"Sets the Z flag if the number is valid.",
"Falls back to a host RNG call when the hardware doesn't support it",
"Sets the Z flag if the number is invalid.",
"RNG hardware is allowed to fail early and return. Software must always check this"
],
"HasSideEffects": true,
+1
View File
@@ -350,6 +350,7 @@ struct JITPointers {
uint64_t MonoBackpatcherWrite {};
uint64_t LUDIV {};
uint64_t LDIV {};
uint64_t RDRANDFallback {};
uint64_t ThunkCallbackRet {};
// Handles returning/calling ARM64EC code from the JIT, expects the target PC in TMP3
+4
View File
@@ -96,6 +96,10 @@ inline int32_t renameat2(int olddirfd, const char* oldpath, int newdirfd, const
inline int32_t pidfd_open(pid_t pid, unsigned int flags) {
return ::syscall(SYS_pidfd_open, pid, flags);
}
inline ssize_t getrandom(void* buf, size_t buflen, unsigned int flags) {
return ::syscall(SYS_getrandom, buf, buflen, flags);
}
#else
inline int32_t getcpu(uint32_t* cpu, uint32_t* node) {
+5
View File
@@ -63,6 +63,11 @@ function(AddTests Tests BinDirectory Bitness)
set_property(TEST "${TEST_CASE}.jit.flt" APPEND PROPERTY ENVIRONMENT "FEX_THUNKCONFIG=${CMAKE_SOURCE_DIR}/Data/CI/FEXLinuxTestsThunks.json")
endif()
if(TEST_NAME STREQUAL "rdrand")
# Force use of the software falllback to excercise that path
set_property(TEST "${TEST_CASE}.jit.flt" APPEND PROPERTY ENVIRONMENT "FEX_HOSTFEATURES=disablerng")
endif()
if(TEST_NAME STREQUAL "vulkan_procaddr")
set_property(TEST "${TEST_CASE}.jit.flt" APPEND PROPERTY ENVIRONMENT
"FEX_THUNKCONFIG=${CMAKE_SOURCE_DIR}/Data/CI/VulkanThunks.json;FEX_THUNKHOSTLIBS=${HOSTLIBS_DATA_DIRECTORY}/HostThunks;FEX_THUNKGUESTLIBS=${CMAKE_INSTALL_PREFIX}/share/fex-emu/GuestThunks")
@@ -0,0 +1,37 @@
#include <catch2/catch_test_macros.hpp>
#include <cstdint>
// Retries until the instruction reports success.
// clang-format off
#define RNG(insn, Value) \
__asm volatile(".Lretry%=:\n" \
insn " %0\n" \
"jnc .Lretry%=\n" \
: "=r"(Value) \
: \
: "cc")
// clang-format on
// This test is designed to test the software fallback implementation for RDRAND and RDSEED.
// To that end, it is executed with FEX_HOSTFEATURES=disablerng.
TEST_CASE("rdrand") {
uint32_t eax, ebx, ecx, edx;
__asm volatile("cpuid" : "=a"(eax), "=b"(ebx), "=c"(ecx), "=d"(edx) : "a"(1), "c"(0));
CHECK((ecx >> 30) & 1); // RDRAND
__asm volatile("cpuid" : "=a"(eax), "=b"(ebx), "=c"(ecx), "=d"(edx) : "a"(7), "c"(0));
CHECK((ebx >> 18) & 1); // RDSEED
uintptr_t A, B;
RNG("rdrand", A);
RNG("rdrand", B);
// This assert is not technically correct,
// but at a 1/(2^32) chance of failure, I like our odds.
CHECK(A != B);
RNG("rdseed", A);
RNG("rdseed", B);
CHECK(A != B);
}