Merge pull request #4835 from pmatos/fix/classify

Fix quiet and signalling nan propagation
This commit is contained in:
Ryan Houdek authored and GitHub committed 2025-09-23 11:17:37 -07:00
commit e5680e031d
36 files changed
+7780 -86

No files matched your search

+3
View File
@@ -7,3 +7,6 @@ FEXCore/Source/Interface/Core/X86Tables/*
# Inline headers with list-like content that can't be processed individually
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/SyscallsNames.inl
Source/Tools/LinuxEmulation/LinuxSyscalls/x*/Ioctl/*.inl
# Include files in unittests
unittests/*ASM/Includes/*.inc
+74
View File
@@ -504,6 +504,47 @@ struct FEX_PACKED X80SoftFloat {
return FEXCore::BitCast<float>(Result);
}
bool IsSignalingNaN() const {
return (Exponent == 0x7FFF) && (Significand & 0x8000000000000000ULL) && !(Significand & 0x4000000000000000ULL) && // Bit 62 clear (signaling)
(Significand & 0x3FFFFFFFFFFFFFFFULL);
}
bool IsQuietNaN() const {
return (Exponent == 0x7FFF) && (Significand & 0x8000000000000000ULL) && (Significand & 0x4000000000000000ULL); // Bit 62 set (quiet)
}
// Helper to detect if this is any NaN
bool IsNaN() const {
return IsSignalingNaN() || IsQuietNaN();
}
// X87 value to F64 while preserving signaling nan property
double ToF64_PreserveNan(softfloat_state* state) const {
if (IsSignalingNaN()) {
// we keep it as a signaling nan in ieee754 in 64bits
uint64_t sign_bit = Sign ? 0x8000000000000000ULL : 0;
uint64_t exp_bits = 0x7FF0000000000000ULL;
uint64_t x87_frac = Significand & 0x3FFFFFFFFFFFFFFFULL;
uint64_t ieee_frac = (x87_frac >> 11) & 0x0007FFFFFFFFFFFFULL;
if (ieee_frac == 0) {
ieee_frac = 1;
}
ieee_frac &= ~0x0008000000000000ULL;
uint64_t result_bits = sign_bit | exp_bits | ieee_frac;
return FEXCore::BitCast<double>(result_bits);
} else if (IsQuietNaN()) {
const float64_t Result = extF80_to_f64(state, *this);
uint64_t result_bits = FEXCore::BitCast<uint64_t>(Result);
result_bits |= 0x0008000000000000ULL;
return FEXCore::BitCast<double>(result_bits);
} else {
const float64_t Result = extF80_to_f64(state, *this);
return FEXCore::BitCast<double>(Result);
}
}
double ToF64(softfloat_state* state) const {
const float64_t Result = extF80_to_f64(state, *this);
return FEXCore::BitCast<double>(Result);
@@ -584,6 +625,39 @@ struct FEX_PACKED X80SoftFloat {
*this = f64_to_extF80(state, FEXCore::BitCast<float64_t>(rhs));
}
// Create X80SoftFloat from double while preserving NaN signaling properties
static X80SoftFloat FromF64_PreserveNaN(softfloat_state* state, double value) {
uint64_t bits = FEXCore::BitCast<uint64_t>(value);
// Check if it's a nan
if ((bits & 0x7FF0000000000000ULL) == 0x7FF0000000000000ULL && (bits & 0x000FFFFFFFFFFFFFULL) != 0) {
X80SoftFloat result;
result.Sign = (bits >> 63) & 1;
result.Exponent = 0x7FFF;
bool is_signaling = !(bits & 0x0008000000000000ULL);
uint64_t ieee_payload = bits & 0x0007FFFFFFFFFFFFULL;
// set bit 63 required for x87
result.Significand = 0x8000000000000000ULL;
if (is_signaling) { // clear bit 62 for signaling nan
result.Significand &= ~0x4000000000000000ULL;
} else { // clear bit 62 for quiet nan
result.Significand |= 0x4000000000000000ULL;
}
// ieee754 51-bit payload -> x87 62-bit payload
result.Significand |= (ieee_payload << 11) & 0x3FFFFFFFFFFFFFFFULL;
return result;
}
// For non-NaN values, use standard conversion
return X80SoftFloat(state, value);
}
X80SoftFloat(softfloat_state* state, BIGFLOAT rhs) {
#if BIGFLOATSIZE == 16
*this = f128_to_extF80(state, FEXCore::BitCast<float128_t>(rhs));
@@ -396,6 +396,14 @@
"Emulates X87 floating point using 64-bit precision. This reduces emulation accuracy and may result in rendering bugs."
]
},
"X87StrictReducedPrecision": {
"Type": "bool",
"Default": "false",
"Desc": [
"Enables stricter X87 floating point behavior when X87ReducedPrecision is enabled.",
"Adds additional checks and implementations like NaN propagation for better compatibility."
]
},
"ABILocalFlags": {
"Type": "bool",
"Default": "false",
@@ -207,6 +207,7 @@ public:
FEX_CONFIG_OPT(GDBSymbols, GDBSYMBOLS);
FEX_CONFIG_OPT(ParanoidTSO, PARANOIDTSO);
FEX_CONFIG_OPT(x87ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(x87StrictReducedPrecision, X87STRICTREDUCEDPRECISION);
FEX_CONFIG_OPT(DisableTelemetry, DISABLETELEMETRY);
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
FEX_CONFIG_OPT(SmallTSCScale, SMALLTSCSCALE);
@@ -322,6 +323,13 @@ protected:
}
}
void UpdateX87PrecisionConfig() {
// If strict reduced precision is enabled, automatically enable reduced precision
if (Config.x87StrictReducedPrecision() && !Config.x87ReducedPrecision()) {
FEXCore::Config::Set(FEXCore::Config::CONFIG_X87REDUCEDPRECISION, "1");
}
}
private:
/**
* @brief Initializes the JIT compilers for the thread
+2 -7
View File
@@ -57,20 +57,13 @@ $end_info$
#include <algorithm>
#include <array>
#include <atomic>
#include <chrono>
#include <condition_variable>
#include <fcntl.h>
#include <functional>
#include <mutex>
#include <queue>
#include <shared_mutex>
#include <signal.h>
#include <stdio.h>
#include <string_view>
#include <sys/stat.h>
#include <type_traits>
#include <unistd.h>
#include <unordered_map>
#include <utility>
#include <xxhash.h>
@@ -100,6 +93,8 @@ ContextImpl::ContextImpl(const FEXCore::HostFeatures& Features)
// Track atomic TSO emulation configuration.
UpdateAtomicTSOEmulationConfig();
// Ensure X87 precision constraints are respected.
UpdateX87PrecisionConfig();
}
struct GetFrameBlockInfoResult {
@@ -2,11 +2,13 @@
#pragma once
#include "Common/SoftFloat.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
#include "Interface/IR/IR.h"
#include <FEXCore/Debug/InternalThreadState.h>
#include <FEXCore/Utils/SHMStats.h>
#include <FEXCore/Config/Config.h>
namespace FEXCore::CPU {
FEXCORE_PRESERVE_ALL_ATTR static softfloat_state SoftFloatStateFromFCW(uint16_t FCW, bool Force80BitPrecision = false) {
@@ -77,6 +79,12 @@ struct OpHandlers<IR::OP_F80CVTTO> {
FEXCORE_PRESERVE_ALL_ATTR static VectorRegType handle8(uint16_t FCW, double src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
auto Context = static_cast<Context::ContextImpl*>(Frame->Thread->CTX);
auto ReducedPrecisionMode = Context->Config.x87ReducedPrecision;
auto StrictReducedPrecisionMode = Context->Config.x87StrictReducedPrecision;
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
return X80SoftFloat::FromF64_PreserveNaN(&State.State, src);
}
return X80SoftFloat(&State.State, src);
}
};
@@ -115,6 +123,12 @@ struct OpHandlers<IR::OP_F80CVT> {
FEXCORE_PRESERVE_ALL_ATTR static double handle8(uint16_t FCW, VectorRegType src, FEXCore::Core::CpuStateFrame* Frame) {
FEXCORE_PROFILE_INSTANT_INCREMENT(Frame->Thread, AccumulatedFloatFallbackCount, 1);
ScopedSoftFloatState State {FCW, Frame};
auto Context = static_cast<Context::ContextImpl*>(Frame->Thread->CTX);
auto ReducedPrecisionMode = Context->Config.x87ReducedPrecision;
auto StrictReducedPrecisionMode = Context->Config.x87StrictReducedPrecision;
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
return X80SoftFloat(src).ToF64_PreserveNan(&State.State);
}
return X80SoftFloat(src).ToF64(&State.State);
}
};
@@ -1329,6 +1329,7 @@ protected:
private:
FEX_CONFIG_OPT(ReducedPrecisionMode, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(StrictReducedPrecisionMode, X87STRICTREDUCEDPRECISION);
struct JumpTargetInfo {
Ref BlockEntry;
@@ -158,7 +158,9 @@ public:
: Features(Features)
, GPROpSize(GPROpSize) {
FEX_CONFIG_OPT(ReducedPrecision, X87REDUCEDPRECISION);
FEX_CONFIG_OPT(StrictReducedPrecision, X87STRICTREDUCEDPRECISION);
ReducedPrecisionMode = ReducedPrecision;
StrictReducedPrecisionMode = StrictReducedPrecision;
}
void Run(IREmitter* Emit) override;
@@ -166,10 +168,12 @@ private:
const FEXCore::HostFeatures& Features;
const OpSize GPROpSize;
bool ReducedPrecisionMode;
bool StrictReducedPrecisionMode;
FEX_CONFIG_OPT(DisableVixlIndirectCalls, DISABLE_VIXL_INDIRECT_RUNTIME_CALLS);
// Helpers
Ref RotateRight8(uint32_t V, Ref Amount);
Ref SilenceNaN(Ref Value);
void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
Ref AddrNode = IR->GetNode(Op->Addr);
@@ -204,6 +208,9 @@ private:
case OpSize::i32Bit:
case OpSize::i64Bit: {
StackNode = IREmit->_F80CVT(Op->StoreSize, StackNode);
if (!ReducedPrecisionMode || StrictReducedPrecisionMode) {
StackNode = SilenceNaN(StackNode);
}
IREmit->_StoreMem(FPRClass, Op->StoreSize, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
break;
}
@@ -235,6 +242,10 @@ private:
MemOffsetType OffsetType = Op->OffsetType;
uint8_t OffsetScale = Op->OffsetScale;
if ((!ReducedPrecisionMode || StrictReducedPrecisionMode) && Op->StoreSize != OpSize::f80Bit) {
StackNode = SilenceNaN(StackNode);
}
switch (Op->StoreSize) {
case OpSize::i32Bit: {
StackNode = IREmit->_Float_FToF(OpSize::i32Bit, OpSize::i64Bit, StackNode);
@@ -489,6 +500,15 @@ inline Ref X87StackOptimization::RotateRight8(uint32_t V, Ref Amount) {
return IREmit->_Lshr(OpSize::i32Bit, GetConstant(V | (V << 8)), Amount);
}
inline Ref X87StackOptimization::SilenceNaN(Ref Value) {
Ref GPRValue = IREmit->_VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, Value, 0);
IREmit->_FCmp(OpSize::i64Bit, Value, Value); // Comparison with itself should set VS if nan
Ref QuietNaNGPR = IREmit->_Or(OpSize::i64Bit, GPRValue, IREmit->_Constant(0x0008000000000000ULL));
Ref SilencedValue = IREmit->_VCastFromGPR(OpSize::i64Bit, OpSize::i64Bit, QuietNaNGPR);
return IREmit->_NZCVSelectV(OpSize::i64Bit, CondClassType {COND_VS}, SilencedValue, Value);
}
inline std::optional<X87StackOptimization::StackMemberInfo> X87StackOptimization::MigrateToSlowPath_IfInvalid(uint8_t Offset) {
const auto& [Valid, StackMember] = StackData.top(Offset);
MigrateToSlowPathIf(Valid != StackSlot::VALID);
+7
View File
@@ -721,10 +721,17 @@ ApplicationWindow {
}
ConfigCheckBox {
id: x87ReducedPrecisionCheckbox
text: qsTr("Reduced x87 precision")
config: "X87ReducedPrecision"
}
ConfigCheckBox {
text: qsTr("Strict reduced x87 precision")
config: "X87StrictReducedPrecision"
enabled: x87ReducedPrecisionCheckbox.checked
}
ConfigCheckBox {
text: qsTr("Unsafe local flags optimization")
config: "ABILocalFlags"
+1 -1
View File
@@ -38,7 +38,7 @@ foreach(ASM_SRC ${ASM_SOURCES})
add_custom_command(OUTPUT ${OUTPUT_NAME}
DEPENDS "${TMP_FILE}"
COMMAND "nasm" ARGS "${TMP_FILE}" "-o" "${OUTPUT_NAME}")
COMMAND "nasm" ARGS "-i" "${CMAKE_SOURCE_DIR}/unittests/32Bit_ASM/Includes/" "${TMP_FILE}" "-o" "${OUTPUT_NAME}")
add_custom_command(OUTPUT ${OUTPUT_CONFIG_NAME}
DEPENDS "${ASM_SRC}"
@@ -0,0 +1,130 @@
; NaN Testing Macros for 32-bit Assembly Tests
; Implements NaN triple testing system:
; - Bit 2: 1 if value is NaN
; - Bit 1: 1 if quiet NaN
; - Bit 0: 1 if signaling NaN
;
; Triple values:
; 0b000 (0): Not a NaN
; 0b101 (5): Signaling NaN
; 0b110 (6): Quiet NaN
;
; ASSUMPTION: All input pointers (edx) are valid and non-null
; Macro: CHECK_NAN_TRIPLE_32
; Checks 32-bit float NaN classification and returns triple in EAX
; Input: 32-bit float value in xmm0
; Output: NaN triple in EAX (bits 2:0)
%macro CHECK_NAN_TRIPLE_32 0
push ecx
push esi
push edx
xor eax, eax
ucomiss xmm0, xmm0
setp al
mov ecx, eax
shl ecx, 2
; Extract and check quiet bit (bit 22)
movd edx, xmm0
and edx, 0x00400000
mov esi, edx
shr esi, 22
and esi, eax
and esi, 1
shl esi, 1
add ecx, esi
; Check for signaling NaN (NaN but not quiet)
test edx, edx
sete dl
and dl, al
movzx eax, dl
or eax, ecx
pop edx
pop esi
pop ecx
%endmacro
; Macro: CHECK_NAN_TRIPLE_64
; Checks 64-bit double NaN classification and returns triple in EAX
; Input: 64-bit double value should be pre-stored at [edx] by caller
; Output: NaN triple in EAX (bits 2:0)
%macro CHECK_NAN_TRIPLE_64 0
push ebx
push esi
sub esp, 12
; Load 64-bit double and use SSE for NaN comparison
movsd xmm0, qword [edx]
xor eax, eax
ucomisd xmm0, xmm0
setp al
mov ecx, eax
movsd qword [esp], xmm0
mov edx, 524288
and edx, [esp + 4]
shl ecx, 2
; Extract quiet bit (bit 51)
mov ebx, edx
shr ebx, 19
and bl, al
movzx esi, bl
lea ecx, [ecx + 2*esi]
; Check for signaling NaN (NaN but not quiet)
test edx, edx
sete dl
and dl, al
movzx eax, dl
or eax, ecx
add esp, 12
pop esi
pop ebx
%endmacro
; Macro: CHECK_NAN_TRIPLE_80
; Checks 80-bit extended precision NaN classification and returns triple in EAX
; Input: 80-bit extended precision value in memory at [eax] (10 bytes)
; Output: NaN triple in EAX (bits 2:0)
%macro CHECK_NAN_TRIPLE_80 0
push ebx
push esi
sub esp, 20
; Load the 80-bit value and store copy for bit manipulation
fld tword [eax]
fld st0
fstp tword [esp]
; Get bits 63:32 from stored significand
mov ecx, [esp + 4]
xor eax, eax
fucomip st0
setp al
mov edx, eax
shl edx, 2
; Extract quiet bit (bit 30 in high dword)
mov ebx, ecx
shr ebx, 30
and bl, al
movzx esi, bl
lea edx, [edx + 2*esi]
; Check for signaling NaN using bt instruction
bt ecx, 30
setae cl
and cl, al
movzx eax, cl
or eax, edx
add esp, 20
pop esi
pop ebx
%endmacro
@@ -0,0 +1,36 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "6"
},
"Mode": "32BIT"
}
%endif
%include "nan_test_macros.inc"
mov esp, 0xe0000040
; Test x87 quiet NaN preservation in non-reduced precision mode (32-bit)
; This test verifies that FLDT loads a quiet NaN and preserves its nature
; We test that loading a quiet nan preserves it
; that then storing it as 32bit, keeps it as a quiet nan.
; Returns NaN triple: 6 (0b110) for quiet NaN
finit
lea edx, [.data]
fld tword [edx] ; load qnan 80bit
fstp dword [edx + 16] ; store qnan as 32bit
; Check the stored 32-bit value using NaN triple macro
lea edx, [.data + 16]
movss xmm0, [edx] ; Load 32-bit float into xmm0
CHECK_NAN_TRIPLE_32
hlt
align 8
.data:
dq 0xc000000000000000 ; quiet NaN significand
dw 0x7fff ; NaN exponent
dd 0 ; space for 32-bit result
@@ -0,0 +1,36 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "6"
},
"Mode": "32BIT"
}
%endif
%include "nan_test_macros.inc"
mov esp, 0xe0000040
; Test x87 signaling NaN non-preservation in non-reduced precision mode (32-bit)
; This test verifies that FLDT loads a signaling NaN and DOES NOT preserve its signaling nature
; We test that loading a signaling nan preserves it but
; that then storing it as 32bit, transforms it to a quiet nan.
; Returns NaN triple: 6 (0b110) for quiet NaN (converted from signaling)
finit
lea edx, [.data]
fld tword [edx] ; load snan
fstp dword [edx + 16] ; store snan as 32bit qnan
; Check the stored 32-bit value using NaN triple macro
lea edx, [.data + 16]
movss xmm0, [edx] ; Load 32-bit float into xmm0
CHECK_NAN_TRIPLE_32
hlt
align 8
.data:
dq 0xa000000000000000 ; signaling NaN significand
dw 0x7fff ; signaling NaN exponent
dd 0 ; space for 32-bit result
@@ -0,0 +1,34 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "5"
},
"Mode": "32BIT"
}
%endif
%include "nan_test_macros.inc"
mov esp, 0xe0000040
; Test x87 signaling negative NaN round-trip preservation in non-reduced precision mode (32-bit)
; This test verifies that FLDT -> FSTPT preserves signaling nan across round-trip
; Returns NaN triple: 5 (0b101) for signaling NaN
finit
lea edx, [.data]
fld tword [edx] ; load snan 80bit
fstp tword [edx + 16] ; store nan as 80bit
; Check the stored 80-bit value using NaN triple macro
lea eax, [.data + 16]
CHECK_NAN_TRIPLE_80
hlt
align 16
.data:
dq 0xa000000000000000 ; signaling nan significand
dw 0xffff ; signaling nan exponent
dw 0, 0, 0 ; padding to 16 bytes
dq 0, 0 ; space for 80-bit result (16 bytes)
@@ -0,0 +1,34 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "6"
},
"Mode": "32BIT"
}
%endif
%include "nan_test_macros.inc"
mov esp, 0xe0000040
; Test x87 quiet NaN preservation in non-reduced precision mode (32-bit)
; This test verifies that quiet NaNs remain quiet during conversion
; Returns NaN triple: 6 (0b110) for quiet NaN
finit
lea edx, [.data]
fld tword [edx] ; load qnan 80bit
fstp qword [edx + 16] ; store qnan as 64bit
; Check the stored 64-bit value using NaN triple macro
lea edx, [.data + 16]
movsd xmm0, [edx] ; Load 64-bit double into xmm0
CHECK_NAN_TRIPLE_64
hlt
align 8
.data:
dq 0xc000000000000000 ; quiet NaN significand
dw 0x7fff ; NaN exponent
dq 0 ; space for 64-bit result
@@ -0,0 +1,36 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "6"
},
"Mode": "32BIT"
}
%endif
%include "nan_test_macros.inc"
mov esp, 0xe0000040
; Test x87 signaling NaN non-preservation in non-reduced precision mode (32-bit)
; This test verifies that FLDT loads a signaling NaN and DOES NOT preserve its signaling nature
; We test that loading a signaling nan preserves it but
; that then storing it as 64bit, transforms it to a quiet nan.
; Returns NaN triple: 6 (0b110) for quiet NaN
finit
mov edx, .data
fld tword [edx] ; load snan
fstp qword [edx + 16] ; store snan as 64bit qnan
; Check the stored 64-bit value using NaN triple macro
mov edx, .data
add edx, 16
CHECK_NAN_TRIPLE_64
hlt
align 8
.data:
dq 0xa000000000000000 ; signaling NaN significand
dw 0x7fff ; signaling NaN exponent
dq 0 ; space for 64-bit result
@@ -0,0 +1,35 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "5"
},
"Mode": "32BIT"
}
%endif
%include "nan_test_macros.inc"
mov esp, 0xe0000040
; Test x87 signaling NaN round-trip preservation in non-reduced precision mode (32-bit)
; This test verifies that FLDT -> FSTPT preserves signaling nan across round-trip
; Returns NaN triple: 5 (0b101) for signaling NaN
finit
mov edx, .data
fld tword [edx] ; load nan 80bit
fstp tword [edx + 16] ; store nan as 80bit
; Check the stored 80-bit value using NaN triple macro
mov eax, edx
add eax, 16
CHECK_NAN_TRIPLE_80
hlt
align 16
.data:
dq 0xa000000000000000 ; signaling nan significand
dw 0x7fff ; signaling nan exponent
dw 0, 0, 0 ; padding to 16 bytes
dq 0, 0 ; space for 80-bit result (16 bytes)
+128
View File
@@ -0,0 +1,128 @@
; NaN Testing Macros for Assembly Tests
; Implements NaN triple testing system:
; - Bit 2: 1 if value is NaN
; - Bit 1: 1 if quiet NaN
; - Bit 0: 1 if signaling NaN
;
; Triple values:
; 0b000 (0): Not a NaN
; 0b101 (5): Signaling NaN
; 0b110 (6): Quiet NaN
;
; ASSUMPTION: All input pointers (rdx) are valid and non-null
; Macro: CHECK_NAN_TRIPLE_32
; Checks 32-bit float NaN classification and returns triple in EAX
; Input: 32-bit float value in xmm0
; Output: NaN triple in EAX (bits 2:0)
%macro CHECK_NAN_TRIPLE_32 0
push rcx
push rsi
push rdx
xor eax, eax
ucomiss xmm0, xmm0
setp al
lea rcx, [4*rax]
; Extract and check quiet bit (bit 22)
movd edx, xmm0
and edx, 0x00400000
mov esi, edx
shr esi, 22
and sil, al
movzx esi, sil
lea rcx, [rcx + 2*rsi]
; Check for signaling NaN (NaN but not quiet)
test edx, edx
sete dl
and dl, al
movzx eax, dl
or eax, ecx
pop rdx
pop rsi
pop rcx
%endmacro
; Macro: CHECK_NAN_TRIPLE_64
; Checks 64-bit double NaN classification and returns triple in RAX
; Input: 64-bit double value in xmm0
; Output: NaN triple in RAX (bits 2:0)
%macro CHECK_NAN_TRIPLE_64 0
push rcx
push rsi
push rdx
xor eax, eax
ucomisd xmm0, xmm0
setp al
lea rcx, [4*rax]
; Extract and check quiet bit (bit 51)
movq rdx, xmm0
mov rsi, 0x0008000000000000
and rsi, rdx
mov rdx, rsi
shr rdx, 51
and dl, al
movzx rdx, dl
lea rcx, [rcx + 2*rdx]
; Check for signaling NaN (NaN but not quiet)
test rsi, rsi
sete dl
and dl, al
movzx eax, dl
or eax, ecx
pop rdx
pop rsi
pop rcx
%endmacro
; Macro: CHECK_NAN_TRIPLE_80
; Checks 80-bit extended precision NaN classification and returns triple in RAX
; Input: 80-bit extended precision value in memory at [rax] (10 bytes)
; Output: NaN triple in RAX (bits 2:0)
%macro CHECK_NAN_TRIPLE_80 0
push rcx
push rdx
push rsi
; Load the 80-bit value twice for comparison
fld tword [rax]
fld tword [rax]
; Store one copy to memory for bit manipulation
sub rsp, 16
fstp tword [rsp]
; Use fucomip for NaN detection
xor eax, eax
fucomip st0, st1
setp al
lea rdx, [4*rax]
; Extract and check quiet bit (bit 62)
mov rcx, 0x4000000000000000
and rcx, [rsp]
mov rsi, rcx
shr rsi, 62
and sil, al
movzx rsi, sil
lea rdx, [rdx + 2*rsi]
; Check for signaling NaN (NaN but not quiet)
test rcx, rcx
sete cl
and cl, al
movzx eax, cl
or eax, edx
add rsp, 16
pop rsi
pop rdx
pop rcx
%endmacro
@@ -0,0 +1,35 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "6"
}
}
%endif
%include "nan_test_macros.inc"
mov rsp, 0xe0000040
; Test x87 quiet NaN preservation in non-reduced precision mode
; This test verifies that FLDT loads a quiet NaN and preserves its nature
; We test that loading a quiet nan preserves it
; that then storing it as 32bit, keeps it as a quiet nan.
; Returns NaN triple: 6 (0b110) for quiet NaN
finit
lea rdx, [rel data]
fld tword [rdx] ; load qnan 80bit
fstp dword [rdx + 16] ; store qnan as 32bit
; Check the stored 32-bit value using NaN triple macro
lea rdx, [rel data + 16]
movss xmm0, [rdx] ; Load 32-bit float into xmm0
CHECK_NAN_TRIPLE_32
hlt
align 8
data:
dq 0xc000000000000000 ; quiet NaN significand
dw 0x7fff ; NaN exponent
dd 0 ; space for 32-bit result
@@ -0,0 +1,35 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "6"
}
}
%endif
%include "nan_test_macros.inc"
mov rsp, 0xe0000040
; Test x87 signaling NaN non-preservation in non-reduced precision mode
; This test verifies that FLDT loads a signaling NaN and DOES NOT preserve its signaling nature
; We test that loading a signaling nan preserves it but
; that then storing it as 32bit, transforms it to a quiet nan.
; Returns NaN triple: 6 (0b110) for quiet NaN (converted from signaling)
finit
lea rdx, [rel data]
fld tword [rdx] ; load snan
fstp dword [rdx + 16] ; store snan as 32bit qnan
; Check the stored 32-bit value using NaN triple macro
lea rdx, [rel data + 16]
movss xmm0, [rdx] ; Load 32-bit float into xmm0
CHECK_NAN_TRIPLE_32
hlt
align 8
data:
dq 0xa000000000000000 ; signaling NaN significand
dw 0x7fff ; signaling NaN exponent
dd 0 ; space for 32-bit result
@@ -0,0 +1,33 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "5"
}
}
%endif
%include "nan_test_macros.inc"
mov rsp, 0xe0000040
; Test x87 signaling negative NaN round-trip preservation in non-reduced precision mode
; This test verifies that FLDT -> FSTPT preserves signaling nan across round-trip
; Returns NaN triple: 5 (0b101) for signaling NaN
finit
lea rdx, [rel data]
fld tword [rdx] ; load snan 80bit
fstp tword [rdx + 16] ; store nan as 80bit
; Check the stored 80-bit value using NaN triple macro
lea rax, [rel data + 16]
CHECK_NAN_TRIPLE_80
hlt
align 16
data:
dq 0xa000000000000000 ; signaling nan significand
dw 0xffff ; signaling nan exponent
dw 0, 0, 0 ; padding to 16 bytes
dq 0, 0 ; space for 80-bit result (16 bytes)
@@ -0,0 +1,33 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "6"
}
}
%endif
%include "nan_test_macros.inc"
mov rsp, 0xe0000040
; Test x87 quiet NaN preservation in non-reduced precision mode
; This test verifies that quiet NaNs remain quiet during conversion
; Returns NaN triple: 6 (0b110) for quiet NaN
finit
lea rdx, [rel data]
fld tword [rdx] ; load qnan 80bit
fstp qword [rdx + 16] ; store qnan as 64bit
; Check the stored 64-bit value using NaN triple macro
lea rdx, [rel data + 16]
movsd xmm0, [rdx] ; Load 64-bit double into xmm0
CHECK_NAN_TRIPLE_64
hlt
align 8
data:
dq 0xc000000000000000 ; quiet NaN significand
dw 0x7fff ; NaN exponent
dq 0 ; space for 64-bit result
@@ -0,0 +1,35 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "6"
}
}
%endif
%include "nan_test_macros.inc"
mov rsp, 0xe0000040
; Test x87 signaling NaN non-preservation in non-reduced precision mode
; This test verifies that FLDT loads a signaling NaN and DOES NOT preserve its signaling nature
; We test that loading a signaling nan preserves it but
; that then storing it as 64bit, transforms it to a quiet nan.
; Returns NaN triple: 6 (0b110) for quiet NaN
finit
lea rdx, [rel data]
fld tword [rdx] ; load snan
fstp qword [rdx + 16] ; store snan as 64bit qnan
; Check the stored 64-bit value using NaN triple macro
lea rdx, [rel data + 16]
movsd xmm0, [rdx] ; Load 64-bit double into xmm0
CHECK_NAN_TRIPLE_64
hlt
align 8
data:
dq 0xa000000000000000 ; signaling NaN significand
dw 0x7fff ; signaling NaN exponent
dq 0 ; space for 64-bit result
+33
View File
@@ -0,0 +1,33 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "5"
}
}
%endif
%include "nan_test_macros.inc"
mov rsp, 0xe0000040
; Test x87 signaling NaN round-trip preservation in non-reduced precision mode
; This test verifies that FLDT -> FSTPT preserves signaling nan across round-trip
; Returns NaN triple: 5 (0b101) for signaling NaN
finit
lea rdx, [rel data]
fld tword [rdx] ; load nan 80bit
fstp tword [rdx + 16] ; store nan as 80bit
; Check the stored 80-bit value using NaN triple macro
lea rax, [rel data + 16]
CHECK_NAN_TRIPLE_80
hlt
align 16
data:
dq 0xa000000000000000 ; signaling nan significand
dw 0x7fff ; signaling nan exponent
dw 0, 0, 0 ; padding to 16 bytes
dq 0, 0 ; space for 80-bit result (16 bytes)
@@ -0,0 +1,35 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "6"
},
"Env": { "FEX_X87STRICTREDUCEDPRECISION" : "1" }
}
%endif
%include "nan_test_macros.inc"
mov esp, 0xe0000040
; Test x87 quiet NaN preservation in reduced precision mode
; This test verifies that quiet NaNs remain quiet during conversion
; Returns NaN triple: 6 (0b110) for quiet NaN
finit
lea rdx, [rel data]
fld tword [rdx] ; load qnan 80bit
fstp dword [rdx + 16] ; store qnan as 32bit
; Check the stored 32-bit value using NaN triple macro
lea rdx, [rel data + 16]
movss xmm0, [rdx] ; Load 32-bit float into xmm0
CHECK_NAN_TRIPLE_32
hlt
align 8
data:
dq 0xc000000000000000 ; quiet NaN significand
dw 0x7fff ; NaN exponent
dd 0 ; space for 32-bit result
@@ -0,0 +1,37 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "6"
},
"Env": { "FEX_X87REDUCEDPRECISION" : "1", "FEX_X87STRICTREDUCEDPRECISION" : "1" }
}
%endif
%include "nan_test_macros.inc"
mov esp, 0xe0000040
; Test x87 signaling NaN non-preservation in reduced precision mode
; This test verifies that FLDT loads a signaling NaN and DOES NOT preserve its signaling nature
; We test that loading a signaling nan preserves it but
; that then storing it as 32bit, transforms it to a quiet nan.
; Returns NaN triple: 6 (0b110) for quiet NaN (converted from signaling)
finit
lea rdx, [rel data]
fld tword [rdx] ; load snan
fstp dword [rdx + 16] ; store snan as 32bit qnan
; Check the stored 32-bit value using NaN triple macro
lea rdx, [rel data + 16]
movss xmm0, [rdx] ; Load 32-bit float into xmm0
CHECK_NAN_TRIPLE_32
hlt
align 8
data:
dq 0xa000000000000000 ; signaling NaN significand
dw 0x7fff ; signaling NaN exponent
dd 0 ; space for 32-bit result
@@ -0,0 +1,35 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "5"
},
"Env": { "FEX_X87REDUCEDPRECISION" : "1", "FEX_X87STRICTREDUCEDPRECISION" : "1" }
}
%endif
%include "nan_test_macros.inc"
mov esp, 0xe0000040
; Test x87 signaling negative NaN round-trip preservation in reduced precision mode
; This test verifies that FLDT -> FSTPT preserves signaling nan across round-trip
; Returns NaN triple: 5 (0b101) for signaling NaN
finit
lea rdx, [rel data]
fld tword [rdx] ; load snan 80bit
fstp tword [rdx + 16] ; store nan as 80bit
; Check the stored 80-bit value using NaN triple macro
lea rax, [rel data + 16]
CHECK_NAN_TRIPLE_80
hlt
align 16
data:
dq 0xa000000000000000 ; signaling nan significand
dw 0xffff ; signaling nan exponent
dw 0, 0, 0 ; padding to 16 bytes
dq 0, 0 ; space for 80-bit result (16 bytes)
@@ -0,0 +1,35 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "6"
},
"Env": { "FEX_X87STRICTREDUCEDPRECISION" : "1" }
}
%endif
%include "nan_test_macros.inc"
mov esp, 0xe0000040
; Test x87 quiet NaN preservation in reduced precision mode
; This test verifies that quiet NaNs remain quiet during conversion
; Returns NaN triple: 6 (0b110) for quiet NaN
finit
lea rdx, [rel data]
fld tword [rdx] ; load qnan 80bit
fstp qword [rdx + 16] ; store qnan as 64bit
; Check the stored 64-bit value using NaN triple macro
lea rdx, [rel data + 16]
movsd xmm0, [rdx] ; Load 64-bit double into xmm0
CHECK_NAN_TRIPLE_64
hlt
align 8
data:
dq 0xc000000000000000 ; quiet NaN significand
dw 0x7fff ; NaN exponent
dq 0 ; space for 64-bit result
@@ -0,0 +1,37 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "6"
},
"Env": { "FEX_X87REDUCEDPRECISION" : "1", "FEX_X87STRICTREDUCEDPRECISION" : "1" }
}
%endif
%include "nan_test_macros.inc"
mov esp, 0xe0000040
; Test x87 signaling NaN non-preservation in reduced precision mode
; This test verifies that FLDT loads a signaling NaN and DOES NOT preserve its signaling nature
; We test that loading a signaling nan preserves it but
; that then storing it as 64bit, transforms it to a quiet nan.
; Returns NaN triple: 6 (0b110) for quiet NaN
finit
lea rdx, [rel data]
fld tword [rdx] ; load snan
fstp qword [rdx + 16] ; store snan as 64bit qnan
; Check the stored 64-bit value using NaN triple macro
lea rdx, [rel data + 16]
movsd xmm0, [rdx] ; Load 64-bit double into xmm0
CHECK_NAN_TRIPLE_64
hlt
align 8
data:
dq 0xa000000000000000 ; signaling NaN significand
dw 0x7fff ; signaling NaN exponent
dq 0 ; space for 64-bit result
@@ -0,0 +1,35 @@
%ifdef CONFIG
{
"RegData": {
"RAX": "5"
},
"Env": { "FEX_X87REDUCEDPRECISION" : "1", "FEX_X87STRICTREDUCEDPRECISION" : "1" }
}
%endif
%include "nan_test_macros.inc"
mov esp, 0xe0000040
; Test x87 signaling NaN round-trip preservation in reduced precision mode
; This test verifies that FLDT -> FSTPT preserves signaling nan across round-trip
; Returns NaN triple: 5 (0b101) for signaling NaN
finit
lea rdx, [rel data]
fld tword [rdx] ; load nan 80bit
fstp tword [rdx + 16] ; store nan as 80bit
; Check the stored 80-bit value using NaN triple macro
lea rax, [rel data + 16]
CHECK_NAN_TRIPLE_80
hlt
align 16
data:
dq 0xa000000000000000 ; signaling nan significand
dw 0x7fff ; signaling nan exponent
dw 0, 0, 0 ; padding to 16 bytes
dq 0, 0 ; space for 80-bit result (16 bytes)
File diff suppressed because it is too large. Load diff
@@ -14,7 +14,7 @@
"Instructions": {
"Block1": {
"x86InstructionCount": 70,
"ExpectedInstructionCount": 412,
"ExpectedInstructionCount": 432,
"x86Insts": [
"sub esp,0x2c",
"mov ecx,dword [esp + 0x34]",
@@ -88,7 +88,7 @@
"fucomi st0,st1"
],
"ExpectedArm64ASM": [
"subs w20, w8, #0x2c (44)",
"sub w20, w8, #0x2c (44)",
"mov x27, x8",
"mov x8, x20",
"ldr w7, [x8, #52]",
@@ -149,6 +149,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s7, s0",
"mov x20, v7.d[0]",
"fcmp d7, d7",
"orr x20, x20, #0x8000000000000",
"fmov d8, x20",
"fcsel d7, d8, d7, vs",
"str s7, [x8, #16]",
"ldr s7, [x7, #8]",
"str x30, [sp, #-16]!",
@@ -181,6 +186,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s9, s0",
"mov x20, v9.d[0]",
"fcmp d9, d9",
"orr x20, x20, #0x8000000000000",
"fmov d10, x20",
"fcsel d9, d10, d9, vs",
"str s9, [x8, #20]",
"ldr s9, [x4]",
"str x30, [sp, #-16]!",
@@ -206,6 +216,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s2, s0",
"mov x20, v2.d[0]",
"fcmp d2, d2",
"orr x20, x20, #0x8000000000000",
"fmov d10, x20",
"fcsel d2, d10, d2, vs",
"str s2, [x8, #24]",
"ldr s2, [x4, #4]",
"str x30, [sp, #-16]!",
@@ -231,6 +246,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s5, s0",
"mov x20, v5.d[0]",
"fcmp d5, d5",
"orr x20, x20, #0x8000000000000",
"fmov d10, x20",
"fcsel d5, d10, d5, vs",
"str s5, [x8, #28]",
"ldr s5, [x4, #8]",
"str x30, [sp, #-16]!",
@@ -762,7 +782,7 @@
},
"Block3": {
"x86InstructionCount": 32,
"ExpectedInstructionCount": 231,
"ExpectedInstructionCount": 236,
"x86Insts": [
"fld dword [ecx]",
"fld dword [edx + 0x4]",
@@ -1022,6 +1042,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s4, s0",
"mov x23, v4.d[0]",
"fcmp d4, d4",
"orr x23, x23, #0x8000000000000",
"fmov d5, x23",
"fcsel d4, d5, d4, vs",
"str s4, [x10]",
"add w21, w21, #0x1 (1)",
"and w21, w21, #0x7",
@@ -1033,7 +1058,7 @@
},
"Block4": {
"x86InstructionCount": 54,
"ExpectedInstructionCount": 75,
"ExpectedInstructionCount": 85,
"x86Insts": [
"push ebp",
"push edi",
@@ -1093,7 +1118,7 @@
"ExpectedArm64ASM": [
"stp w11, w9, [x8, #-8]!",
"stp w6, w10, [x8, #-8]!",
"subs w26, w8, #0x4c (76)",
"sub w26, w8, #0x4c (76)",
"mov x27, x8",
"mov x8, x26",
"ldr w4, [x8, #104]",
@@ -1138,6 +1163,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s2, s0",
"mov x20, v2.d[0]",
"fcmp d2, d2",
"orr x20, x20, #0x8000000000000",
"fmov d3, x20",
"fcsel d2, d3, d2, vs",
"str s2, [x8, #44]",
"ldr s2, [x8, #44]",
"str x30, [sp, #-16]!",
@@ -1154,6 +1184,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov d2, d0",
"mov x20, v2.d[0]",
"fcmp d2, d2",
"orr x20, x20, #0x8000000000000",
"fmov d3, x20",
"fcsel d2, d3, d2, vs",
"str d2, [x8]",
"mov w20, #0x44",
"movk w20, #0x1, lsl #16",
@@ -1170,7 +1205,7 @@
},
"Block5": {
"x86InstructionCount": 49,
"ExpectedInstructionCount": 300,
"ExpectedInstructionCount": 325,
"x86Insts": [
"fld dword [esp + 0x80]",
"fsub dword [esp + 0x7c]",
@@ -1258,6 +1293,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s3, s0",
"mov x20, v3.d[0]",
"fcmp d3, d3",
"orr x20, x20, #0x8000000000000",
"fmov d4, x20",
"fcsel d3, d4, d3, vs",
"str s3, [x8, #52]",
"ldr s3, [x8, #36]",
"str x30, [sp, #-16]!",
@@ -1299,6 +1339,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s2, s0",
"mov x20, v2.d[0]",
"fcmp d2, d2",
"orr x20, x20, #0x8000000000000",
"fmov d4, x20",
"fcsel d2, d4, d2, vs",
"str s2, [x8, #44]",
"ldr s2, [x4]",
"str x30, [sp, #-16]!",
@@ -1356,6 +1401,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s2, s0",
"mov x20, v2.d[0]",
"fcmp d2, d2",
"orr x20, x20, #0x8000000000000",
"fmov d4, x20",
"fcsel d2, d4, d2, vs",
"str s2, [x8, #68]",
"ldr s2, [x4, #4]",
"str x30, [sp, #-16]!",
@@ -1412,6 +1462,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s2, s0",
"mov x20, v2.d[0]",
"fcmp d2, d2",
"orr x20, x20, #0x8000000000000",
"fmov d4, x20",
"fcsel d2, d4, d2, vs",
"str s2, [x8, #72]",
"ldr s2, [x4, #8]",
"str x30, [sp, #-16]!",
@@ -1474,6 +1529,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s2, s0",
"mov x20, v2.d[0]",
"fcmp d2, d2",
"orr x20, x20, #0x8000000000000",
"fmov d3, x20",
"fcsel d2, d3, d2, vs",
"str s2, [x8, #76]",
"movi v2.2d, #0x0",
"ldr s3, [x8, #40]",
@@ -1868,7 +1928,7 @@
},
"Block7": {
"x86InstructionCount": 25,
"ExpectedInstructionCount": 244,
"ExpectedInstructionCount": 249,
"x86Insts": [
"fld dword [ebx + 0x4]",
"fld dword [ebx]",
@@ -2131,6 +2191,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s3, s0",
"mov x22, v3.d[0]",
"fcmp d3, d3",
"orr x22, x22, #0x8000000000000",
"fmov d4, x22",
"fcsel d3, d4, d3, vs",
"str s3, [x11]",
"strb w21, [x28, #1019]",
"str q2, [x20, #1040]",
@@ -2145,7 +2210,7 @@
},
"Block8": {
"x86InstructionCount": 25,
"ExpectedInstructionCount": 72,
"ExpectedInstructionCount": 92,
"x86Insts": [
"fstp st0",
"fstp st3",
@@ -2204,6 +2269,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s5, s0",
"mov x12, v5.d[0]",
"fcmp d5, d5",
"orr x12, x12, #0x8000000000000",
"fmov d6, x12",
"fcsel d5, d6, d5, vs",
"str s5, [x8, #56]",
"strb wzr, [x28, #1017]",
"str x30, [sp, #-16]!",
@@ -2213,6 +2283,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s5, s0",
"mov x12, v5.d[0]",
"fcmp d5, d5",
"orr x12, x12, #0x8000000000000",
"fmov d6, x12",
"fcsel d5, d6, d5, vs",
"str s5, [x8, #44]",
"add w20, w20, #0x1 (1)",
"and w20, w20, #0x7",
@@ -2226,6 +2301,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s6, s0",
"mov x13, v6.d[0]",
"fcmp d6, d6",
"orr x13, x13, #0x8000000000000",
"fmov d7, x13",
"fcsel d6, d7, d6, vs",
"str s6, [x8, #40]",
"str x30, [sp, #-16]!",
"mov v0.16b, v2.16b",
@@ -2234,6 +2314,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov d6, d0",
"mov x13, v6.d[0]",
"fcmp d6, d6",
"orr x13, x13, #0x8000000000000",
"fmov d7, x13",
"fcsel d6, d7, d6, vs",
"str d6, [x8]",
"add w20, w20, #0x1 (1)",
"and w20, w20, #0x7",
@@ -2250,7 +2335,7 @@
},
"Block9": {
"x86InstructionCount": 25,
"ExpectedInstructionCount": 72,
"ExpectedInstructionCount": 92,
"x86Insts": [
"fstp st0",
"fstp st3",
@@ -2309,6 +2394,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s5, s0",
"mov x12, v5.d[0]",
"fcmp d5, d5",
"orr x12, x12, #0x8000000000000",
"fmov d6, x12",
"fcsel d5, d6, d5, vs",
"str s5, [x8, #56]",
"strb wzr, [x28, #1017]",
"str x30, [sp, #-16]!",
@@ -2318,6 +2408,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s5, s0",
"mov x12, v5.d[0]",
"fcmp d5, d5",
"orr x12, x12, #0x8000000000000",
"fmov d6, x12",
"fcsel d5, d6, d5, vs",
"str s5, [x8, #44]",
"add w20, w20, #0x1 (1)",
"and w20, w20, #0x7",
@@ -2331,6 +2426,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s6, s0",
"mov x13, v6.d[0]",
"fcmp d6, d6",
"orr x13, x13, #0x8000000000000",
"fmov d7, x13",
"fcsel d6, d7, d6, vs",
"str s6, [x8, #40]",
"str x30, [sp, #-16]!",
"mov v0.16b, v2.16b",
@@ -2339,6 +2439,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov d6, d0",
"mov x13, v6.d[0]",
"fcmp d6, d6",
"orr x13, x13, #0x8000000000000",
"fmov d7, x13",
"fcsel d6, d7, d6, vs",
"str d6, [x8]",
"add w20, w20, #0x1 (1)",
"and w20, w20, #0x7",
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
+24 -4
View File
@@ -2001,7 +2001,7 @@
]
},
"fst dword [rax]": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 16,
"Comment": [
"0xd9 !11b /2"
],
@@ -2016,11 +2016,16 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s2, s0",
"mov x20, v2.d[0]",
"fcmp d2, d2",
"orr x20, x20, #0x8000000000000",
"fmov d3, x20",
"fcsel d2, d3, d2, vs",
"str s2, [x4]"
]
},
"fstp dword [rax]": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 24,
"Comment": [
"0xd9 !11b /3"
],
@@ -2035,6 +2040,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s2, s0",
"mov x21, v2.d[0]",
"fcmp d2, d2",
"orr x21, x21, #0x8000000000000",
"fmov d3, x21",
"fcsel d2, d3, d2, vs",
"str s2, [x4]",
"add w21, w20, #0x1 (1)",
"and w21, w21, #0x7",
@@ -6736,7 +6746,7 @@
]
},
"fst qword [rax]": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 16,
"Comment": [
"0xdd !11b /2"
],
@@ -6751,11 +6761,16 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov d2, d0",
"mov x20, v2.d[0]",
"fcmp d2, d2",
"orr x20, x20, #0x8000000000000",
"fmov d3, x20",
"fcsel d2, d3, d2, vs",
"str d2, [x4]"
]
},
"fstp qword [rax]": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 24,
"Comment": [
"0xdd !11b /3"
],
@@ -6770,6 +6785,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov d2, d0",
"mov x21, v2.d[0]",
"fcmp d2, d2",
"orr x21, x21, #0x8000000000000",
"fmov d3, x21",
"fcsel d2, d3, d2, vs",
"str d2, [x4]",
"add w21, w20, #0x1 (1)",
"and w21, w21, #0x7",
+24 -4
View File
@@ -2000,7 +2000,7 @@
]
},
"fst dword [rax]": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 16,
"Comment": [
"0xd9 !11b /2"
],
@@ -2015,11 +2015,16 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s2, s0",
"mov x20, v2.d[0]",
"fcmp d2, d2",
"orr x20, x20, #0x8000000000000",
"fmov d3, x20",
"fcsel d2, d3, d2, vs",
"str s2, [x4]"
]
},
"fstp dword [rax]": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 24,
"Comment": [
"0xd9 !11b /3"
],
@@ -2034,6 +2039,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov s2, s0",
"mov x21, v2.d[0]",
"fcmp d2, d2",
"orr x21, x21, #0x8000000000000",
"fmov d3, x21",
"fcsel d2, d3, d2, vs",
"str s2, [x4]",
"add w21, w20, #0x1 (1)",
"and w21, w21, #0x7",
@@ -6767,7 +6777,7 @@
]
},
"fst qword [rax]": {
"ExpectedInstructionCount": 11,
"ExpectedInstructionCount": 16,
"Comment": [
"0xdd !11b /2"
],
@@ -6782,11 +6792,16 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov d2, d0",
"mov x20, v2.d[0]",
"fcmp d2, d2",
"orr x20, x20, #0x8000000000000",
"fmov d3, x20",
"fcsel d2, d3, d2, vs",
"str d2, [x4]"
]
},
"fstp qword [rax]": {
"ExpectedInstructionCount": 19,
"ExpectedInstructionCount": 24,
"Comment": [
"0xdd !11b /3"
],
@@ -6801,6 +6816,11 @@
"blr x0",
"ldr x30, [sp], #16",
"fmov d2, d0",
"mov x21, v2.d[0]",
"fcmp d2, d2",
"orr x21, x21, #0x8000000000000",
"fmov d3, x21",
"fcsel d2, d3, d2, vs",
"str d2, [x4]",
"add w21, w20, #0x1 (1)",
"and w21, w21, #0x7",