diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp index d8820abd5..1243f7642 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp @@ -860,22 +860,103 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) { void OpDispatchBuilder::X87FXAM(OpcodeArgs) { auto a = _ReadStackValue(0); - Ref Result = - ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1); + Ref Value = ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1); - // Extract the sign bit - Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Result) : _Bfe(OpSize::i64Bit, 1, 15, Result); + // Extract the sign bit, which goes in C1 + Ref Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Value) : _Bfe(OpSize::i64Bit, 1, 15, Value); SetRFLAG(Result); - // Claim this is a normal number - // We don't support anything else - auto TopValid = _StackValidTag(0); + auto NotEmpty = _StackValidTag(0); + Ref IsEmpty = _Xor(OpSize::i64Bit, NotEmpty, Constant(1)); + Ref IsNaN {}; + Ref IsDenormal {}; + Ref IsInf {}; + Ref IsZero {}; + Ref IsUnsupported {}; + Ref NoSignBit {}; - // In the case of top being invalid then C3:C2:C0 is 0b101 - auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1)); + // TODO: The codegen for this is not optimal, and can probably be improved + // if FXAM ends up on the hot path for some workload. + + if (ReducedPrecisionMode) { + constexpr uint64_t ExponentMask = 0x7FF0'0000'0000'0000ULL; + NoSignBit = _Bfe(OpSize::i64Bit, 63, 0, Value); + IsInf = Select01(OpSize::i64Bit, CondClass::EQ, NoSignBit, Constant(ExponentMask)); + IsNaN = Select01(OpSize::i64Bit, CondClass::UGT, NoSignBit, Constant(ExponentMask)); + + IsZero = Select01(OpSize::i64Bit, CondClass::EQ, NoSignBit, Constant(0)); + // 64 bit floats can't represent an x87 denormal, nor any of the + // unsupported encodings. + IsDenormal = Constant(0); + IsUnsupported = Constant(0); + } else { + Ref Mantissa = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 0); + + // "J" is the name given to the msb of the mantissa in the SDM. + Ref JBit = _Bfe(OpSize::i64Bit, 1, 63, Mantissa); + Ref Exponent = _Bfe(OpSize::i64Bit, 15, 0, Value); + Ref IsExponentZero = Select01(OpSize::i64Bit, CondClass::EQ, Exponent, Constant(0)); + Ref IsExponentMax = Select01(OpSize::i64Bit, CondClass::EQ, Exponent, Constant(0x7FFF)); + + // Inf is when mantissa only has the J bit set, exponent is all 1's. + Ref IsOnlyJBit = Select01(OpSize::i64Bit, CondClass::EQ, Mantissa, Constant(1ULL << 63)); + IsInf = _And(OpSize::i64Bit, IsExponentMax, IsOnlyJBit); + + // NaN is when the low 63 bits of the mantissa are non-zero + // and exponent is max, and the J bit is set. + Ref Fraction = _Bfe(OpSize::i64Bit, 63, 0, Mantissa); + Ref FractionNonZero = Select01(OpSize::i64Bit, CondClass::NEQ, Fraction, Constant(0)); + Ref IsExponentMaxWithJBit = _And(OpSize::i64Bit, IsExponentMax, JBit); + IsNaN = _And(OpSize::i64Bit, IsExponentMaxWithJBit, FractionNonZero); + + // Zero and Denormal are basically the same as the 64-bit case. + Ref MantissaNonZero = Select01(OpSize::i64Bit, CondClass::NEQ, Mantissa, Constant(0)); + Ref MantissaZero = _Xor(OpSize::i64Bit, MantissaNonZero, Constant(1)); + IsZero = _And(OpSize::i64Bit, IsExponentZero, MantissaZero); + IsDenormal = _And(OpSize::i64Bit, IsExponentZero, MantissaNonZero); + + // This is where things are weird. If the J bit is not set + // and the exponent is non-zero, then this is an "unsupported" + // encoding, which I believe is left in for legacy reasons. + Ref IsSupported = _Or(OpSize::i64Bit, IsExponentZero, JBit); + IsUnsupported = _Xor(OpSize::i64Bit, IsSupported, Constant(1)); + } + + // NormalFiniteNumber = !Zero && !Denormal && !Inf && !NaN && !Empty && !Unsupported + Ref temp1 = _Or(OpSize::i64Bit, IsZero, IsDenormal); + Ref temp2 = _Or(OpSize::i64Bit, IsInf, IsNaN); + Ref temp3 = _Or(OpSize::i64Bit, IsUnsupported, IsEmpty); + temp1 = _Or(OpSize::i64Bit, temp1, temp2); + temp2 = _Or(OpSize::i64Bit, temp1, temp3); + Ref NormalFiniteNumber = _Xor(OpSize::i64Bit, temp2, Constant(1)); + + // Set C3, C2, C0 based on the class of the FP value in ST(0) + // Table is from "FXAM" page in the SDM. + // +----------------------+----+----+----+ + // | Class | C3 | C2 | C0 | + // +----------------------+----+----+----+ + // | Unsupported | 0 | 0 | 0 | + // | NaN | 0 | 0 | 1 | + // | Normal finite number | 0 | 1 | 0 | + // | Infinity | 0 | 1 | 1 | + // | Zero | 1 | 0 | 0 | + // | Empty | 1 | 0 | 1 | + // | Denormal number | 1 | 1 | 0 | + // +----------------------+----+----+----+ + + // c0 = IsNaN || IsInf || IsEmpty + // c2 = (IsInf || Denormal || NormalFiniteNumber) && !IsEmpty + // c3 = Zero || IsEmpty || Denormal + Ref C0 = _Or(OpSize::i64Bit, IsNaN, IsInf); + C0 = _Or(OpSize::i64Bit, C0, IsEmpty); + + Ref C2 = _Or(OpSize::i64Bit, IsInf, IsDenormal); + C2 = _Or(OpSize::i64Bit, C2, NormalFiniteNumber); + C2 = _And(OpSize::i64Bit, C2, NotEmpty); + + Ref C3 = _Or(OpSize::i64Bit, IsZero, IsEmpty); + C3 = _Or(OpSize::i64Bit, C3, IsDenormal); - auto C2 = TopValid; - auto C0 = C3; // Mirror C3 until something other than zero is supported SetRFLAG(C0); SetRFLAG(C2); SetRFLAG(C3); diff --git a/unittests/ASM/Known_Failures_host b/unittests/ASM/Known_Failures_host index 2158fda4e..2e985db75 100644 --- a/unittests/ASM/Known_Failures_host +++ b/unittests/ASM/Known_Failures_host @@ -1,3 +1,2 @@ -Test_X87/FXAM_Simple.asm ## Tag bits not completely modelled Test_X87/X87MMXInteraction.asm \ No newline at end of file diff --git a/unittests/ASM/X87/FXAM_Simple.asm b/unittests/ASM/X87/FXAM_Simple.asm index 7e76fbaaa..c0c7635de 100644 --- a/unittests/ASM/X87/FXAM_Simple.asm +++ b/unittests/ASM/X87/FXAM_Simple.asm @@ -1,13 +1,15 @@ ;; Simpler versions of FXAM_Push* tests. -;; In hostrunner tests this will fail because we mentioned below there's no support -;; for the zero flag. In hostrunner RCX should contain 0x4000 instead of 0x400. %ifdef CONFIG { "RegData": { "RAX": "0x6", "RBX": "0x0400", - "RCX": "0x0400", - "RDX": "0x4100" + "RCX": "0x4000", + "RDX": "0x4100", + "R8": "0x4200440044000700", + "R9": "0x0100030000000000", + "R10": "0x0000000004000400", + "R11": "0x0000050006004400" } } %endif @@ -21,15 +23,51 @@ fxam fwait fnstsw ax -and ax, 0x4500 ; should be 0x4100 for zero +and ax, 0x4500 ; should be 0x4100 for empty mov edx, eax +;; Get the C0 - C3 flags and +;; pack up to 4 results into a GPR +%macro examine_packed 2 + fld tword [rel %2] + fxam + fwait + fnstsw ax + and eax, 0x4700 + shl %1, 16 + or %1, rax + fstp st0 +%endmacro + +xor r8d, r8d +examine_packed r8, .data_neg_zero +examine_packed r8, .data_denormal +examine_packed r8, .data_pseudo_denormal +examine_packed r8, .data_neg_infinity + +xor r9d, r9d +examine_packed r9, .data_qnan +examine_packed r9, .data_neg_snan +examine_packed r9, .data_unnormal +examine_packed r9, .data_pseudo_infinity + +xor r10d, r10d +examine_packed r10, .data_pseudo_nan +examine_packed r10, .data_pseudo_zero +examine_packed r10, .data_min_exponent +examine_packed r10, .data_max_exponent + +xor r11d, r11d +examine_packed r11, .data_infinity +examine_packed r11, .data_neg_one +examine_packed r11, .data_pseudo_denormal_fraction + fldz fxam fwait fnstsw ax -and ax, 0x4500 ; should be 0x4000 for zero, but there's no support for it at the moment, so it'll return 0x0400 as it does for a normal number. +and ax, 0x4500 ; should be 0x4000 for zero mov ecx, eax fld1 @@ -47,3 +85,64 @@ and eax, 0x7 hlt + +align 16 +.data_neg_zero: +dq 0 +dw 0x8000 + +.data_denormal: +dq 1 +dw 0x0000 + +.data_pseudo_denormal: +dq (1 << 63) +dw 0x0000 + +.data_neg_infinity: +dq (1 << 63) +dw 0xFFFF + +.data_qnan: +dq (11b << 62) +dw 0x7FFF + +.data_neg_snan: +dq (10b << 62) | 1 +dw 0xFFFF + +.data_unnormal: +dq (1 << 62) +dw 0x3FFF + +.data_pseudo_infinity: +dq 0 +dw 0x7FFF + +.data_pseudo_nan: +dq (1 << 62) +dw 0x7FFF + +.data_pseudo_zero: +dq 0 +dw 0x3FFF + +.data_min_exponent: +dq (1 << 63) +dw 0x0001 + +.data_max_exponent: +dq (1 << 63) +dw 0x7FFE + +.data_infinity: +dq (1 << 63) +dw 0x7FFF + +.data_neg_one: +dq (1 << 63) +dw 0xBFFF + +.data_pseudo_denormal_fraction: +dq (1 << 63) | 1 +dw 0x0000