mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 10:00:16 +02:00
x87: complete FXAM implementation
This commit is contained in:
1 parent
0df84d3844
commit
6788bcf264
3 files changed
+197
-18
No files matched your search
@@ -860,22 +860,103 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
auto a = _ReadStackValue(0);
|
||||
Ref Result =
|
||||
ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1);
|
||||
Ref Value = ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1);
|
||||
|
||||
// Extract the sign bit
|
||||
Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Result) : _Bfe(OpSize::i64Bit, 1, 15, Result);
|
||||
// Extract the sign bit, which goes in C1
|
||||
Ref Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Value) : _Bfe(OpSize::i64Bit, 1, 15, Value);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Result);
|
||||
|
||||
// Claim this is a normal number
|
||||
// We don't support anything else
|
||||
auto TopValid = _StackValidTag(0);
|
||||
auto NotEmpty = _StackValidTag(0);
|
||||
Ref IsEmpty = _Xor(OpSize::i64Bit, NotEmpty, Constant(1));
|
||||
Ref IsNaN {};
|
||||
Ref IsDenormal {};
|
||||
Ref IsInf {};
|
||||
Ref IsZero {};
|
||||
Ref IsUnsupported {};
|
||||
Ref NoSignBit {};
|
||||
|
||||
// In the case of top being invalid then C3:C2:C0 is 0b101
|
||||
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1));
|
||||
// TODO: The codegen for this is not optimal, and can probably be improved
|
||||
// if FXAM ends up on the hot path for some workload.
|
||||
|
||||
if (ReducedPrecisionMode) {
|
||||
constexpr uint64_t ExponentMask = 0x7FF0'0000'0000'0000ULL;
|
||||
NoSignBit = _Bfe(OpSize::i64Bit, 63, 0, Value);
|
||||
IsInf = Select01(OpSize::i64Bit, CondClass::EQ, NoSignBit, Constant(ExponentMask));
|
||||
IsNaN = Select01(OpSize::i64Bit, CondClass::UGT, NoSignBit, Constant(ExponentMask));
|
||||
|
||||
IsZero = Select01(OpSize::i64Bit, CondClass::EQ, NoSignBit, Constant(0));
|
||||
// 64 bit floats can't represent an x87 denormal, nor any of the
|
||||
// unsupported encodings.
|
||||
IsDenormal = Constant(0);
|
||||
IsUnsupported = Constant(0);
|
||||
} else {
|
||||
Ref Mantissa = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 0);
|
||||
|
||||
// "J" is the name given to the msb of the mantissa in the SDM.
|
||||
Ref JBit = _Bfe(OpSize::i64Bit, 1, 63, Mantissa);
|
||||
Ref Exponent = _Bfe(OpSize::i64Bit, 15, 0, Value);
|
||||
Ref IsExponentZero = Select01(OpSize::i64Bit, CondClass::EQ, Exponent, Constant(0));
|
||||
Ref IsExponentMax = Select01(OpSize::i64Bit, CondClass::EQ, Exponent, Constant(0x7FFF));
|
||||
|
||||
// Inf is when mantissa only has the J bit set, exponent is all 1's.
|
||||
Ref IsOnlyJBit = Select01(OpSize::i64Bit, CondClass::EQ, Mantissa, Constant(1ULL << 63));
|
||||
IsInf = _And(OpSize::i64Bit, IsExponentMax, IsOnlyJBit);
|
||||
|
||||
// NaN is when the low 63 bits of the mantissa are non-zero
|
||||
// and exponent is max, and the J bit is set.
|
||||
Ref Fraction = _Bfe(OpSize::i64Bit, 63, 0, Mantissa);
|
||||
Ref FractionNonZero = Select01(OpSize::i64Bit, CondClass::NEQ, Fraction, Constant(0));
|
||||
Ref IsExponentMaxWithJBit = _And(OpSize::i64Bit, IsExponentMax, JBit);
|
||||
IsNaN = _And(OpSize::i64Bit, IsExponentMaxWithJBit, FractionNonZero);
|
||||
|
||||
// Zero and Denormal are basically the same as the 64-bit case.
|
||||
Ref MantissaNonZero = Select01(OpSize::i64Bit, CondClass::NEQ, Mantissa, Constant(0));
|
||||
Ref MantissaZero = _Xor(OpSize::i64Bit, MantissaNonZero, Constant(1));
|
||||
IsZero = _And(OpSize::i64Bit, IsExponentZero, MantissaZero);
|
||||
IsDenormal = _And(OpSize::i64Bit, IsExponentZero, MantissaNonZero);
|
||||
|
||||
// This is where things are weird. If the J bit is not set
|
||||
// and the exponent is non-zero, then this is an "unsupported"
|
||||
// encoding, which I believe is left in for legacy reasons.
|
||||
Ref IsSupported = _Or(OpSize::i64Bit, IsExponentZero, JBit);
|
||||
IsUnsupported = _Xor(OpSize::i64Bit, IsSupported, Constant(1));
|
||||
}
|
||||
|
||||
// NormalFiniteNumber = !Zero && !Denormal && !Inf && !NaN && !Empty && !Unsupported
|
||||
Ref temp1 = _Or(OpSize::i64Bit, IsZero, IsDenormal);
|
||||
Ref temp2 = _Or(OpSize::i64Bit, IsInf, IsNaN);
|
||||
Ref temp3 = _Or(OpSize::i64Bit, IsUnsupported, IsEmpty);
|
||||
temp1 = _Or(OpSize::i64Bit, temp1, temp2);
|
||||
temp2 = _Or(OpSize::i64Bit, temp1, temp3);
|
||||
Ref NormalFiniteNumber = _Xor(OpSize::i64Bit, temp2, Constant(1));
|
||||
|
||||
// Set C3, C2, C0 based on the class of the FP value in ST(0)
|
||||
// Table is from "FXAM" page in the SDM.
|
||||
// +----------------------+----+----+----+
|
||||
// | Class | C3 | C2 | C0 |
|
||||
// +----------------------+----+----+----+
|
||||
// | Unsupported | 0 | 0 | 0 |
|
||||
// | NaN | 0 | 0 | 1 |
|
||||
// | Normal finite number | 0 | 1 | 0 |
|
||||
// | Infinity | 0 | 1 | 1 |
|
||||
// | Zero | 1 | 0 | 0 |
|
||||
// | Empty | 1 | 0 | 1 |
|
||||
// | Denormal number | 1 | 1 | 0 |
|
||||
// +----------------------+----+----+----+
|
||||
|
||||
// c0 = IsNaN || IsInf || IsEmpty
|
||||
// c2 = (IsInf || Denormal || NormalFiniteNumber) && !IsEmpty
|
||||
// c3 = Zero || IsEmpty || Denormal
|
||||
Ref C0 = _Or(OpSize::i64Bit, IsNaN, IsInf);
|
||||
C0 = _Or(OpSize::i64Bit, C0, IsEmpty);
|
||||
|
||||
Ref C2 = _Or(OpSize::i64Bit, IsInf, IsDenormal);
|
||||
C2 = _Or(OpSize::i64Bit, C2, NormalFiniteNumber);
|
||||
C2 = _And(OpSize::i64Bit, C2, NotEmpty);
|
||||
|
||||
Ref C3 = _Or(OpSize::i64Bit, IsZero, IsEmpty);
|
||||
C3 = _Or(OpSize::i64Bit, C3, IsDenormal);
|
||||
|
||||
auto C2 = TopValid;
|
||||
auto C0 = C3; // Mirror C3 until something other than zero is supported
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(C0);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(C2);
|
||||
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(C3);
|
||||
|
||||
@@ -1,3 +1,2 @@
|
||||
Test_X87/FXAM_Simple.asm
|
||||
## Tag bits not completely modelled
|
||||
Test_X87/X87MMXInteraction.asm
|
||||
@@ -1,13 +1,15 @@
|
||||
;; Simpler versions of FXAM_Push* tests.
|
||||
;; In hostrunner tests this will fail because we mentioned below there's no support
|
||||
;; for the zero flag. In hostrunner RCX should contain 0x4000 instead of 0x400.
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0x6",
|
||||
"RBX": "0x0400",
|
||||
"RCX": "0x0400",
|
||||
"RDX": "0x4100"
|
||||
"RCX": "0x4000",
|
||||
"RDX": "0x4100",
|
||||
"R8": "0x4200440044000700",
|
||||
"R9": "0x0100030000000000",
|
||||
"R10": "0x0000000004000400",
|
||||
"R11": "0x0000050006004400"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
@@ -21,15 +23,51 @@ fxam
|
||||
fwait
|
||||
|
||||
fnstsw ax
|
||||
and ax, 0x4500 ; should be 0x4100 for zero
|
||||
and ax, 0x4500 ; should be 0x4100 for empty
|
||||
mov edx, eax
|
||||
|
||||
;; Get the C0 - C3 flags and
|
||||
;; pack up to 4 results into a GPR
|
||||
%macro examine_packed 2
|
||||
fld tword [rel %2]
|
||||
fxam
|
||||
fwait
|
||||
fnstsw ax
|
||||
and eax, 0x4700
|
||||
shl %1, 16
|
||||
or %1, rax
|
||||
fstp st0
|
||||
%endmacro
|
||||
|
||||
xor r8d, r8d
|
||||
examine_packed r8, .data_neg_zero
|
||||
examine_packed r8, .data_denormal
|
||||
examine_packed r8, .data_pseudo_denormal
|
||||
examine_packed r8, .data_neg_infinity
|
||||
|
||||
xor r9d, r9d
|
||||
examine_packed r9, .data_qnan
|
||||
examine_packed r9, .data_neg_snan
|
||||
examine_packed r9, .data_unnormal
|
||||
examine_packed r9, .data_pseudo_infinity
|
||||
|
||||
xor r10d, r10d
|
||||
examine_packed r10, .data_pseudo_nan
|
||||
examine_packed r10, .data_pseudo_zero
|
||||
examine_packed r10, .data_min_exponent
|
||||
examine_packed r10, .data_max_exponent
|
||||
|
||||
xor r11d, r11d
|
||||
examine_packed r11, .data_infinity
|
||||
examine_packed r11, .data_neg_one
|
||||
examine_packed r11, .data_pseudo_denormal_fraction
|
||||
|
||||
fldz
|
||||
fxam
|
||||
fwait
|
||||
|
||||
fnstsw ax
|
||||
and ax, 0x4500 ; should be 0x4000 for zero, but there's no support for it at the moment, so it'll return 0x0400 as it does for a normal number.
|
||||
and ax, 0x4500 ; should be 0x4000 for zero
|
||||
mov ecx, eax
|
||||
|
||||
fld1
|
||||
@@ -47,3 +85,64 @@ and eax, 0x7
|
||||
|
||||
|
||||
hlt
|
||||
|
||||
align 16
|
||||
.data_neg_zero:
|
||||
dq 0
|
||||
dw 0x8000
|
||||
|
||||
.data_denormal:
|
||||
dq 1
|
||||
dw 0x0000
|
||||
|
||||
.data_pseudo_denormal:
|
||||
dq (1 << 63)
|
||||
dw 0x0000
|
||||
|
||||
.data_neg_infinity:
|
||||
dq (1 << 63)
|
||||
dw 0xFFFF
|
||||
|
||||
.data_qnan:
|
||||
dq (11b << 62)
|
||||
dw 0x7FFF
|
||||
|
||||
.data_neg_snan:
|
||||
dq (10b << 62) | 1
|
||||
dw 0xFFFF
|
||||
|
||||
.data_unnormal:
|
||||
dq (1 << 62)
|
||||
dw 0x3FFF
|
||||
|
||||
.data_pseudo_infinity:
|
||||
dq 0
|
||||
dw 0x7FFF
|
||||
|
||||
.data_pseudo_nan:
|
||||
dq (1 << 62)
|
||||
dw 0x7FFF
|
||||
|
||||
.data_pseudo_zero:
|
||||
dq 0
|
||||
dw 0x3FFF
|
||||
|
||||
.data_min_exponent:
|
||||
dq (1 << 63)
|
||||
dw 0x0001
|
||||
|
||||
.data_max_exponent:
|
||||
dq (1 << 63)
|
||||
dw 0x7FFE
|
||||
|
||||
.data_infinity:
|
||||
dq (1 << 63)
|
||||
dw 0x7FFF
|
||||
|
||||
.data_neg_one:
|
||||
dq (1 << 63)
|
||||
dw 0xBFFF
|
||||
|
||||
.data_pseudo_denormal_fraction:
|
||||
dq (1 << 63) | 1
|
||||
dw 0x0000
|
||||
Reference in new issue
Block a user