Merge pull request #6002 from javelina-pkwy/fix/fxam

X87: Implement full FXAM support
This commit is contained in:
Ryan Houdek authored and GitHub committed 2026-10-02 16:33:07 -07:00
commit 7325cb09be
110 files changed
+498 -171

No files matched your search

+1 -1
View File
@@ -249,7 +249,7 @@ namespace DiskCache {
// The current version of the diskcache.
// This must be changed any time codegen changes occur!
// Be aware of the impact of changing this frequently!
static constexpr uint16_t FormatVersion = 28;
static constexpr uint16_t FormatVersion = 29;
static constexpr uint32_t LOOKUP_KEY_MAX_BUCKET_DEPTH = 500;
@@ -860,22 +860,103 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
auto a = _ReadStackValue(0);
Ref Result =
ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1);
Ref Value = ReducedPrecisionMode ? _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, a, 0) : _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 1);
// Extract the sign bit
Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Result) : _Bfe(OpSize::i64Bit, 1, 15, Result);
// Extract the sign bit, which goes in C1
Ref Result = ReducedPrecisionMode ? _Bfe(OpSize::i64Bit, 1, 63, Value) : _Bfe(OpSize::i64Bit, 1, 15, Value);
SetRFLAG<FEXCore::X86State::X87FLAG_C1_LOC>(Result);
// Claim this is a normal number
// We don't support anything else
auto TopValid = _StackValidTag(0);
auto NotEmpty = _StackValidTag(0);
Ref IsEmpty = _Xor(OpSize::i64Bit, NotEmpty, Constant(1));
Ref IsNaN {};
Ref IsDenormal {};
Ref IsInf {};
Ref IsZero {};
Ref IsUnsupported {};
Ref NoSignBit {};
// In the case of top being invalid then C3:C2:C0 is 0b101
auto C3 = Select01(OpSize::i32Bit, CondClass::NEQ, TopValid, Constant(1));
// TODO: The codegen for this is not optimal, and can probably be improved
// if FXAM ends up on the hot path for some workload.
if (ReducedPrecisionMode) {
constexpr uint64_t ExponentMask = 0x7FF0'0000'0000'0000ULL;
NoSignBit = _Bfe(OpSize::i64Bit, 63, 0, Value);
IsInf = Select01(OpSize::i64Bit, CondClass::EQ, NoSignBit, Constant(ExponentMask));
IsNaN = Select01(OpSize::i64Bit, CondClass::UGT, NoSignBit, Constant(ExponentMask));
IsZero = Select01(OpSize::i64Bit, CondClass::EQ, NoSignBit, Constant(0));
// 64 bit floats can't represent an x87 denormal, nor any of the
// unsupported encodings.
IsDenormal = Constant(0);
IsUnsupported = Constant(0);
} else {
Ref Mantissa = _VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, a, 0);
// "J" is the name given to the msb of the mantissa in the SDM.
Ref JBit = _Bfe(OpSize::i64Bit, 1, 63, Mantissa);
Ref Exponent = _Bfe(OpSize::i64Bit, 15, 0, Value);
Ref IsExponentZero = Select01(OpSize::i64Bit, CondClass::EQ, Exponent, Constant(0));
Ref IsExponentMax = Select01(OpSize::i64Bit, CondClass::EQ, Exponent, Constant(0x7FFF));
// Inf is when mantissa only has the J bit set, exponent is all 1's.
Ref IsOnlyJBit = Select01(OpSize::i64Bit, CondClass::EQ, Mantissa, Constant(1ULL << 63));
IsInf = _And(OpSize::i64Bit, IsExponentMax, IsOnlyJBit);
// NaN is when the low 63 bits of the mantissa are non-zero
// and exponent is max, and the J bit is set.
Ref Fraction = _Bfe(OpSize::i64Bit, 63, 0, Mantissa);
Ref FractionNonZero = Select01(OpSize::i64Bit, CondClass::NEQ, Fraction, Constant(0));
Ref IsExponentMaxWithJBit = _And(OpSize::i64Bit, IsExponentMax, JBit);
IsNaN = _And(OpSize::i64Bit, IsExponentMaxWithJBit, FractionNonZero);
// Zero and Denormal are basically the same as the 64-bit case.
Ref MantissaNonZero = Select01(OpSize::i64Bit, CondClass::NEQ, Mantissa, Constant(0));
Ref MantissaZero = _Xor(OpSize::i64Bit, MantissaNonZero, Constant(1));
IsZero = _And(OpSize::i64Bit, IsExponentZero, MantissaZero);
IsDenormal = _And(OpSize::i64Bit, IsExponentZero, MantissaNonZero);
// This is where things are weird. If the J bit is not set
// and the exponent is non-zero, then this is an "unsupported"
// encoding, which I believe is left in for legacy reasons.
Ref IsSupported = _Or(OpSize::i64Bit, IsExponentZero, JBit);
IsUnsupported = _Xor(OpSize::i64Bit, IsSupported, Constant(1));
}
// NormalFiniteNumber = !Zero && !Denormal && !Inf && !NaN && !Empty && !Unsupported
Ref temp1 = _Or(OpSize::i64Bit, IsZero, IsDenormal);
Ref temp2 = _Or(OpSize::i64Bit, IsInf, IsNaN);
Ref temp3 = _Or(OpSize::i64Bit, IsUnsupported, IsEmpty);
temp1 = _Or(OpSize::i64Bit, temp1, temp2);
temp2 = _Or(OpSize::i64Bit, temp1, temp3);
Ref NormalFiniteNumber = _Xor(OpSize::i64Bit, temp2, Constant(1));
// Set C3, C2, C0 based on the class of the FP value
// Table is from "FXAM" page in the SDM.
// +----------------------+----+----+----+
// | Class | C3 | C2 | C0 |
// +----------------------+----+----+----+
// | Unsupported | 0 | 0 | 0 |
// | NaN | 0 | 0 | 1 |
// | Normal finite number | 0 | 1 | 0 |
// | Infinity | 0 | 1 | 1 |
// | Zero | 1 | 0 | 0 |
// | Empty | 1 | 0 | 1 |
// | Denormal number | 1 | 1 | 0 |
// +----------------------+----+----+----+
// C0 = IsNaN || IsInf || IsEmpty
Ref C0 = _Or(OpSize::i64Bit, IsNaN, IsInf);
C0 = _Or(OpSize::i64Bit, C0, IsEmpty);
// C2 = (IsInf || Denormal || NormalFiniteNumber) && !IsEmpty
Ref C2 = _Or(OpSize::i64Bit, IsInf, IsDenormal);
C2 = _Or(OpSize::i64Bit, C2, NormalFiniteNumber);
C2 = _And(OpSize::i64Bit, C2, NotEmpty);
// C3 = Zero || IsEmpty || Denormal
Ref C3 = _Or(OpSize::i64Bit, IsZero, IsEmpty);
C3 = _Or(OpSize::i64Bit, C3, IsDenormal);
auto C2 = TopValid;
auto C0 = C3; // Mirror C3 until something other than zero is supported
SetRFLAG<FEXCore::X86State::X87FLAG_C0_LOC>(C0);
SetRFLAG<FEXCore::X86State::X87FLAG_C2_LOC>(C2);
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(C3);
-1
View File
@@ -1,3 +1,2 @@
Test_X87/FXAM_Simple.asm
## Tag bits not completely modelled
Test_X87/X87MMXInteraction.asm
+105 -6
View File
@@ -1,13 +1,15 @@
;; Simpler versions of FXAM_Push* tests.
;; In hostrunner tests this will fail because we mentioned below there's no support
;; for the zero flag. In hostrunner RCX should contain 0x4000 instead of 0x400.
%ifdef CONFIG
{
"RegData": {
"RAX": "0x6",
"RBX": "0x0400",
"RCX": "0x0400",
"RDX": "0x4100"
"RCX": "0x4000",
"RDX": "0x4100",
"R8": "0x4200440044000700",
"R9": "0x0100030000000000",
"R10": "0x0000000004000400",
"R11": "0x0000050006004400"
}
}
%endif
@@ -21,15 +23,51 @@ fxam
fwait
fnstsw ax
and ax, 0x4500 ; should be 0x4100 for zero
and ax, 0x4500 ; should be 0x4100 for empty
mov edx, eax
;; Get the C0 - C3 flags and
;; pack up to 4 results into a GPR
%macro examine_packed 2
fld tword [rel %2]
fxam
fwait
fnstsw ax
and eax, 0x4700
shl %1, 16
or %1, rax
fstp st0
%endmacro
xor r8d, r8d
examine_packed r8, .data_neg_zero
examine_packed r8, .data_denormal
examine_packed r8, .data_pseudo_denormal
examine_packed r8, .data_neg_infinity
xor r9d, r9d
examine_packed r9, .data_qnan
examine_packed r9, .data_neg_snan
examine_packed r9, .data_unnormal
examine_packed r9, .data_pseudo_infinity
xor r10d, r10d
examine_packed r10, .data_pseudo_nan
examine_packed r10, .data_pseudo_zero
examine_packed r10, .data_min_exponent
examine_packed r10, .data_max_exponent
xor r11d, r11d
examine_packed r11, .data_infinity
examine_packed r11, .data_neg_one
examine_packed r11, .data_pseudo_denormal_fraction
fldz
fxam
fwait
fnstsw ax
and ax, 0x4500 ; should be 0x4000 for zero, but there's no support for it at the moment, so it'll return 0x0400 as it does for a normal number.
and ax, 0x4500 ; should be 0x4000 for zero
mov ecx, eax
fld1
@@ -47,3 +85,64 @@ and eax, 0x7
hlt
align 16
.data_neg_zero:
dq 0
dw 0x8000
.data_denormal:
dq 1
dw 0x0000
.data_pseudo_denormal:
dq (1 << 63)
dw 0x0000
.data_neg_infinity:
dq (1 << 63)
dw 0xFFFF
.data_qnan:
dq (11b << 62)
dw 0x7FFF
.data_neg_snan:
dq (10b << 62) | 1
dw 0xFFFF
.data_unnormal:
dq (1 << 62)
dw 0x3FFF
.data_pseudo_infinity:
dq 0
dw 0x7FFF
.data_pseudo_nan:
dq (1 << 62)
dw 0x7FFF
.data_pseudo_zero:
dq 0
dw 0x3FFF
.data_min_exponent:
dq (1 << 63)
dw 0x0001
.data_max_exponent:
dq (1 << 63)
dw 0x7FFE
.data_infinity:
dq (1 << 63)
dw 0x7FFF
.data_neg_one:
dq (1 << 63)
dw 0xBFFF
.data_pseudo_denormal_fraction:
dq (1 << 63) | 1
dw 0x0000
+1 -1
View File
@@ -8,7 +8,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"roundss xmm0, xmm1, 00000000b": {
@@ -7,7 +7,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"cvtpi2ps xmm0, [rax]": {
@@ -9,7 +9,7 @@
"DisabledHostFeatures": [
"RPRES"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"cvtsi2ss xmm0, eax": {
@@ -7,7 +7,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"cvtsi2sd xmm0, eax": {
@@ -8,7 +8,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"cvtpi2ps xmm0, [rax]": {
@@ -9,7 +9,7 @@
"SVE256",
"RPRES"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"cvtsi2ss xmm0, eax": {
@@ -8,7 +8,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"cvtsi2sd xmm0, eax": {
@@ -9,7 +9,7 @@
"DisabledHostFeatures": [
"RPRES"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vsqrtss xmm0, xmm1, xmm2": {
@@ -7,7 +7,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vroundss xmm0, xmm1, 00000000b": {
@@ -7,7 +7,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vfmaddsubps xmm0, xmm1, xmm2, xmm3": {
@@ -13,7 +13,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vmovups xmm0, xmm0": {
@@ -9,7 +9,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vaddsubpd xmm0, xmm1, xmm2": {
@@ -10,7 +10,7 @@
"FLAGM2",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vmovntps [rax], xmm0": {
@@ -10,7 +10,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vucomiss xmm0, xmm1": {
@@ -11,7 +11,7 @@
"I8MM",
"DOTPROD"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vpshufb xmm0, xmm1, xmm2": {
@@ -8,7 +8,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vfmadd132ss xmm0, xmm1, xmm2": {
@@ -12,7 +12,7 @@
"SVE256",
"I8MM"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vpdpbusd xmm0, xmm1, xmm2": {
@@ -11,7 +11,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vpdpbusd xmm0, xmm1, xmm2": {
@@ -10,7 +10,7 @@
"FLAGM2",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vmovntdqa xmm0, [rax]": {
@@ -11,7 +11,7 @@
"SVE256",
"SVEBITPERM"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vtestps xmm0, xmm1": {
@@ -7,7 +7,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vpermq ymm0, ymm1, 1": {
@@ -8,7 +8,7 @@
"AFP",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vblendvps xmm0, xmm1, xmm2, xmm3": {
@@ -9,7 +9,7 @@
"SVE256",
"SVE128"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vpsrlw xmm0, xmm1, 0": {
+1 -1
View File
@@ -9,7 +9,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"lock add byte [rax], cl": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"sha1nexte xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"sha1nexte xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"pclmulqdq xmm0, xmm1, 00000b": {
+1 -1
View File
@@ -10,7 +10,7 @@
"AFP",
"RPRES"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Comment": [
"These 3DNow! instructions are optimal assuming that FEX doesn't SRA MMX registers",
@@ -8,7 +8,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Comment": [
"Instructions that explicitly push against the limits of ARM's loadstore instructions"
@@ -8,7 +8,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Comment": [
"Instructions that explicitly push against the limits of ARM's loadstore instructions"
@@ -13,7 +13,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -11,7 +11,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -9,7 +9,7 @@
"SVE256",
"RPRES"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -14,7 +14,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -14,7 +14,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Comment": [
"These are instruction combinations that could be more optimal if FEX optimized for them"
@@ -12,7 +12,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Comment": [],
"Instructions": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"lock add byte [rax], cl": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"Chained add": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"ptest xmm0, xmm1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"The Witcher 3": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"Sonic Mania movie player": {
@@ -10,7 +10,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"FMOD scalar loop": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"The Sims 1 hot block": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"add bl, cl": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"add al, 1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"push es": {
@@ -11,7 +11,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"ucomiss xmm0, xmm1": {
@@ -11,7 +11,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"sgdt [rax]": {
@@ -11,7 +11,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"xgetbv": {
@@ -11,7 +11,7 @@
"AFP",
"FCMA"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"ucomisd xmm0, xmm1": {
@@ -12,7 +12,7 @@
"AFP",
"CSSC"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"popcnt ax, bx": {
@@ -12,7 +12,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"popcnt ax, bx": {
@@ -12,7 +12,7 @@
"RPRES",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vucomiss xmm0, xmm1": {
@@ -10,7 +10,7 @@
"AFP",
"SVEBITPERM"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vtestps xmm0, xmm1": {
@@ -9,7 +9,7 @@
"DisabledHostFeatures": [
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"blsr eax, ebx": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"Block1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"Block1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"Block1": {
@@ -10,7 +10,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"Block1": {
+57 -13
View File
@@ -11,7 +11,7 @@
"CSSC",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"fadd dword [rax]": {
@@ -2607,27 +2607,71 @@
]
},
"fxam": {
"ExpectedInstructionCount": 16,
"ExpectedInstructionCount": 60,
"Comment": [
"0xd9 11b 0xe5 /4"
],
"ExpectedArm64ASM": [
"sub sp, sp, #0x60 (96)",
"ldrb w20, [x28, #1051]",
"add x21, x28, x20, lsl #4",
"ldr q2, [x21, #1056]",
"mov x21, v2.d[1]",
"ubfx x21, x21, #15, #1",
"strb w21, [x28, #1049]",
"ldrb w21, [x28, #1202]",
"lsr w20, w21, w20",
"ubfx x22, x21, #15, #1",
"strb w22, [x28, #1049]",
"ldrb w22, [x28, #1202]",
"lsr w20, w22, w20",
"and w20, w20, #0x1",
"mrs x21, nzcv",
"cmp w20, #0x1 (1)",
"cset x22, ne",
"strb w22, [x28, #1048]",
"strb w20, [x28, #1050]",
"strb w22, [x28, #1054]",
"msr nzcv, x21"
"eor x22, x20, #0x1",
"mov x23, v2.d[0]",
"lsr x24, x23, #63",
"ubfx x21, x21, #0, #15",
"mrs x30, nzcv",
"cmp x21, #0x0 (0)",
"cset x18, eq",
"str w30, [sp]",
"mov w30, #0x7fff",
"cmp x21, x30",
"cset x21, eq",
"mov x30, #0x8000000000000000",
"cmp x23, x30",
"cset x30, eq",
"and x30, x21, x30",
"str w20, [sp, #32]",
"ubfx x20, x23, #0, #63",
"cmp x20, #0x0 (0)",
"cset x20, ne",
"and x21, x21, x24",
"and x20, x21, x20",
"cmp x23, #0x0 (0)",
"cset x21, ne",
"eor x23, x21, #0x1",
"and x23, x18, x23",
"and x21, x18, x21",
"orr x24, x18, x24",
"eor x24, x24, #0x1",
"orr x18, x23, x21",
"str x23, [sp, #64]",
"orr x23, x30, x20",
"orr x24, x24, x22",
"orr x23, x18, x23",
"orr x23, x23, x24",
"eor x23, x23, #0x1",
"orr x20, x20, x30",
"orr x20, x20, x22",
"orr x24, x30, x21",
"orr x23, x24, x23",
"ldr w24, [sp, #32]",
"and x23, x23, x24",
"ldr x24, [sp, #64]",
"orr x22, x24, x22",
"orr x21, x22, x21",
"strb w20, [x28, #1048]",
"strb w23, [x28, #1050]",
"strb w21, [x28, #1054]",
"ldr w20, [sp]",
"msr nzcv, x20",
"add sp, sp, #0x60 (96)"
]
},
"fld1": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"Block1": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"Block1": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"Block1": {
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"Block1": {
+42 -12
View File
@@ -13,7 +13,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"fadd dword [rax]": {
@@ -2006,27 +2006,57 @@
]
},
"fxam": {
"ExpectedInstructionCount": 16,
"ExpectedInstructionCount": 46,
"Comment": [
"0xd9 11b 0xe5 /4"
],
"ExpectedArm64ASM": [
"sub sp, sp, #0x60 (96)",
"ldrb w20, [x28, #1051]",
"add x21, x28, x20, lsl #4",
"ldr d2, [x21, #1056]",
"mov x21, v2.d[0]",
"lsr x21, x21, #63",
"strb w21, [x28, #1049]",
"ldrb w21, [x28, #1202]",
"lsr w20, w21, w20",
"lsr x22, x21, #63",
"strb w22, [x28, #1049]",
"ldrb w22, [x28, #1202]",
"lsr w20, w22, w20",
"and w20, w20, #0x1",
"mrs x21, nzcv",
"cmp w20, #0x1 (1)",
"cset x22, ne",
"strb w22, [x28, #1048]",
"eor x22, x20, #0x1",
"ubfx x21, x21, #0, #63",
"mov x23, #0x7ff0000000000000",
"mrs x24, nzcv",
"cmp x21, x23",
"cset x30, eq",
"cmp x21, x23",
"cset x23, hi",
"mov w18, #0x0",
"cmp x21, #0x0 (0)",
"cset x21, eq",
"str w24, [sp]",
"orr x24, x21, x18",
"orr x18, x30, x23",
"str x21, [sp, #32]",
"mov w21, #0x0",
"str w20, [sp, #64]",
"orr x20, x21, x22",
"orr x24, x24, x18",
"orr x20, x24, x20",
"eor x20, x20, #0x1",
"orr x23, x23, x30",
"orr x23, x23, x22",
"orr x24, x30, x21",
"orr x20, x24, x20",
"ldr w24, [sp, #64]",
"and x20, x20, x24",
"ldr x24, [sp, #32]",
"orr x22, x24, x22",
"orr x21, x22, x21",
"strb w23, [x28, #1048]",
"strb w20, [x28, #1050]",
"strb w22, [x28, #1054]",
"msr nzcv, x21"
"strb w21, [x28, #1054]",
"ldr w20, [sp]",
"msr nzcv, x20",
"add sp, sp, #0x60 (96)"
]
},
"fld1": {
+1 -1
View File
@@ -10,7 +10,7 @@
"FLAGM2",
"CRYPTO"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"pshufb mm0, mm1": {
+1 -1
View File
@@ -8,7 +8,7 @@
"AFP",
"CRYPTO"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Comment": [
"SSE4.2 string instructions are skipped here.",
@@ -8,7 +8,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"dpps xmm0, xmm1, 00000000b": {
@@ -11,7 +11,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"rep movsb": {
+1 -1
View File
@@ -10,7 +10,7 @@
"FLAGM2",
"MOPS"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"add bl, cl": {
@@ -9,7 +9,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Comment": [
"Instructions in this table that are marked optimal don't have their flag calculation part of this assumption",
@@ -9,7 +9,7 @@
"FlagM",
"FlagM2"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"push es": {
+1 -1
View File
@@ -9,7 +9,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"pfrcpv mm0, mm1": {
@@ -9,7 +9,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"rsqrtps xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"rsqrtss xmm0, xmm1": {
@@ -8,7 +8,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vrsqrtps xmm0, xmm1": {
+1 -1
View File
@@ -6,7 +6,7 @@
"SVE128",
"SVE256"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {}
}
@@ -33,7 +33,7 @@
" - 1b: ECX = MSB",
"[7] - Reserved"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"pcmpestrm xmm0, xmm1, 0_0_00_00_00b": {
+1 -1
View File
@@ -12,7 +12,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Comment": [
"MMX instructions are defined as optimal without SRA being used for these instructions.",
@@ -11,7 +11,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"sgdt [rax]": {
@@ -11,7 +11,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"xgetbv": {
@@ -7,7 +7,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"push fs": {
@@ -12,7 +12,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"movupd xmm0, xmm0": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"addsubpd xmm0, xmm1": {
@@ -8,7 +8,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"psrlw xmm0, xmm1": {
@@ -7,7 +7,7 @@
"AFP"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"pmulhuw xmm0, xmm1": {
@@ -12,7 +12,7 @@
"FRINTTS",
"CSSC"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"movss xmm0, xmm1": {
@@ -10,7 +10,7 @@
"FCMA",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"movsd xmm0, xmm1": {
@@ -9,7 +9,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"addsubps xmm0, xmm1": {
@@ -10,7 +10,7 @@
"FCMA",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"cvtpd2dq xmm0, xmm1": {
@@ -12,7 +12,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"cvttss2si eax, xmm0": {
@@ -8,7 +8,7 @@
"SVE256",
"AFP"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"movmskps eax, xmm0": {
+1 -1
View File
@@ -13,7 +13,7 @@
"FLAGM2",
"FRINTTS"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vmovups xmm0, xmm0": {
@@ -8,7 +8,7 @@
"FCMA"
],
"DisabledHostFeatures": [],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vaddsubpd xmm0, xmm1, xmm2": {
@@ -13,7 +13,7 @@
"FLAGM",
"FLAGM2"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vcvttss2si eax, xmm0": {
+1 -1
View File
@@ -13,7 +13,7 @@
"I8MM",
"DOTPROD"
],
"BinaryCacheVersion": 28
"BinaryCacheVersion": 29
},
"Instructions": {
"vpshufb xmm0, xmm1, xmm2": {
Loaded 100 of 110 files, more files were not shown because too many files have changed in this diff. Show more