Merge pull request #3080 from Sonicadvance1/defer_softfloat

FEXCore: Defer setting x87 softflow rounding mode until use
This commit is contained in:
Ryan Houdek authored and GitHub committed 2023-09-15 08:08:04 -07:00
commit 9866e238d5
13 files changed
+2608 -2696

No files matched your search

@@ -14,15 +14,11 @@ $end_info$
namespace FEXCore::CPU {
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
DEF_OP(F80LOADFCW) {
FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle(*GetSrc<uint16_t*>(Data->SSAData, IROp->Args[0]));
}
DEF_OP(F80ADD) {
auto Op = IROp->C<IR::IROp_F80Add>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FADD(Src1, Src2);
const auto Tmp = CPU::OpHandlers<IR::OP_F80ADD>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -31,7 +27,7 @@ DEF_OP(F80SUB) {
auto Op = IROp->C<IR::IROp_F80Sub>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FSUB(Src1, Src2);
const auto Tmp = CPU::OpHandlers<IR::OP_F80SUB>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -40,7 +36,7 @@ DEF_OP(F80MUL) {
auto Op = IROp->C<IR::IROp_F80Mul>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FMUL(Src1, Src2);
const auto Tmp = CPU::OpHandlers<IR::OP_F80MUL>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -49,7 +45,7 @@ DEF_OP(F80DIV) {
auto Op = IROp->C<IR::IROp_F80Div>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FDIV(Src1, Src2);
const auto Tmp = CPU::OpHandlers<IR::OP_F80DIV>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -58,7 +54,7 @@ DEF_OP(F80FYL2X) {
auto Op = IROp->C<IR::IROp_F80FYL2X>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FYL2X(Src1, Src2);
const auto Tmp = CPU::OpHandlers<IR::OP_F80FYL2X>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -67,7 +63,7 @@ DEF_OP(F80ATAN) {
auto Op = IROp->C<IR::IROp_F80ATAN>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FATAN(Src1, Src2);
const auto Tmp = CPU::OpHandlers<IR::OP_F80ATAN>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -76,7 +72,7 @@ DEF_OP(F80FPREM1) {
auto Op = IROp->C<IR::IROp_F80FPREM1>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FREM1(Src1, Src2);
const auto Tmp = CPU::OpHandlers<IR::OP_F80FPREM1>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -85,7 +81,7 @@ DEF_OP(F80FPREM) {
auto Op = IROp->C<IR::IROp_F80FPREM>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FREM(Src1, Src2);
const auto Tmp = CPU::OpHandlers<IR::OP_F80FPREM>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -94,7 +90,7 @@ DEF_OP(F80SCALE) {
auto Op = IROp->C<IR::IROp_F80SCALE>();
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
const auto Tmp = X80SoftFloat::FSCALE(Src1, Src2);
const auto Tmp = CPU::OpHandlers<IR::OP_F80SCALE>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -107,12 +103,12 @@ DEF_OP(F80CVT) {
switch (OpSize) {
case 4: {
float Tmp = Src;
const auto Tmp = CPU::OpHandlers<IR::OP_F80CVT>::handle4(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, OpSize);
break;
}
case 8: {
double Tmp = Src;
const auto Tmp = CPU::OpHandlers<IR::OP_F80CVT>::handle8(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, OpSize);
break;
}
@@ -128,17 +124,17 @@ DEF_OP(F80CVTINT) {
switch (OpSize) {
case 2: {
int16_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2)(Src);
int16_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2)(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(Tmp));
break;
}
case 4: {
int32_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4)(Src);
int32_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4)(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(Tmp));
break;
}
case 8: {
int64_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8)(Src);
int64_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8)(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(Tmp));
break;
}
@@ -152,13 +148,13 @@ DEF_OP(F80CVTTO) {
switch (Op->SrcSize) {
case 4: {
float Src = *GetSrc<float *>(Data->SSAData, Op->X80Src);
X80SoftFloat Tmp = Src;
const auto Tmp = CPU::OpHandlers<IR::OP_F80CVTTO>::handle4(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
case 8: {
double Src = *GetSrc<double *>(Data->SSAData, Op->X80Src);
X80SoftFloat Tmp = Src;
const auto Tmp = CPU::OpHandlers<IR::OP_F80CVTTO>::handle8(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
@@ -172,13 +168,13 @@ DEF_OP(F80CVTTOINT) {
switch (Op->SrcSize) {
case 2: {
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Src);
X80SoftFloat Tmp = Src;
const auto Tmp = CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
case 4: {
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Src);
X80SoftFloat Tmp = Src;
const auto Tmp = CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
break;
}
@@ -189,7 +185,7 @@ DEF_OP(F80CVTTOINT) {
DEF_OP(F80ROUND) {
auto Op = IROp->C<IR::IROp_F80Round>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FRNDINT(Src);
const auto Tmp = CPU::OpHandlers<IR::OP_F80ROUND>::handle(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -197,7 +193,7 @@ DEF_OP(F80ROUND) {
DEF_OP(F80F2XM1) {
auto Op = IROp->C<IR::IROp_F80F2XM1>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::F2XM1(Src);
const auto Tmp = CPU::OpHandlers<IR::OP_F80F2XM1>::handle(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -205,7 +201,7 @@ DEF_OP(F80F2XM1) {
DEF_OP(F80TAN) {
auto Op = IROp->C<IR::IROp_F80TAN>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FTAN(Src);
const auto Tmp = CPU::OpHandlers<IR::OP_F80TAN>::handle(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -213,7 +209,7 @@ DEF_OP(F80TAN) {
DEF_OP(F80SQRT) {
auto Op = IROp->C<IR::IROp_F80SQRT>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FSQRT(Src);
const auto Tmp = CPU::OpHandlers<IR::OP_F80SQRT>::handle(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -221,7 +217,7 @@ DEF_OP(F80SQRT) {
DEF_OP(F80SIN) {
auto Op = IROp->C<IR::IROp_F80SIN>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FSIN(Src);
const auto Tmp = CPU::OpHandlers<IR::OP_F80SIN>::handle(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -229,7 +225,7 @@ DEF_OP(F80SIN) {
DEF_OP(F80COS) {
auto Op = IROp->C<IR::IROp_F80COS>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FCOS(Src);
const auto Tmp = CPU::OpHandlers<IR::OP_F80COS>::handle(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -237,7 +233,7 @@ DEF_OP(F80COS) {
DEF_OP(F80XTRACT_EXP) {
auto Op = IROp->C<IR::IROp_F80XTRACT_EXP>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FXTRACT_EXP(Src);
const auto Tmp = CPU::OpHandlers<IR::OP_F80XTRACT_EXP>::handle(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
@@ -245,102 +241,32 @@ DEF_OP(F80XTRACT_EXP) {
DEF_OP(F80XTRACT_SIG) {
auto Op = IROp->C<IR::IROp_F80XTRACT_SIG>();
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = X80SoftFloat::FXTRACT_SIG(Src);
const auto Tmp = CPU::OpHandlers<IR::OP_F80XTRACT_SIG>::handle(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80CMP) {
auto Op = IROp->C<IR::IROp_F80Cmp>();
uint32_t ResultFlags{};
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
bool eq, lt, nan;
X80SoftFloat::FCMP(Src1, Src2, &eq, &lt, &nan);
if (Op->Flags & (1 << IR::FCMP_FLAG_LT) &&
lt) {
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
}
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
nan) {
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
}
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ) &&
eq) {
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
}
const auto ResultFlags = CPU::OpHandlers<IR::OP_F80CMP>::handle<IR::FCMP_FLAG_LT | IR::FCMP_FLAG_UNORDERED | IR::FCMP_FLAG_EQ>(Data->State->CurrentFrame->State.FCW, Src1, Src2);
GD = ResultFlags;
}
DEF_OP(F80BCDLOAD) {
auto Op = IROp->C<IR::IROp_F80BCDLoad>();
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->X80Src);
uint64_t BCD{};
// We walk through each uint8_t and pull out the BCD encoding
// Each 4bit split is a digit
// Only 0-9 is supported, A-F results in undefined data
// | 4 bit | 4 bit |
// | 10s place | 1s place |
// EG 0x48 = 48
// EG 0x4847 = 4847
// This gives us an 18digit value encoded in BCD
// The last byte lets us know if it negative or not
for (size_t i = 0; i < 9; ++i) {
uint8_t Digit = Src1[8 - i];
// First shift our last value over
BCD *= 100;
// Add the tens place digit
BCD += (Digit >> 4) * 10;
// Add the ones place digit
BCD += Digit & 0xF;
}
// Set negative flag once converted to x87
bool Negative = Src1[9] & 0x80;
X80SoftFloat Tmp;
Tmp = BCD;
Tmp.Sign = Negative;
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F80BCDSTORE) {
auto Op = IROp->C<IR::IROp_F80BCDStore>();
X80SoftFloat Src1 = X80SoftFloat::FRNDINT(*GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src));
bool Negative = Src1.Sign;
// Clear the Sign bit
Src1.Sign = 0;
uint64_t Tmp = Src1;
uint8_t BCD[10]{};
for (size_t i = 0; i < 9; ++i) {
if (Tmp == 0) {
// Nothing left? Just leave
break;
}
// Extract the lower 100 values
uint8_t Digit = Tmp % 100;
// Now divide it for the next iteration
Tmp /= 100;
uint8_t UpperNibble = Digit / 10;
uint8_t LowerNibble = Digit % 10;
// Now store the BCD
BCD[i] = (UpperNibble << 4) | LowerNibble;
}
// Set negative flag once converted to x87
BCD[9] = Negative ? 0x80 : 0;
memcpy(GDP, BCD, 10);
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
const auto Tmp = CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle(Data->State->CurrentFrame->State.FCW, Src);
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
}
DEF_OP(F64SIN) {
@@ -7,13 +7,41 @@
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
namespace FEXCore::CPU {
static void LoadDeferredFCW(uint16_t NewFCW) {
auto PC = (NewFCW >> 8) & 3;
switch(PC) {
case 0: extF80_roundingPrecision = 32; break;
case 2: extF80_roundingPrecision = 64; break;
case 3: extF80_roundingPrecision = 80; break;
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
}
auto RC = (NewFCW >> 10) & 3;
switch(RC) {
case 0:
softfloat_roundingMode = softfloat_round_near_even;
break;
case 1:
softfloat_roundingMode = softfloat_round_min;
break;
case 2:
softfloat_roundingMode = softfloat_round_max;
break;
case 3:
softfloat_roundingMode = softfloat_round_minMag;
break;
}
}
template<>
struct OpHandlers<IR::OP_F80CVTTO> {
static X80SoftFloat handle4(float src) {
static X80SoftFloat handle4(uint16_t NewFCW, float src) {
LoadDeferredFCW(NewFCW);
return src;
}
static X80SoftFloat handle8(double src) {
static X80SoftFloat handle8(uint16_t NewFCW, double src) {
LoadDeferredFCW(NewFCW);
return src;
}
};
@@ -21,7 +49,9 @@ struct OpHandlers<IR::OP_F80CVTTO> {
template<>
struct OpHandlers<IR::OP_F80CMP> {
template<uint32_t Flags>
static uint64_t handle(X80SoftFloat Src1, X80SoftFloat Src2) {
static uint64_t handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
bool eq, lt, nan;
uint64_t ResultFlags = 0;
@@ -44,30 +74,36 @@ struct OpHandlers<IR::OP_F80CMP> {
template<>
struct OpHandlers<IR::OP_F80CVT> {
static float handle4(X80SoftFloat src) {
static float handle4(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
static double handle8(X80SoftFloat src) {
static double handle8(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
};
template<>
struct OpHandlers<IR::OP_F80CVTINT> {
static int16_t handle2(X80SoftFloat src) {
static int16_t handle2(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
static int32_t handle4(X80SoftFloat src) {
static int32_t handle4(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
static int64_t handle8(X80SoftFloat src) {
static int64_t handle8(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return src;
}
static int16_t handle2t(X80SoftFloat src) {
static int16_t handle2t(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
auto rv = extF80_to_i32(src, softfloat_round_minMag, false);
if (rv > INT16_MAX) {
@@ -79,216 +115,243 @@ struct OpHandlers<IR::OP_F80CVTINT> {
}
}
static int32_t handle4t(X80SoftFloat src) {
static int32_t handle4t(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return extF80_to_i32(src, softfloat_round_minMag, false);
}
static int64_t handle8t(X80SoftFloat src) {
static int64_t handle8t(uint16_t NewFCW, X80SoftFloat src) {
LoadDeferredFCW(NewFCW);
return extF80_to_i64(src, softfloat_round_minMag, false);
}
};
template<>
struct OpHandlers<IR::OP_F80CVTTOINT> {
static X80SoftFloat handle2(int16_t src) {
static X80SoftFloat handle2(uint16_t NewFCW, int16_t src) {
LoadDeferredFCW(NewFCW);
return src;
}
static X80SoftFloat handle4(int32_t src) {
static X80SoftFloat handle4(uint16_t NewFCW, int32_t src) {
LoadDeferredFCW(NewFCW);
return src;
}
};
template<>
struct OpHandlers<IR::OP_F80ROUND> {
static X80SoftFloat handle(X80SoftFloat Src1) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FRNDINT(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80F2XM1> {
static X80SoftFloat handle(X80SoftFloat Src1) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::F2XM1(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80TAN> {
static X80SoftFloat handle(X80SoftFloat Src1) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FTAN(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80SQRT> {
static X80SoftFloat handle(X80SoftFloat Src1) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FSQRT(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80SIN> {
static X80SoftFloat handle(X80SoftFloat Src1) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FSIN(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80COS> {
static X80SoftFloat handle(X80SoftFloat Src1) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FCOS(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
static X80SoftFloat handle(X80SoftFloat Src1) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FXTRACT_EXP(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
static X80SoftFloat handle(X80SoftFloat Src1) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FXTRACT_SIG(Src1);
}
};
template<>
struct OpHandlers<IR::OP_F80ADD> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FADD(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80SUB> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FSUB(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80MUL> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FMUL(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80DIV> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FDIV(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80FYL2X> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FYL2X(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80ATAN> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FATAN(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80FPREM1> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FREM1(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80FPREM> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FREM(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F80SCALE> {
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
LoadDeferredFCW(NewFCW);
return X80SoftFloat::FSCALE(Src1, Src2);
}
};
template<>
struct OpHandlers<IR::OP_F64SIN> {
static double handle(double src) {
static double handle(uint16_t NewFCW, double src) {
LoadDeferredFCW(NewFCW);
return sin(src);
}
};
template<>
struct OpHandlers<IR::OP_F64COS> {
static double handle(double src) {
static double handle(uint16_t NewFCW, double src) {
LoadDeferredFCW(NewFCW);
return cos(src);
}
};
template<>
struct OpHandlers<IR::OP_F64TAN> {
static double handle(double src) {
static double handle(uint16_t NewFCW, double src) {
LoadDeferredFCW(NewFCW);
return tan(src);
}
};
template<>
struct OpHandlers<IR::OP_F64F2XM1> {
static double handle(double src) {
static double handle(uint16_t NewFCW, double src) {
LoadDeferredFCW(NewFCW);
return exp2(src) - 1.0;
}
};
template<>
struct OpHandlers<IR::OP_F64ATAN> {
static double handle(double src1, double src2) {
static double handle(uint16_t NewFCW, double src1, double src2) {
LoadDeferredFCW(NewFCW);
return atan2(src1, src2);
}
};
template<>
struct OpHandlers<IR::OP_F64FPREM> {
static double handle(double src1, double src2) {
static double handle(uint16_t NewFCW, double src1, double src2) {
LoadDeferredFCW(NewFCW);
return fmod(src1, src2);
}
};
template<>
struct OpHandlers<IR::OP_F64FPREM1> {
static double handle(double src1, double src2) {
static double handle(uint16_t NewFCW, double src1, double src2) {
LoadDeferredFCW(NewFCW);
return remainder(src1, src2);
}
};
template<>
struct OpHandlers<IR::OP_F64FYL2X> {
static double handle(double src1, double src2) {
static double handle(uint16_t NewFCW, double src1, double src2) {
LoadDeferredFCW(NewFCW);
return src2 * log2(src1);
}
};
template<>
struct OpHandlers<IR::OP_F64SCALE> {
static double handle(double src1, double src2) {
static double handle(uint16_t NewFCW, double src1, double src2) {
LoadDeferredFCW(NewFCW);
double trunc = (double)(int64_t)(src2); //truncate
return src1 * exp2(trunc);
}
};
template<>
struct OpHandlers<IR::OP_F80BCDSTORE> {
static X80SoftFloat handle(X80SoftFloat Src1) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
LoadDeferredFCW(NewFCW);
bool Negative = Src1.Sign;
Src1 = X80SoftFloat::FRNDINT(Src1);
@@ -328,7 +391,8 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
template<>
struct OpHandlers<IR::OP_F80BCDLOAD> {
static X80SoftFloat handle(X80SoftFloat Src) {
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src) {
LoadDeferredFCW(NewFCW);
uint8_t *Src1 = reinterpret_cast<uint8_t *>(&Src);
uint64_t BCD{};
// We walk through each uint8_t and pull out the BCD encoding
@@ -362,34 +426,4 @@ struct OpHandlers<IR::OP_F80BCDLOAD> {
}
};
template<>
struct OpHandlers<IR::OP_F80LOADFCW> {
static void handle(uint16_t NewFCW) {
auto PC = (NewFCW >> 8) & 3;
switch(PC) {
case 0: extF80_roundingPrecision = 32; break;
case 2: extF80_roundingPrecision = 64; break;
case 3: extF80_roundingPrecision = 80; break;
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
}
auto RC = (NewFCW >> 10) & 3;
switch(RC) {
case 0:
softfloat_roundingMode = softfloat_round_near_even;
break;
case 1:
softfloat_roundingMode = softfloat_round_min;
break;
case 2:
softfloat_roundingMode = softfloat_round_max;
break;
case 3:
softfloat_roundingMode = softfloat_round_minMag;
break;
}
}
};
} // namespace FEXCore::CPU
@@ -15,78 +15,73 @@ static FallbackInfo GetFallbackInfo(R(*fn)(Args...), FEXCore::Core::FallbackHand
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F32, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(uint16_t, float), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I16_F32, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F64, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I16_F64, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I16, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(uint16_t, int16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I16_I16, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_VOID_U16, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(uint16_t, int32_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I16_I32, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I32, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(float(*fn)(uint16_t, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F32_I16_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F32_F80, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(double(*fn)(uint16_t, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_I16_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_F80, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(double(*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_I16_F64, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(double(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_F64, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(double(*fn)(uint16_t, double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_I16_F64_F64, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(double(*fn)(double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F64_F64_F64, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(int16_t(*fn)(uint16_t, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I16_I16_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I16_F80, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(int32_t(*fn)(uint16_t, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I32_I16_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I32_F80, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(int64_t(*fn)(uint16_t, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I64_I16_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I64_F80, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(uint64_t(*fn)(uint16_t, X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I64_I16_F80_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_I64_F80_F80, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(uint16_t, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I16_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F80, (void*)fn, HandlerIndex};
}
template<>
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_F80_F80, (void*)fn, HandlerIndex};
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(uint16_t, X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
return {FABI_F80_I16_F80_F80, (void*)fn, HandlerIndex};
}
template<>
@@ -100,7 +95,6 @@ FallbackInfo GetFallbackInfo(uint32_t(*fn)(__uint128_t, __uint128_t, uint16_t),
}
void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
Info[Core::OPINDEX_F80LOADFCW] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW).fn);
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4).fn);
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8).fn);
Info[Core::OPINDEX_F80CVT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4).fn);
@@ -164,11 +158,6 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info) {
uint8_t OpSize = IROp->Size;
switch(IROp->Op) {
case IR::OP_F80LOADFCW: {
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW);
return true;
}
case IR::OP_F80CVTTO: {
auto Op = IROp->C<IR::IROp_F80CVTTo>();
@@ -300,7 +300,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
REGISTER_OP(PCLMUL, PCLMUL);
// F80 ops
REGISTER_OP(F80LOADFCW, F80LOADFCW);
REGISTER_OP(F80ADD, F80ADD);
REGISTER_OP(F80SUB, F80SUB);
REGISTER_OP(F80MUL, F80MUL);
@@ -24,21 +24,20 @@ namespace FEXCore::Core{
namespace FEXCore::CPU {
enum FallbackABI {
FABI_UNKNOWN,
FABI_VOID_U16,
FABI_F80_F32,
FABI_F80_F64,
FABI_F80_I16,
FABI_F80_I32,
FABI_F32_F80,
FABI_F64_F80,
FABI_F64_F64,
FABI_F64_F64_F64,
FABI_I16_F80,
FABI_I32_F80,
FABI_I64_F80,
FABI_I64_F80_F80,
FABI_F80_F80,
FABI_F80_F80_F80,
FABI_F80_I16_F32,
FABI_F80_I16_F64,
FABI_F80_I16_I16,
FABI_F80_I16_I32,
FABI_F32_I16_F80,
FABI_F64_I16_F80,
FABI_F64_I16_F64,
FABI_F64_I16_F64_F64,
FABI_I16_I16_F80,
FABI_I32_I16_F80,
FABI_I64_I16_F80,
FABI_I64_I16_F80_F80,
FABI_F80_I16_F80,
FABI_F80_I16_F80_F80,
FABI_I32_I64_I64_I128_I128_I16,
FABI_I32_I128_I128_I16,
};
+110 -117
View File
@@ -83,37 +83,18 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
#endif
} else {
switch(Info.ABI) {
case FABI_VOID_U16:{
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
const auto Src1 = GetReg(IROp->Args[0].ID());
uxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Src1);
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<void, uint16_t>(ARMEmitter::Reg::r1);
#else
blr(ARMEmitter::Reg::r1);
#endif
PopDynamicRegsAndLR();
FillStaticRegs();
}
break;
case FABI_F80_F32:{
case FABI_F80_I16_F32:{
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
const auto Src1 = GetVReg(IROp->Args[0].ID());
fmov(ARMEmitter::SReg::s0, Src1.S());
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<__uint128_t, float>(ARMEmitter::Reg::r0);
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, float>(ARMEmitter::Reg::r1);
#else
blr(ARMEmitter::Reg::r0);
blr(ARMEmitter::Reg::r1);
#endif
PopDynamicRegsAndLR();
@@ -127,47 +108,17 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
}
break;
case FABI_F80_F64:{
case FABI_F80_I16_F64:{
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
const auto Src1 = GetVReg(IROp->Args[0].ID());
mov(ARMEmitter::DReg::d0, Src1.D());
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<__uint128_t, double>(ARMEmitter::Reg::r0);
#else
blr(ARMEmitter::Reg::r0);
#endif
PopDynamicRegsAndLR();
FillStaticRegs();
const auto Dst = GetVReg(Node);
eor(Dst.Q(), Dst.Q(), Dst.Q());
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
}
break;
case FABI_F80_I16:
case FABI_F80_I32: {
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
const auto Src1 = GetReg(IROp->Args[0].ID());
if (Info.ABI == FABI_F80_I16) {
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Src1);
}
else {
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Src1);
}
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<__uint128_t, uint32_t>(ARMEmitter::Reg::r1);
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, double>(ARMEmitter::Reg::r1);
#else
blr(ARMEmitter::Reg::r1);
#endif
@@ -183,21 +134,54 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
}
break;
case FABI_F32_F80:{
case FABI_F80_I16_I16:
case FABI_F80_I16_I32: {
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
const auto Src1 = GetReg(IROp->Args[0].ID());
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
if (Info.ABI == FABI_F80_I16_I16) {
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
}
else {
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
}
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint32_t>(ARMEmitter::Reg::r2);
#else
blr(ARMEmitter::Reg::r2);
#endif
PopDynamicRegsAndLR();
FillStaticRegs();
const auto Dst = GetVReg(Node);
eor(Dst.Q(), Dst.Q(), Dst.Q());
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
}
break;
case FABI_F32_I16_F80:{
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
const auto Src1 = GetVReg(IROp->Args[0].ID());
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<float, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
GenerateIndirectRuntimeCall<float, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
#else
blr(ARMEmitter::Reg::r2);
blr(ARMEmitter::Reg::r3);
#endif
PopDynamicRegsAndLR();
@@ -209,21 +193,22 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
}
break;
case FABI_F64_F80:{
case FABI_F64_I16_F80:{
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
const auto Src1 = GetVReg(IROp->Args[0].ID());
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<double, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
GenerateIndirectRuntimeCall<double, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
#else
blr(ARMEmitter::Reg::r2);
blr(ARMEmitter::Reg::r3);
#endif
PopDynamicRegsAndLR();
@@ -235,7 +220,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
}
break;
case FABI_F64_F64: {
case FABI_F64_I16_F64: {
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
@@ -243,11 +228,12 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
const auto Src1 = GetVReg(IROp->Args[0].ID());
mov(ARMEmitter::DReg::d0, Src1.D());
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<double, double>(ARMEmitter::Reg::r0);
GenerateIndirectRuntimeCall<double, uint16_t, double>(ARMEmitter::Reg::r1);
#else
blr(ARMEmitter::Reg::r0);
blr(ARMEmitter::Reg::r1);
#endif
PopDynamicRegsAndLR();
@@ -259,7 +245,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
}
break;
case FABI_F64_F64_F64: {
case FABI_F64_I16_F64_F64: {
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
@@ -269,11 +255,12 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
mov(ARMEmitter::DReg::d0, Src1.D());
mov(ARMEmitter::DReg::d1, Src2.D());
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<double, double, double>(ARMEmitter::Reg::r0);
GenerateIndirectRuntimeCall<double, uint16_t, double, double>(ARMEmitter::Reg::r1);
#else
blr(ARMEmitter::Reg::r0);
blr(ARMEmitter::Reg::r1);
#endif
PopDynamicRegsAndLR();
@@ -285,21 +272,22 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
}
break;
case FABI_I16_F80:{
case FABI_I16_I16_F80:{
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
const auto Src1 = GetVReg(IROp->Args[0].ID());
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
#else
blr(ARMEmitter::Reg::r2);
blr(ARMEmitter::Reg::r3);
#endif
PopDynamicRegsAndLR();
@@ -310,21 +298,22 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
sxth(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
}
break;
case FABI_I32_F80:{
case FABI_I32_I16_F80:{
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
const auto Src1 = GetVReg(IROp->Args[0].ID());
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
#else
blr(ARMEmitter::Reg::r2);
blr(ARMEmitter::Reg::r3);
#endif
PopDynamicRegsAndLR();
@@ -335,21 +324,22 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
mov(ARMEmitter::Size::i32Bit, Dst, ARMEmitter::Reg::r0);
}
break;
case FABI_I64_F80:{
case FABI_I64_I16_F80:{
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
const auto Src1 = GetVReg(IROp->Args[0].ID());
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
#else
blr(ARMEmitter::Reg::r2);
blr(ARMEmitter::Reg::r3);
#endif
PopDynamicRegsAndLR();
@@ -360,7 +350,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
mov(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
}
break;
case FABI_I64_F80_F80:{
case FABI_I64_I16_F80_F80:{
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
@@ -368,17 +358,18 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
const auto Src1 = GetVReg(IROp->Args[0].ID());
const auto Src2 = GetVReg(IROp->Args[1].ID());
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r3, Src2, 4);
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
ldr(ARMEmitter::XReg::x4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
#else
blr(ARMEmitter::Reg::r4);
blr(ARMEmitter::Reg::r5);
#endif
PopDynamicRegsAndLR();
@@ -388,21 +379,22 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
mov(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
}
break;
case FABI_F80_F80:{
case FABI_F80_I16_F80:{
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
const auto Src1 = GetVReg(IROp->Args[0].ID());
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<__uint128_t, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
#else
blr(ARMEmitter::Reg::r2);
blr(ARMEmitter::Reg::r3);
#endif
PopDynamicRegsAndLR();
@@ -415,7 +407,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
}
break;
case FABI_F80_F80_F80:{
case FABI_F80_I16_F80_F80:{
SpillStaticRegs(TMP1);
PushDynamicRegsAndLR(TMP1);
@@ -423,17 +415,18 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
const auto Src1 = GetVReg(IROp->Args[0].ID());
const auto Src2 = GetVReg(IROp->Args[1].ID());
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r3, Src2, 4);
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
ldr(ARMEmitter::XReg::x4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
#ifdef VIXL_SIMULATOR
GenerateIndirectRuntimeCall<__uint128_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
#else
blr(ARMEmitter::Reg::r4);
blr(ARMEmitter::Reg::r5);
#endif
PopDynamicRegsAndLR();
@@ -104,18 +104,11 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
#endif
} else {
switch(Info.ABI) {
case FABI_VOID_U16: {
PushRegs();
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
break;
}
case FABI_F80_F32:{
case FABI_F80_I16_F32:{
PushRegs();
movss(xmm0, GetSrc(IROp->Args[0].ID()));
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -126,10 +119,11 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
}
break;
case FABI_F80_F64:{
case FABI_F80_I16_F64:{
PushRegs();
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
PopRegs();
@@ -140,15 +134,17 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
}
break;
case FABI_F80_I16:
case FABI_F80_I32: {
case FABI_F80_I16_I16:
case FABI_F80_I16_I32: {
PushRegs();
if (Info.ABI == FABI_F80_I16) {
movsx(rdi, GetSrc<RA_32>(IROp->Args[0].ID()).cvt16());
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
if (Info.ABI == FABI_F80_I16_I16) {
movsx(rsi, GetSrc<RA_32>(IROp->Args[0].ID()).cvt16());
}
else {
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
mov(esi, GetSrc<RA_32>(IROp->Args[0].ID()));
}
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
@@ -160,11 +156,12 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
}
break;
case FABI_F32_F80:{
case FABI_F32_I16_F80:{
PushRegs();
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
movq(rsi, GetSrc(IROp->Args[0].ID()));
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
@@ -174,11 +171,12 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
}
break;
case FABI_F64_F80:{
case FABI_F64_I16_F80:{
PushRegs();
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
movq(rsi, GetSrc(IROp->Args[0].ID()));
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
@@ -188,9 +186,10 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
}
break;
case FABI_F64_F64: {
case FABI_F64_I16_F64: {
PushRegs();
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
@@ -201,9 +200,10 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
}
break;
case FABI_F64_F64_F64: {
case FABI_F64_I16_F64_F64: {
PushRegs();
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
movsd(xmm1, GetSrc(IROp->Args[1].ID()));
@@ -215,11 +215,12 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
}
break;
case FABI_I16_F80:{
case FABI_I16_I16_F80:{
PushRegs();
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
movq(rsi, GetSrc(IROp->Args[0].ID()));
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
@@ -228,11 +229,12 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
movsx(GetDst<RA_64>(Node), ax);
}
break;
case FABI_I32_F80:{
case FABI_I32_I16_F80:{
PushRegs();
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
movq(rsi, GetSrc(IROp->Args[0].ID()));
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
@@ -241,11 +243,12 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
mov(GetDst<RA_32>(Node), eax);
}
break;
case FABI_I64_F80:{
case FABI_I64_I16_F80:{
PushRegs();
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
movq(rsi, GetSrc(IROp->Args[0].ID()));
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
@@ -254,14 +257,15 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
mov(GetDst<RA_64>(Node), rax);
}
break;
case FABI_I64_F80_F80:{
case FABI_I64_I16_F80_F80:{
PushRegs();
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
movq(rsi, GetSrc(IROp->Args[0].ID()));
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
movq(rdx, GetSrc(IROp->Args[1].ID()));
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
movq(rcx, GetSrc(IROp->Args[1].ID()));
pextrq(r8, GetSrc(IROp->Args[1].ID()), 1);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
@@ -270,11 +274,12 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
mov(GetDst<RA_64>(Node), rax);
}
break;
case FABI_F80_F80:{
case FABI_F80_I16_F80:{
PushRegs();
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
movq(rsi, GetSrc(IROp->Args[0].ID()));
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
@@ -285,14 +290,15 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
pinsrw(GetDst(Node), edx, 4);
}
break;
case FABI_F80_F80_F80:{
case FABI_F80_I16_F80_F80:{
PushRegs();
movq(rdi, GetSrc(IROp->Args[0].ID()));
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
movq(rsi, GetSrc(IROp->Args[0].ID()));
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
movq(rdx, GetSrc(IROp->Args[1].ID()));
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
movq(rcx, GetSrc(IROp->Args[1].ID()));
pextrq(r8, GetSrc(IROp->Args[1].ID()), 1);
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
@@ -2971,7 +2971,6 @@ void OpDispatchBuilder::RestoreX87State(OrderedNode *MemBase) {
const auto OpSize = IR::SizeToOpSize(CTX->GetGPRSize());
auto NewFCW = _LoadMem(GPRClass, 2, MemBase, 2);
_F80LoadFCW(NewFCW);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
{
@@ -694,7 +694,6 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
auto Zero = _Constant(0);
// Init FCW to 0x037F
auto NewFCW = _Constant(16, 0x037F);
_F80LoadFCW(NewFCW);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
// Init FSW to 0
@@ -1015,7 +1014,6 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
Mem = AppendSegmentOffset(Mem, Op->Flags);
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
_F80LoadFCW(NewFCW);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
@@ -1098,7 +1096,6 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
_F80LoadFCW(NewFCW);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
}
@@ -1216,7 +1213,6 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
Mem = AppendSegmentOffset(Mem, Op->Flags);
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
_F80LoadFCW(NewFCW);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
@@ -46,7 +46,6 @@ class OrderedNode;
void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
// Init FCW to 0x037F
auto NewFCW = _Constant(16, 0x037F);
_F80LoadFCW(NewFCW);
// Init host rounding mode to zero
auto Zero = _Constant(0);
_SetRoundingMode(Zero);
@@ -78,8 +77,6 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
roundingMode = _And(OpSize::i32Bit, roundingMode, roundMask);
_SetRoundingMode(roundingMode);
_F80LoadFCW(NewFCW);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
@@ -96,7 +93,6 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
_F80LoadFCW(NewFCW); //keeps BCD code working
//ignore the rounding precision, we're always 64-bit in F64.
//extract rounding mode
OrderedNode *roundingMode = NewFCW;
@@ -1026,7 +1022,6 @@ void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
roundingMode = _And(OpSize::i32Bit, roundingMode, roundMask);
_SetRoundingMode(roundingMode);
_F80LoadFCW(NewFCW);
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
-3
View File
@@ -1912,9 +1912,6 @@
}
},
"F80": {
"F80LoadFCW GPR:$Src": {
"HasSideEffects": true
},
"FPR = F80Add FPR:$X80Src1, FPR:$X80Src2": {
"DestSize": "16"
},
@@ -1039,52 +1039,11 @@
]
},
"fxrstor [rax]": {
"ExpectedInstructionCount": 105,
"ExpectedInstructionCount": 64,
"Optimal": "No",
"Comment": "GROUP15 0x0F 0xAE /1",
"ExpectedArm64ASM": [
"ldrh w20, [x4]",
"stp x4, x5, [x28, #8]",
"stp x6, x7, [x28, #24]",
"stp x8, x9, [x28, #40]",
"stp x10, x11, [x28, #56]",
"stp x12, x13, [x28, #72]",
"stp x14, x15, [x28, #88]",
"stp x16, x17, [x28, #104]",
"stp x19, x29, [x28, #120]",
"add x0, x28, #0xc0 (192)",
"st1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x0], #64",
"st1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x0], #64",
"st1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x0], #64",
"st1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x0], #64",
"sub sp, sp, #0xf0 (240)",
"mov x0, sp",
"st1 {v2.2d, v3.2d}, [x0], #32",
"st1 {v4.2d, v5.2d, v6.2d, v7.2d}, [x0], #64",
"st1 {v8.2d, v9.2d, v10.2d, v11.2d}, [x0], #64",
"st1 {v12.2d, v13.2d, v14.2d, v15.2d}, [x0], #64",
"str x30, [x0]",
"uxth w0, w20",
"ldr x1, [x28, #1144]",
"blr x1",
"ld1 {v2.2d, v3.2d}, [sp], #32",
"ld1 {v4.2d, v5.2d, v6.2d, v7.2d}, [sp], #64",
"ld1 {v8.2d, v9.2d, v10.2d, v11.2d}, [sp], #64",
"ld1 {v12.2d, v13.2d, v14.2d, v15.2d}, [sp], #64",
"ldr x30, [sp], #16",
"add x5, x28, #0xc0 (192)",
"ld1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x5], #64",
"ld1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x5], #64",
"ld1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x5], #64",
"ld1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x5], #64",
"ldp x4, x5, [x28, #8]",
"ldp x6, x7, [x28, #24]",
"ldp x8, x9, [x28, #40]",
"ldp x10, x11, [x28, #56]",
"ldp x12, x13, [x28, #72]",
"ldp x14, x15, [x28, #88]",
"ldp x16, x17, [x28, #104]",
"ldp x19, x29, [x28, #120]",
"strh w20, [x28, #1008]",
"ldrh w20, [x4, #2]",
"ubfx w21, w20, #11, #3",
@@ -1350,7 +1309,7 @@
]
},
"xrstor [rax]": {
"ExpectedInstructionCount": 194,
"ExpectedInstructionCount": 112,
"Optimal": "No",
"Comment": "GROUP15 0x0F 0xAE /5",
"ExpectedArm64ASM": [
@@ -1358,49 +1317,8 @@
"ldr x21, [x20, #512]",
"ubfx x22, x21, #0, #1",
"cbnz x22, #+0x8",
"b #+0x128",
"b #+0x84",
"ldrh w22, [x20]",
"stp x4, x5, [x28, #8]",
"stp x6, x7, [x28, #24]",
"stp x8, x9, [x28, #40]",
"stp x10, x11, [x28, #56]",
"stp x12, x13, [x28, #72]",
"stp x14, x15, [x28, #88]",
"stp x16, x17, [x28, #104]",
"stp x19, x29, [x28, #120]",
"add x0, x28, #0xc0 (192)",
"st1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x0], #64",
"st1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x0], #64",
"st1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x0], #64",
"st1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x0], #64",
"sub sp, sp, #0xf0 (240)",
"mov x0, sp",
"st1 {v2.2d, v3.2d}, [x0], #32",
"st1 {v4.2d, v5.2d, v6.2d, v7.2d}, [x0], #64",
"st1 {v8.2d, v9.2d, v10.2d, v11.2d}, [x0], #64",
"st1 {v12.2d, v13.2d, v14.2d, v15.2d}, [x0], #64",
"str x30, [x0]",
"uxth w0, w22",
"ldr x1, [x28, #1144]",
"blr x1",
"ld1 {v2.2d, v3.2d}, [sp], #32",
"ld1 {v4.2d, v5.2d, v6.2d, v7.2d}, [sp], #64",
"ld1 {v8.2d, v9.2d, v10.2d, v11.2d}, [sp], #64",
"ld1 {v12.2d, v13.2d, v14.2d, v15.2d}, [sp], #64",
"ldr x30, [sp], #16",
"add x5, x28, #0xc0 (192)",
"ld1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x5], #64",
"ld1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x5], #64",
"ld1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x5], #64",
"ld1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x5], #64",
"ldp x4, x5, [x28, #8]",
"ldp x6, x7, [x28, #24]",
"ldp x8, x9, [x28, #40]",
"ldp x10, x11, [x28, #56]",
"ldp x12, x13, [x28, #72]",
"ldp x14, x15, [x28, #88]",
"ldp x16, x17, [x28, #104]",
"ldp x19, x29, [x28, #120]",
"strh w22, [x28, #1008]",
"ldrh w22, [x20, #2]",
"ubfx w23, w22, #11, #3",
@@ -1431,50 +1349,9 @@
"str q2, [x28, #848]",
"ldr q2, [x20, #144]",
"str q2, [x28, #864]",
"b #+0xf0",
"b #+0x4c",
"mov w22, #0x0",
"mov w23, #0x37f",
"stp x4, x5, [x28, #8]",
"stp x6, x7, [x28, #24]",
"stp x8, x9, [x28, #40]",
"stp x10, x11, [x28, #56]",
"stp x12, x13, [x28, #72]",
"stp x14, x15, [x28, #88]",
"stp x16, x17, [x28, #104]",
"stp x19, x29, [x28, #120]",
"add x0, x28, #0xc0 (192)",
"st1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x0], #64",
"st1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x0], #64",
"st1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x0], #64",
"st1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x0], #64",
"sub sp, sp, #0xf0 (240)",
"mov x0, sp",
"st1 {v2.2d, v3.2d}, [x0], #32",
"st1 {v4.2d, v5.2d, v6.2d, v7.2d}, [x0], #64",
"st1 {v8.2d, v9.2d, v10.2d, v11.2d}, [x0], #64",
"st1 {v12.2d, v13.2d, v14.2d, v15.2d}, [x0], #64",
"str x30, [x0]",
"uxth w0, w23",
"ldr x1, [x28, #1144]",
"blr x1",
"ld1 {v2.2d, v3.2d}, [sp], #32",
"ld1 {v4.2d, v5.2d, v6.2d, v7.2d}, [sp], #64",
"ld1 {v8.2d, v9.2d, v10.2d, v11.2d}, [sp], #64",
"ld1 {v12.2d, v13.2d, v14.2d, v15.2d}, [sp], #64",
"ldr x30, [sp], #16",
"add x5, x28, #0xc0 (192)",
"ld1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x5], #64",
"ld1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x5], #64",
"ld1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x5], #64",
"ld1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x5], #64",
"ldp x4, x5, [x28, #8]",
"ldp x6, x7, [x28, #24]",
"ldp x8, x9, [x28, #40]",
"ldp x10, x11, [x28, #56]",
"ldp x12, x13, [x28, #72]",
"ldp x14, x15, [x28, #88]",
"ldp x16, x17, [x28, #104]",
"ldp x19, x29, [x28, #120]",
"strh w23, [x28, #1008]",
"strb w22, [x28, #747]",
"strb w22, [x28, #744]",
File diff suppressed because it is too large. Load diff