mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 10:00:16 +02:00
Merge pull request #3080 from Sonicadvance1/defer_softfloat
FEXCore: Defer setting x87 softflow rounding mode until use
This commit is contained in:
13 files changed
+2608
-2696
No files matched your search
@@ -14,15 +14,11 @@ $end_info$
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
#define DEF_OP(x) void InterpreterOps::Op_##x(IR::IROp_Header *IROp, IROpData *Data, IR::NodeID Node)
|
||||
DEF_OP(F80LOADFCW) {
|
||||
FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle(*GetSrc<uint16_t*>(Data->SSAData, IROp->Args[0]));
|
||||
}
|
||||
|
||||
DEF_OP(F80ADD) {
|
||||
auto Op = IROp->C<IR::IROp_F80Add>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FADD(Src1, Src2);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80ADD>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -31,7 +27,7 @@ DEF_OP(F80SUB) {
|
||||
auto Op = IROp->C<IR::IROp_F80Sub>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FSUB(Src1, Src2);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80SUB>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -40,7 +36,7 @@ DEF_OP(F80MUL) {
|
||||
auto Op = IROp->C<IR::IROp_F80Mul>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FMUL(Src1, Src2);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80MUL>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -49,7 +45,7 @@ DEF_OP(F80DIV) {
|
||||
auto Op = IROp->C<IR::IROp_F80Div>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FDIV(Src1, Src2);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80DIV>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -58,7 +54,7 @@ DEF_OP(F80FYL2X) {
|
||||
auto Op = IROp->C<IR::IROp_F80FYL2X>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FYL2X(Src1, Src2);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80FYL2X>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -67,7 +63,7 @@ DEF_OP(F80ATAN) {
|
||||
auto Op = IROp->C<IR::IROp_F80ATAN>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FATAN(Src1, Src2);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80ATAN>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -76,7 +72,7 @@ DEF_OP(F80FPREM1) {
|
||||
auto Op = IROp->C<IR::IROp_F80FPREM1>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FREM1(Src1, Src2);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80FPREM1>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -85,7 +81,7 @@ DEF_OP(F80FPREM) {
|
||||
auto Op = IROp->C<IR::IROp_F80FPREM>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FREM(Src1, Src2);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80FPREM>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -94,7 +90,7 @@ DEF_OP(F80SCALE) {
|
||||
auto Op = IROp->C<IR::IROp_F80SCALE>();
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
const auto Tmp = X80SoftFloat::FSCALE(Src1, Src2);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80SCALE>::handle(Data->State->CurrentFrame->State.FCW, Src1, Src2);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -107,12 +103,12 @@ DEF_OP(F80CVT) {
|
||||
|
||||
switch (OpSize) {
|
||||
case 4: {
|
||||
float Tmp = Src;
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80CVT>::handle4(Data->State->CurrentFrame->State.FCW, Src);
|
||||
memcpy(GDP, &Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
double Tmp = Src;
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80CVT>::handle8(Data->State->CurrentFrame->State.FCW, Src);
|
||||
memcpy(GDP, &Tmp, OpSize);
|
||||
break;
|
||||
}
|
||||
@@ -128,17 +124,17 @@ DEF_OP(F80CVTINT) {
|
||||
|
||||
switch (OpSize) {
|
||||
case 2: {
|
||||
int16_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2)(Src);
|
||||
int16_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle2)(Data->State->CurrentFrame->State.FCW, Src);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
int32_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4)(Src);
|
||||
int32_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle4)(Data->State->CurrentFrame->State.FCW, Src);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
int64_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8)(Src);
|
||||
int64_t Tmp = (Op->Truncate? FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8t : FEXCore::CPU::OpHandlers<IR::OP_F80CVTINT>::handle8)(Data->State->CurrentFrame->State.FCW, Src);
|
||||
memcpy(GDP, &Tmp, sizeof(Tmp));
|
||||
break;
|
||||
}
|
||||
@@ -152,13 +148,13 @@ DEF_OP(F80CVTTO) {
|
||||
switch (Op->SrcSize) {
|
||||
case 4: {
|
||||
float Src = *GetSrc<float *>(Data->SSAData, Op->X80Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80CVTTO>::handle4(Data->State->CurrentFrame->State.FCW, Src);
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
case 8: {
|
||||
double Src = *GetSrc<double *>(Data->SSAData, Op->X80Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80CVTTO>::handle8(Data->State->CurrentFrame->State.FCW, Src);
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
@@ -172,13 +168,13 @@ DEF_OP(F80CVTTOINT) {
|
||||
switch (Op->SrcSize) {
|
||||
case 2: {
|
||||
int16_t Src = *GetSrc<int16_t*>(Data->SSAData, Op->Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle2(Data->State->CurrentFrame->State.FCW, Src);
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
case 4: {
|
||||
int32_t Src = *GetSrc<int32_t*>(Data->SSAData, Op->Src);
|
||||
X80SoftFloat Tmp = Src;
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80CVTTOINT>::handle4(Data->State->CurrentFrame->State.FCW, Src);
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
break;
|
||||
}
|
||||
@@ -189,7 +185,7 @@ DEF_OP(F80CVTTOINT) {
|
||||
DEF_OP(F80ROUND) {
|
||||
auto Op = IROp->C<IR::IROp_F80Round>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FRNDINT(Src);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80ROUND>::handle(Data->State->CurrentFrame->State.FCW, Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -197,7 +193,7 @@ DEF_OP(F80ROUND) {
|
||||
DEF_OP(F80F2XM1) {
|
||||
auto Op = IROp->C<IR::IROp_F80F2XM1>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::F2XM1(Src);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80F2XM1>::handle(Data->State->CurrentFrame->State.FCW, Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -205,7 +201,7 @@ DEF_OP(F80F2XM1) {
|
||||
DEF_OP(F80TAN) {
|
||||
auto Op = IROp->C<IR::IROp_F80TAN>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FTAN(Src);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80TAN>::handle(Data->State->CurrentFrame->State.FCW, Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -213,7 +209,7 @@ DEF_OP(F80TAN) {
|
||||
DEF_OP(F80SQRT) {
|
||||
auto Op = IROp->C<IR::IROp_F80SQRT>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FSQRT(Src);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80SQRT>::handle(Data->State->CurrentFrame->State.FCW, Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -221,7 +217,7 @@ DEF_OP(F80SQRT) {
|
||||
DEF_OP(F80SIN) {
|
||||
auto Op = IROp->C<IR::IROp_F80SIN>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FSIN(Src);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80SIN>::handle(Data->State->CurrentFrame->State.FCW, Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -229,7 +225,7 @@ DEF_OP(F80SIN) {
|
||||
DEF_OP(F80COS) {
|
||||
auto Op = IROp->C<IR::IROp_F80COS>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FCOS(Src);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80COS>::handle(Data->State->CurrentFrame->State.FCW, Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -237,7 +233,7 @@ DEF_OP(F80COS) {
|
||||
DEF_OP(F80XTRACT_EXP) {
|
||||
auto Op = IROp->C<IR::IROp_F80XTRACT_EXP>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FXTRACT_EXP(Src);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80XTRACT_EXP>::handle(Data->State->CurrentFrame->State.FCW, Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
@@ -245,102 +241,32 @@ DEF_OP(F80XTRACT_EXP) {
|
||||
DEF_OP(F80XTRACT_SIG) {
|
||||
auto Op = IROp->C<IR::IROp_F80XTRACT_SIG>();
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = X80SoftFloat::FXTRACT_SIG(Src);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80XTRACT_SIG>::handle(Data->State->CurrentFrame->State.FCW, Src);
|
||||
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80CMP) {
|
||||
auto Op = IROp->C<IR::IROp_F80Cmp>();
|
||||
uint32_t ResultFlags{};
|
||||
const auto Src1 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src1);
|
||||
const auto Src2 = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src2);
|
||||
bool eq, lt, nan;
|
||||
X80SoftFloat::FCMP(Src1, Src2, &eq, <, &nan);
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_LT) &&
|
||||
lt) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_LT);
|
||||
}
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_UNORDERED) &&
|
||||
nan) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_UNORDERED);
|
||||
}
|
||||
if (Op->Flags & (1 << IR::FCMP_FLAG_EQ) &&
|
||||
eq) {
|
||||
ResultFlags |= (1 << IR::FCMP_FLAG_EQ);
|
||||
}
|
||||
const auto ResultFlags = CPU::OpHandlers<IR::OP_F80CMP>::handle<IR::FCMP_FLAG_LT | IR::FCMP_FLAG_UNORDERED | IR::FCMP_FLAG_EQ>(Data->State->CurrentFrame->State.FCW, Src1, Src2);
|
||||
|
||||
GD = ResultFlags;
|
||||
}
|
||||
|
||||
DEF_OP(F80BCDLOAD) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDLoad>();
|
||||
const uint8_t *Src1 = GetSrc<uint8_t*>(Data->SSAData, Op->X80Src);
|
||||
uint64_t BCD{};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
// Each 4bit split is a digit
|
||||
// Only 0-9 is supported, A-F results in undefined data
|
||||
// | 4 bit | 4 bit |
|
||||
// | 10s place | 1s place |
|
||||
// EG 0x48 = 48
|
||||
// EG 0x4847 = 4847
|
||||
// This gives us an 18digit value encoded in BCD
|
||||
// The last byte lets us know if it negative or not
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
uint8_t Digit = Src1[8 - i];
|
||||
// First shift our last value over
|
||||
BCD *= 100;
|
||||
|
||||
// Add the tens place digit
|
||||
BCD += (Digit >> 4) * 10;
|
||||
|
||||
// Add the ones place digit
|
||||
BCD += Digit & 0xF;
|
||||
}
|
||||
|
||||
// Set negative flag once converted to x87
|
||||
bool Negative = Src1[9] & 0x80;
|
||||
X80SoftFloat Tmp;
|
||||
|
||||
Tmp = BCD;
|
||||
Tmp.Sign = Negative;
|
||||
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80BCDLOAD>::handle(Data->State->CurrentFrame->State.FCW, Src);
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F80BCDSTORE) {
|
||||
auto Op = IROp->C<IR::IROp_F80BCDStore>();
|
||||
X80SoftFloat Src1 = X80SoftFloat::FRNDINT(*GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src));
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
// Clear the Sign bit
|
||||
Src1.Sign = 0;
|
||||
|
||||
uint64_t Tmp = Src1;
|
||||
uint8_t BCD[10]{};
|
||||
|
||||
for (size_t i = 0; i < 9; ++i) {
|
||||
if (Tmp == 0) {
|
||||
// Nothing left? Just leave
|
||||
break;
|
||||
}
|
||||
// Extract the lower 100 values
|
||||
uint8_t Digit = Tmp % 100;
|
||||
|
||||
// Now divide it for the next iteration
|
||||
Tmp /= 100;
|
||||
|
||||
uint8_t UpperNibble = Digit / 10;
|
||||
uint8_t LowerNibble = Digit % 10;
|
||||
|
||||
// Now store the BCD
|
||||
BCD[i] = (UpperNibble << 4) | LowerNibble;
|
||||
}
|
||||
|
||||
// Set negative flag once converted to x87
|
||||
BCD[9] = Negative ? 0x80 : 0;
|
||||
|
||||
memcpy(GDP, BCD, 10);
|
||||
const auto Src = *GetSrc<X80SoftFloat*>(Data->SSAData, Op->X80Src);
|
||||
const auto Tmp = CPU::OpHandlers<IR::OP_F80BCDSTORE>::handle(Data->State->CurrentFrame->State.FCW, Src);
|
||||
memcpy(GDP, &Tmp, sizeof(X80SoftFloat));
|
||||
}
|
||||
|
||||
DEF_OP(F64SIN) {
|
||||
|
||||
@@ -7,13 +7,41 @@
|
||||
#include "Interface/Core/Interpreter/Fallbacks/FallbackOpHandler.h"
|
||||
|
||||
namespace FEXCore::CPU {
|
||||
static void LoadDeferredFCW(uint16_t NewFCW) {
|
||||
auto PC = (NewFCW >> 8) & 3;
|
||||
switch(PC) {
|
||||
case 0: extF80_roundingPrecision = 32; break;
|
||||
case 2: extF80_roundingPrecision = 64; break;
|
||||
case 3: extF80_roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
}
|
||||
|
||||
auto RC = (NewFCW >> 10) & 3;
|
||||
switch(RC) {
|
||||
case 0:
|
||||
softfloat_roundingMode = softfloat_round_near_even;
|
||||
break;
|
||||
case 1:
|
||||
softfloat_roundingMode = softfloat_round_min;
|
||||
break;
|
||||
case 2:
|
||||
softfloat_roundingMode = softfloat_round_max;
|
||||
break;
|
||||
case 3:
|
||||
softfloat_roundingMode = softfloat_round_minMag;
|
||||
break;
|
||||
}
|
||||
}
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
static X80SoftFloat handle4(float src) {
|
||||
static X80SoftFloat handle4(uint16_t NewFCW, float src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
|
||||
static X80SoftFloat handle8(double src) {
|
||||
static X80SoftFloat handle8(uint16_t NewFCW, double src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
};
|
||||
@@ -21,7 +49,9 @@ struct OpHandlers<IR::OP_F80CVTTO> {
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CMP> {
|
||||
template<uint32_t Flags>
|
||||
static uint64_t handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
static uint64_t handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
|
||||
bool eq, lt, nan;
|
||||
uint64_t ResultFlags = 0;
|
||||
|
||||
@@ -44,30 +74,36 @@ struct OpHandlers<IR::OP_F80CMP> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVT> {
|
||||
static float handle4(X80SoftFloat src) {
|
||||
static float handle4(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
|
||||
static double handle8(X80SoftFloat src) {
|
||||
static double handle8(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
static int16_t handle2(X80SoftFloat src) {
|
||||
static int16_t handle2(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
|
||||
static int32_t handle4(X80SoftFloat src) {
|
||||
static int32_t handle4(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
|
||||
static int64_t handle8(X80SoftFloat src) {
|
||||
static int64_t handle8(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
|
||||
static int16_t handle2t(X80SoftFloat src) {
|
||||
static int16_t handle2t(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
auto rv = extF80_to_i32(src, softfloat_round_minMag, false);
|
||||
|
||||
if (rv > INT16_MAX) {
|
||||
@@ -79,216 +115,243 @@ struct OpHandlers<IR::OP_F80CVTINT> {
|
||||
}
|
||||
}
|
||||
|
||||
static int32_t handle4t(X80SoftFloat src) {
|
||||
static int32_t handle4t(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return extF80_to_i32(src, softfloat_round_minMag, false);
|
||||
}
|
||||
|
||||
static int64_t handle8t(X80SoftFloat src) {
|
||||
static int64_t handle8t(uint16_t NewFCW, X80SoftFloat src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return extF80_to_i64(src, softfloat_round_minMag, false);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80CVTTOINT> {
|
||||
static X80SoftFloat handle2(int16_t src) {
|
||||
static X80SoftFloat handle2(uint16_t NewFCW, int16_t src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
|
||||
static X80SoftFloat handle4(int32_t src) {
|
||||
static X80SoftFloat handle4(uint16_t NewFCW, int32_t src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ROUND> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FRNDINT(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80F2XM1> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::F2XM1(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80TAN> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FTAN(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SQRT> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FSQRT(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SIN> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FSIN(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80COS> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FCOS(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_EXP> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FXTRACT_EXP(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80XTRACT_SIG> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FXTRACT_SIG(Src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ADD> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FADD(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SUB> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FSUB(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80MUL> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FMUL(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80DIV> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FDIV(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FYL2X> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FYL2X(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80ATAN> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FATAN(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM1> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FREM1(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80FPREM> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FREM(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80SCALE> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1, X80SoftFloat Src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return X80SoftFloat::FSCALE(Src1, Src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SIN> {
|
||||
static double handle(double src) {
|
||||
static double handle(uint16_t NewFCW, double src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return sin(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64COS> {
|
||||
static double handle(double src) {
|
||||
static double handle(uint16_t NewFCW, double src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return cos(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64TAN> {
|
||||
static double handle(double src) {
|
||||
static double handle(uint16_t NewFCW, double src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return tan(src);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64F2XM1> {
|
||||
static double handle(double src) {
|
||||
static double handle(uint16_t NewFCW, double src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return exp2(src) - 1.0;
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64ATAN> {
|
||||
static double handle(double src1, double src2) {
|
||||
static double handle(uint16_t NewFCW, double src1, double src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return atan2(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM> {
|
||||
static double handle(double src1, double src2) {
|
||||
static double handle(uint16_t NewFCW, double src1, double src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return fmod(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FPREM1> {
|
||||
static double handle(double src1, double src2) {
|
||||
static double handle(uint16_t NewFCW, double src1, double src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return remainder(src1, src2);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64FYL2X> {
|
||||
static double handle(double src1, double src2) {
|
||||
static double handle(uint16_t NewFCW, double src1, double src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
return src2 * log2(src1);
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F64SCALE> {
|
||||
static double handle(double src1, double src2) {
|
||||
static double handle(uint16_t NewFCW, double src1, double src2) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
double trunc = (double)(int64_t)(src2); //truncate
|
||||
return src1 * exp2(trunc);
|
||||
}
|
||||
};
|
||||
|
||||
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src1) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src1) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
bool Negative = Src1.Sign;
|
||||
|
||||
Src1 = X80SoftFloat::FRNDINT(Src1);
|
||||
@@ -328,7 +391,8 @@ struct OpHandlers<IR::OP_F80BCDSTORE> {
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80BCDLOAD> {
|
||||
static X80SoftFloat handle(X80SoftFloat Src) {
|
||||
static X80SoftFloat handle(uint16_t NewFCW, X80SoftFloat Src) {
|
||||
LoadDeferredFCW(NewFCW);
|
||||
uint8_t *Src1 = reinterpret_cast<uint8_t *>(&Src);
|
||||
uint64_t BCD{};
|
||||
// We walk through each uint8_t and pull out the BCD encoding
|
||||
@@ -362,34 +426,4 @@ struct OpHandlers<IR::OP_F80BCDLOAD> {
|
||||
}
|
||||
};
|
||||
|
||||
template<>
|
||||
struct OpHandlers<IR::OP_F80LOADFCW> {
|
||||
static void handle(uint16_t NewFCW) {
|
||||
|
||||
auto PC = (NewFCW >> 8) & 3;
|
||||
switch(PC) {
|
||||
case 0: extF80_roundingPrecision = 32; break;
|
||||
case 2: extF80_roundingPrecision = 64; break;
|
||||
case 3: extF80_roundingPrecision = 80; break;
|
||||
case 1: LOGMAN_MSG_A_FMT("Invalid x87 precision mode, {}", PC);
|
||||
}
|
||||
|
||||
auto RC = (NewFCW >> 10) & 3;
|
||||
switch(RC) {
|
||||
case 0:
|
||||
softfloat_roundingMode = softfloat_round_near_even;
|
||||
break;
|
||||
case 1:
|
||||
softfloat_roundingMode = softfloat_round_min;
|
||||
break;
|
||||
case 2:
|
||||
softfloat_roundingMode = softfloat_round_max;
|
||||
break;
|
||||
case 3:
|
||||
softfloat_roundingMode = softfloat_round_minMag;
|
||||
break;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
} // namespace FEXCore::CPU
|
||||
@@ -15,78 +15,73 @@ static FallbackInfo GetFallbackInfo(R(*fn)(Args...), FEXCore::Core::FallbackHand
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(float), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_F32, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(uint16_t, float), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_I16_F32, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_F64, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_I16_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_I16, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(uint16_t, int16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_I16_I16, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(void(*fn)(uint16_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_VOID_U16, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(uint16_t, int32_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_I16_I32, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(int32_t), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_I32, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(float(*fn)(uint16_t, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F32_I16_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(float(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F32_F80, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(uint16_t, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F80, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(uint16_t, double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F64, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(uint16_t, double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_I16_F64_F64, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(double(*fn)(double,double), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F64_F64_F64, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(int16_t(*fn)(uint16_t, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I16_I16_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int16_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I16_F80, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(int32_t(*fn)(uint16_t, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I32_I16_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int32_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I32_F80, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(int64_t(*fn)(uint16_t, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I64_I16_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(int64_t(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I64_F80, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(uint64_t(*fn)(uint16_t, X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I64_I16_F80_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(uint64_t(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_I64_F80_F80, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(uint16_t, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_I16_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_F80_F80, (void*)fn, HandlerIndex};
|
||||
FallbackInfo GetFallbackInfo(X80SoftFloat(*fn)(uint16_t, X80SoftFloat, X80SoftFloat), FEXCore::Core::FallbackHandlerIndex HandlerIndex) {
|
||||
return {FABI_F80_I16_F80_F80, (void*)fn, HandlerIndex};
|
||||
}
|
||||
|
||||
template<>
|
||||
@@ -100,7 +95,6 @@ FallbackInfo GetFallbackInfo(uint32_t(*fn)(__uint128_t, __uint128_t, uint16_t),
|
||||
}
|
||||
|
||||
void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
Info[Core::OPINDEX_F80LOADFCW] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW).fn);
|
||||
Info[Core::OPINDEX_F80CVTTO_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle4, Core::OPINDEX_F80CVTTO_4).fn);
|
||||
Info[Core::OPINDEX_F80CVTTO_8] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVTTO>::handle8, Core::OPINDEX_F80CVTTO_8).fn);
|
||||
Info[Core::OPINDEX_F80CVT_4] = reinterpret_cast<uint64_t>(GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80CVT>::handle4, Core::OPINDEX_F80CVT_4).fn);
|
||||
@@ -164,11 +158,6 @@ void InterpreterOps::FillFallbackIndexPointers(uint64_t *Info) {
|
||||
bool InterpreterOps::GetFallbackHandler(IR::IROp_Header const *IROp, FallbackInfo *Info) {
|
||||
uint8_t OpSize = IROp->Size;
|
||||
switch(IROp->Op) {
|
||||
case IR::OP_F80LOADFCW: {
|
||||
*Info = GetFallbackInfo(&FEXCore::CPU::OpHandlers<IR::OP_F80LOADFCW>::handle, Core::OPINDEX_F80LOADFCW);
|
||||
return true;
|
||||
}
|
||||
|
||||
case IR::OP_F80CVTTO: {
|
||||
auto Op = IROp->C<IR::IROp_F80CVTTo>();
|
||||
|
||||
|
||||
@@ -300,7 +300,6 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] {
|
||||
REGISTER_OP(PCLMUL, PCLMUL);
|
||||
|
||||
// F80 ops
|
||||
REGISTER_OP(F80LOADFCW, F80LOADFCW);
|
||||
REGISTER_OP(F80ADD, F80ADD);
|
||||
REGISTER_OP(F80SUB, F80SUB);
|
||||
REGISTER_OP(F80MUL, F80MUL);
|
||||
|
||||
@@ -24,21 +24,20 @@ namespace FEXCore::Core{
|
||||
namespace FEXCore::CPU {
|
||||
enum FallbackABI {
|
||||
FABI_UNKNOWN,
|
||||
FABI_VOID_U16,
|
||||
FABI_F80_F32,
|
||||
FABI_F80_F64,
|
||||
FABI_F80_I16,
|
||||
FABI_F80_I32,
|
||||
FABI_F32_F80,
|
||||
FABI_F64_F80,
|
||||
FABI_F64_F64,
|
||||
FABI_F64_F64_F64,
|
||||
FABI_I16_F80,
|
||||
FABI_I32_F80,
|
||||
FABI_I64_F80,
|
||||
FABI_I64_F80_F80,
|
||||
FABI_F80_F80,
|
||||
FABI_F80_F80_F80,
|
||||
FABI_F80_I16_F32,
|
||||
FABI_F80_I16_F64,
|
||||
FABI_F80_I16_I16,
|
||||
FABI_F80_I16_I32,
|
||||
FABI_F32_I16_F80,
|
||||
FABI_F64_I16_F80,
|
||||
FABI_F64_I16_F64,
|
||||
FABI_F64_I16_F64_F64,
|
||||
FABI_I16_I16_F80,
|
||||
FABI_I32_I16_F80,
|
||||
FABI_I64_I16_F80,
|
||||
FABI_I64_I16_F80_F80,
|
||||
FABI_F80_I16_F80,
|
||||
FABI_F80_I16_F80_F80,
|
||||
FABI_I32_I64_I64_I128_I128_I16,
|
||||
FABI_I32_I128_I128_I16,
|
||||
};
|
||||
|
||||
@@ -83,37 +83,18 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
#endif
|
||||
} else {
|
||||
switch(Info.ABI) {
|
||||
case FABI_VOID_U16:{
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
uxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Src1);
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<void, uint16_t>(ARMEmitter::Reg::r1);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_F32:{
|
||||
case FABI_F80_I16_F32:{
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
fmov(ARMEmitter::SReg::s0, Src1.S());
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, float>(ARMEmitter::Reg::r0);
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, float>(ARMEmitter::Reg::r1);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r0);
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -127,47 +108,17 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_F64:{
|
||||
case FABI_F80_I16_F64:{
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, double>(ARMEmitter::Reg::r0);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r0);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_I16:
|
||||
case FABI_F80_I32: {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
if (Info.ABI == FABI_F80_I16) {
|
||||
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Src1);
|
||||
}
|
||||
else {
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r0, Src1);
|
||||
}
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint32_t>(ARMEmitter::Reg::r1);
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, double>(ARMEmitter::Reg::r1);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
#endif
|
||||
@@ -183,21 +134,54 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F32_F80:{
|
||||
case FABI_F80_I16_I16:
|
||||
case FABI_F80_I16_I32: {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Src1 = GetReg(IROp->Args[0].ID());
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
if (Info.ABI == FABI_F80_I16_I16) {
|
||||
sxth(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
else {
|
||||
mov(ARMEmitter::Size::i32Bit, ARMEmitter::Reg::r1, Src1);
|
||||
}
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint32_t>(ARMEmitter::Reg::r2);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
FillStaticRegs();
|
||||
|
||||
const auto Dst = GetVReg(Node);
|
||||
eor(Dst.Q(), Dst.Q(), Dst.Q());
|
||||
ins(ARMEmitter::SubRegSize::i64Bit, Dst, 0, ARMEmitter::Reg::r0);
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F32_I16_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<float, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
|
||||
GenerateIndirectRuntimeCall<float, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -209,21 +193,22 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F80:{
|
||||
case FABI_F64_I16_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<double, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -235,7 +220,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
case FABI_F64_I16_F64: {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
@@ -243,11 +228,12 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<double, double>(ARMEmitter::Reg::r0);
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double>(ARMEmitter::Reg::r1);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r0);
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -259,7 +245,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
case FABI_F64_I16_F64_F64: {
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
@@ -269,11 +255,12 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
|
||||
mov(ARMEmitter::DReg::d0, Src1.D());
|
||||
mov(ARMEmitter::DReg::d1, Src2.D());
|
||||
ldr(ARMEmitter::XReg::x0, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
ldr(ARMEmitter::XReg::x1, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<double, double, double>(ARMEmitter::Reg::r0);
|
||||
GenerateIndirectRuntimeCall<double, uint16_t, double, double>(ARMEmitter::Reg::r1);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r0);
|
||||
blr(ARMEmitter::Reg::r1);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -285,21 +272,22 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
case FABI_I16_I16_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -310,21 +298,22 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
sxth(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_F80:{
|
||||
case FABI_I32_I16_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
|
||||
GenerateIndirectRuntimeCall<uint32_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -335,21 +324,22 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
mov(ARMEmitter::Size::i32Bit, Dst, ARMEmitter::Reg::r0);
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80:{
|
||||
case FABI_I64_I16_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -360,7 +350,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80_F80:{
|
||||
case FABI_I64_I16_F80_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
@@ -368,17 +358,18 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r3, Src2, 4);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
GenerateIndirectRuntimeCall<uint64_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r4);
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
#endif
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -388,21 +379,22 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
mov(ARMEmitter::Size::i64Bit, Dst, ARMEmitter::Reg::r0);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80:{
|
||||
case FABI_F80_I16_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x2, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x3, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint64_t, uint64_t>(ARMEmitter::Reg::r2);
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t>(ARMEmitter::Reg::r3);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r2);
|
||||
blr(ARMEmitter::Reg::r3);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
@@ -415,7 +407,7 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
ins(ARMEmitter::SubRegSize::i16Bit, Dst, 4, ARMEmitter::Reg::r1);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80_F80:{
|
||||
case FABI_F80_I16_F80_F80:{
|
||||
SpillStaticRegs(TMP1);
|
||||
|
||||
PushDynamicRegsAndLR(TMP1);
|
||||
@@ -423,17 +415,18 @@ void Arm64JITCore::Op_Unhandled(IR::IROp_Header const *IROp, IR::NodeID Node) {
|
||||
const auto Src1 = GetVReg(IROp->Args[0].ID());
|
||||
const auto Src2 = GetVReg(IROp->Args[1].ID());
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r0, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r1, Src1, 4);
|
||||
ldrh(ARMEmitter::WReg::w0, STATE, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r1, Src1, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r2, Src1, 4);
|
||||
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r2, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r3, Src2, 4);
|
||||
umov<ARMEmitter::SubRegSize::i64Bit>(ARMEmitter::Reg::r3, Src2, 0);
|
||||
umov<ARMEmitter::SubRegSize::i16Bit>(ARMEmitter::Reg::r4, Src2, 4);
|
||||
|
||||
ldr(ARMEmitter::XReg::x4, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
ldr(ARMEmitter::XReg::x5, STATE_PTR(CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex]));
|
||||
#ifdef VIXL_SIMULATOR
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r4);
|
||||
GenerateIndirectRuntimeCall<__uint128_t, uint16_t, uint64_t, uint64_t, uint64_t, uint64_t>(ARMEmitter::Reg::r5);
|
||||
#else
|
||||
blr(ARMEmitter::Reg::r4);
|
||||
blr(ARMEmitter::Reg::r5);
|
||||
#endif
|
||||
|
||||
PopDynamicRegsAndLR();
|
||||
|
||||
@@ -104,18 +104,11 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
#endif
|
||||
} else {
|
||||
switch(Info.ABI) {
|
||||
case FABI_VOID_U16: {
|
||||
PushRegs();
|
||||
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
break;
|
||||
}
|
||||
case FABI_F80_F32:{
|
||||
case FABI_F80_I16_F32:{
|
||||
PushRegs();
|
||||
|
||||
movss(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
@@ -126,10 +119,11 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_F64:{
|
||||
case FABI_F80_I16_F64:{
|
||||
PushRegs();
|
||||
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
PopRegs();
|
||||
@@ -140,15 +134,17 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F80_I16:
|
||||
case FABI_F80_I32: {
|
||||
case FABI_F80_I16_I16:
|
||||
case FABI_F80_I16_I32: {
|
||||
PushRegs();
|
||||
|
||||
if (Info.ABI == FABI_F80_I16) {
|
||||
movsx(rdi, GetSrc<RA_32>(IROp->Args[0].ID()).cvt16());
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
|
||||
if (Info.ABI == FABI_F80_I16_I16) {
|
||||
movsx(rsi, GetSrc<RA_32>(IROp->Args[0].ID()).cvt16());
|
||||
}
|
||||
else {
|
||||
mov(edi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
mov(esi, GetSrc<RA_32>(IROp->Args[0].ID()));
|
||||
}
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
@@ -160,11 +156,12 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F32_F80:{
|
||||
case FABI_F32_I16_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
movq(rsi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
@@ -174,11 +171,12 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F80:{
|
||||
case FABI_F64_I16_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
movq(rsi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
@@ -188,9 +186,10 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64: {
|
||||
case FABI_F64_I16_F64: {
|
||||
PushRegs();
|
||||
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
@@ -201,9 +200,10 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_F64_F64_F64: {
|
||||
case FABI_F64_I16_F64_F64: {
|
||||
PushRegs();
|
||||
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
movsd(xmm0, GetSrc(IROp->Args[0].ID()));
|
||||
movsd(xmm1, GetSrc(IROp->Args[1].ID()));
|
||||
|
||||
@@ -215,11 +215,12 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
}
|
||||
break;
|
||||
|
||||
case FABI_I16_F80:{
|
||||
case FABI_I16_I16_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
movq(rsi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
@@ -228,11 +229,12 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
movsx(GetDst<RA_64>(Node), ax);
|
||||
}
|
||||
break;
|
||||
case FABI_I32_F80:{
|
||||
case FABI_I32_I16_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
movq(rsi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
@@ -241,11 +243,12 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
mov(GetDst<RA_32>(Node), eax);
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80:{
|
||||
case FABI_I64_I16_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
movq(rsi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
@@ -254,14 +257,15 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
mov(GetDst<RA_64>(Node), rax);
|
||||
}
|
||||
break;
|
||||
case FABI_I64_F80_F80:{
|
||||
case FABI_I64_I16_F80_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
movq(rsi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
movq(rdx, GetSrc(IROp->Args[1].ID()));
|
||||
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
|
||||
movq(rcx, GetSrc(IROp->Args[1].ID()));
|
||||
pextrq(r8, GetSrc(IROp->Args[1].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
@@ -270,11 +274,12 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
mov(GetDst<RA_64>(Node), rax);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80:{
|
||||
case FABI_F80_I16_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
movq(rsi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
@@ -285,14 +290,15 @@ void X86JITCore::Op_Unhandled(IR::IROp_Header *IROp, IR::NodeID Node) {
|
||||
pinsrw(GetDst(Node), edx, 4);
|
||||
}
|
||||
break;
|
||||
case FABI_F80_F80_F80:{
|
||||
case FABI_F80_I16_F80_F80:{
|
||||
PushRegs();
|
||||
|
||||
movq(rdi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rsi, GetSrc(IROp->Args[0].ID()), 1);
|
||||
mov(rdi, word [STATE + offsetof(FEXCore::Core::CPUState, FCW)]);
|
||||
movq(rsi, GetSrc(IROp->Args[0].ID()));
|
||||
pextrq(rdx, GetSrc(IROp->Args[0].ID()), 1);
|
||||
|
||||
movq(rdx, GetSrc(IROp->Args[1].ID()));
|
||||
pextrq(rcx, GetSrc(IROp->Args[1].ID()), 1);
|
||||
movq(rcx, GetSrc(IROp->Args[1].ID()));
|
||||
pextrq(r8, GetSrc(IROp->Args[1].ID()), 1);
|
||||
|
||||
call(qword [STATE + offsetof(FEXCore::Core::CpuStateFrame, Pointers.Common.FallbackHandlerPointers[Info.HandlerIndex])]);
|
||||
|
||||
|
||||
@@ -2971,7 +2971,6 @@ void OpDispatchBuilder::RestoreX87State(OrderedNode *MemBase) {
|
||||
const auto OpSize = IR::SizeToOpSize(CTX->GetGPRSize());
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, MemBase, 2);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
{
|
||||
|
||||
@@ -694,7 +694,6 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
|
||||
auto Zero = _Constant(0);
|
||||
// Init FCW to 0x037F
|
||||
auto NewFCW = _Constant(16, 0x037F);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
// Init FSW to 0
|
||||
@@ -1015,7 +1014,6 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
@@ -1098,7 +1096,6 @@ void OpDispatchBuilder::X87FNSTENV(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FLDCW(OpcodeArgs) {
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
}
|
||||
|
||||
@@ -1216,7 +1213,6 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
Mem = AppendSegmentOffset(Mem, Op->Flags);
|
||||
|
||||
auto NewFCW = _LoadMem(GPRClass, 2, Mem, 2);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
|
||||
@@ -46,7 +46,6 @@ class OrderedNode;
|
||||
void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
|
||||
// Init FCW to 0x037F
|
||||
auto NewFCW = _Constant(16, 0x037F);
|
||||
_F80LoadFCW(NewFCW);
|
||||
// Init host rounding mode to zero
|
||||
auto Zero = _Constant(0);
|
||||
_SetRoundingMode(Zero);
|
||||
@@ -78,8 +77,6 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
|
||||
roundingMode = _And(OpSize::i32Bit, roundingMode, roundMask);
|
||||
_SetRoundingMode(roundingMode);
|
||||
_F80LoadFCW(NewFCW);
|
||||
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
OrderedNode *MemLocation = _Add(OpSize::i64Bit, Mem, _Constant(Size * 1));
|
||||
@@ -96,7 +93,6 @@ void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
|
||||
|
||||
void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
|
||||
OrderedNode *NewFCW = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
_F80LoadFCW(NewFCW); //keeps BCD code working
|
||||
//ignore the rounding precision, we're always 64-bit in F64.
|
||||
//extract rounding mode
|
||||
OrderedNode *roundingMode = NewFCW;
|
||||
@@ -1026,7 +1022,6 @@ void OpDispatchBuilder::X87FRSTORF64(OpcodeArgs) {
|
||||
roundingMode = _Lshr(OpSize::i32Bit, roundingMode, roundShift);
|
||||
roundingMode = _And(OpSize::i32Bit, roundingMode, roundMask);
|
||||
_SetRoundingMode(roundingMode);
|
||||
_F80LoadFCW(NewFCW);
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
_StoreContext(2, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
|
||||
@@ -1912,9 +1912,6 @@
|
||||
}
|
||||
},
|
||||
"F80": {
|
||||
"F80LoadFCW GPR:$Src": {
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"FPR = F80Add FPR:$X80Src1, FPR:$X80Src2": {
|
||||
"DestSize": "16"
|
||||
},
|
||||
|
||||
@@ -1039,52 +1039,11 @@
|
||||
]
|
||||
},
|
||||
"fxrstor [rax]": {
|
||||
"ExpectedInstructionCount": 105,
|
||||
"ExpectedInstructionCount": 64,
|
||||
"Optimal": "No",
|
||||
"Comment": "GROUP15 0x0F 0xAE /1",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldrh w20, [x4]",
|
||||
"stp x4, x5, [x28, #8]",
|
||||
"stp x6, x7, [x28, #24]",
|
||||
"stp x8, x9, [x28, #40]",
|
||||
"stp x10, x11, [x28, #56]",
|
||||
"stp x12, x13, [x28, #72]",
|
||||
"stp x14, x15, [x28, #88]",
|
||||
"stp x16, x17, [x28, #104]",
|
||||
"stp x19, x29, [x28, #120]",
|
||||
"add x0, x28, #0xc0 (192)",
|
||||
"st1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x0], #64",
|
||||
"st1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x0], #64",
|
||||
"st1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x0], #64",
|
||||
"st1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x0], #64",
|
||||
"sub sp, sp, #0xf0 (240)",
|
||||
"mov x0, sp",
|
||||
"st1 {v2.2d, v3.2d}, [x0], #32",
|
||||
"st1 {v4.2d, v5.2d, v6.2d, v7.2d}, [x0], #64",
|
||||
"st1 {v8.2d, v9.2d, v10.2d, v11.2d}, [x0], #64",
|
||||
"st1 {v12.2d, v13.2d, v14.2d, v15.2d}, [x0], #64",
|
||||
"str x30, [x0]",
|
||||
"uxth w0, w20",
|
||||
"ldr x1, [x28, #1144]",
|
||||
"blr x1",
|
||||
"ld1 {v2.2d, v3.2d}, [sp], #32",
|
||||
"ld1 {v4.2d, v5.2d, v6.2d, v7.2d}, [sp], #64",
|
||||
"ld1 {v8.2d, v9.2d, v10.2d, v11.2d}, [sp], #64",
|
||||
"ld1 {v12.2d, v13.2d, v14.2d, v15.2d}, [sp], #64",
|
||||
"ldr x30, [sp], #16",
|
||||
"add x5, x28, #0xc0 (192)",
|
||||
"ld1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x5], #64",
|
||||
"ld1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x5], #64",
|
||||
"ld1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x5], #64",
|
||||
"ld1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x5], #64",
|
||||
"ldp x4, x5, [x28, #8]",
|
||||
"ldp x6, x7, [x28, #24]",
|
||||
"ldp x8, x9, [x28, #40]",
|
||||
"ldp x10, x11, [x28, #56]",
|
||||
"ldp x12, x13, [x28, #72]",
|
||||
"ldp x14, x15, [x28, #88]",
|
||||
"ldp x16, x17, [x28, #104]",
|
||||
"ldp x19, x29, [x28, #120]",
|
||||
"strh w20, [x28, #1008]",
|
||||
"ldrh w20, [x4, #2]",
|
||||
"ubfx w21, w20, #11, #3",
|
||||
@@ -1350,7 +1309,7 @@
|
||||
]
|
||||
},
|
||||
"xrstor [rax]": {
|
||||
"ExpectedInstructionCount": 194,
|
||||
"ExpectedInstructionCount": 112,
|
||||
"Optimal": "No",
|
||||
"Comment": "GROUP15 0x0F 0xAE /5",
|
||||
"ExpectedArm64ASM": [
|
||||
@@ -1358,49 +1317,8 @@
|
||||
"ldr x21, [x20, #512]",
|
||||
"ubfx x22, x21, #0, #1",
|
||||
"cbnz x22, #+0x8",
|
||||
"b #+0x128",
|
||||
"b #+0x84",
|
||||
"ldrh w22, [x20]",
|
||||
"stp x4, x5, [x28, #8]",
|
||||
"stp x6, x7, [x28, #24]",
|
||||
"stp x8, x9, [x28, #40]",
|
||||
"stp x10, x11, [x28, #56]",
|
||||
"stp x12, x13, [x28, #72]",
|
||||
"stp x14, x15, [x28, #88]",
|
||||
"stp x16, x17, [x28, #104]",
|
||||
"stp x19, x29, [x28, #120]",
|
||||
"add x0, x28, #0xc0 (192)",
|
||||
"st1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x0], #64",
|
||||
"st1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x0], #64",
|
||||
"st1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x0], #64",
|
||||
"st1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x0], #64",
|
||||
"sub sp, sp, #0xf0 (240)",
|
||||
"mov x0, sp",
|
||||
"st1 {v2.2d, v3.2d}, [x0], #32",
|
||||
"st1 {v4.2d, v5.2d, v6.2d, v7.2d}, [x0], #64",
|
||||
"st1 {v8.2d, v9.2d, v10.2d, v11.2d}, [x0], #64",
|
||||
"st1 {v12.2d, v13.2d, v14.2d, v15.2d}, [x0], #64",
|
||||
"str x30, [x0]",
|
||||
"uxth w0, w22",
|
||||
"ldr x1, [x28, #1144]",
|
||||
"blr x1",
|
||||
"ld1 {v2.2d, v3.2d}, [sp], #32",
|
||||
"ld1 {v4.2d, v5.2d, v6.2d, v7.2d}, [sp], #64",
|
||||
"ld1 {v8.2d, v9.2d, v10.2d, v11.2d}, [sp], #64",
|
||||
"ld1 {v12.2d, v13.2d, v14.2d, v15.2d}, [sp], #64",
|
||||
"ldr x30, [sp], #16",
|
||||
"add x5, x28, #0xc0 (192)",
|
||||
"ld1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x5], #64",
|
||||
"ld1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x5], #64",
|
||||
"ld1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x5], #64",
|
||||
"ld1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x5], #64",
|
||||
"ldp x4, x5, [x28, #8]",
|
||||
"ldp x6, x7, [x28, #24]",
|
||||
"ldp x8, x9, [x28, #40]",
|
||||
"ldp x10, x11, [x28, #56]",
|
||||
"ldp x12, x13, [x28, #72]",
|
||||
"ldp x14, x15, [x28, #88]",
|
||||
"ldp x16, x17, [x28, #104]",
|
||||
"ldp x19, x29, [x28, #120]",
|
||||
"strh w22, [x28, #1008]",
|
||||
"ldrh w22, [x20, #2]",
|
||||
"ubfx w23, w22, #11, #3",
|
||||
@@ -1431,50 +1349,9 @@
|
||||
"str q2, [x28, #848]",
|
||||
"ldr q2, [x20, #144]",
|
||||
"str q2, [x28, #864]",
|
||||
"b #+0xf0",
|
||||
"b #+0x4c",
|
||||
"mov w22, #0x0",
|
||||
"mov w23, #0x37f",
|
||||
"stp x4, x5, [x28, #8]",
|
||||
"stp x6, x7, [x28, #24]",
|
||||
"stp x8, x9, [x28, #40]",
|
||||
"stp x10, x11, [x28, #56]",
|
||||
"stp x12, x13, [x28, #72]",
|
||||
"stp x14, x15, [x28, #88]",
|
||||
"stp x16, x17, [x28, #104]",
|
||||
"stp x19, x29, [x28, #120]",
|
||||
"add x0, x28, #0xc0 (192)",
|
||||
"st1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x0], #64",
|
||||
"st1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x0], #64",
|
||||
"st1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x0], #64",
|
||||
"st1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x0], #64",
|
||||
"sub sp, sp, #0xf0 (240)",
|
||||
"mov x0, sp",
|
||||
"st1 {v2.2d, v3.2d}, [x0], #32",
|
||||
"st1 {v4.2d, v5.2d, v6.2d, v7.2d}, [x0], #64",
|
||||
"st1 {v8.2d, v9.2d, v10.2d, v11.2d}, [x0], #64",
|
||||
"st1 {v12.2d, v13.2d, v14.2d, v15.2d}, [x0], #64",
|
||||
"str x30, [x0]",
|
||||
"uxth w0, w23",
|
||||
"ldr x1, [x28, #1144]",
|
||||
"blr x1",
|
||||
"ld1 {v2.2d, v3.2d}, [sp], #32",
|
||||
"ld1 {v4.2d, v5.2d, v6.2d, v7.2d}, [sp], #64",
|
||||
"ld1 {v8.2d, v9.2d, v10.2d, v11.2d}, [sp], #64",
|
||||
"ld1 {v12.2d, v13.2d, v14.2d, v15.2d}, [sp], #64",
|
||||
"ldr x30, [sp], #16",
|
||||
"add x5, x28, #0xc0 (192)",
|
||||
"ld1 {v16.2d, v17.2d, v18.2d, v19.2d}, [x5], #64",
|
||||
"ld1 {v20.2d, v21.2d, v22.2d, v23.2d}, [x5], #64",
|
||||
"ld1 {v24.2d, v25.2d, v26.2d, v27.2d}, [x5], #64",
|
||||
"ld1 {v28.2d, v29.2d, v30.2d, v31.2d}, [x5], #64",
|
||||
"ldp x4, x5, [x28, #8]",
|
||||
"ldp x6, x7, [x28, #24]",
|
||||
"ldp x8, x9, [x28, #40]",
|
||||
"ldp x10, x11, [x28, #56]",
|
||||
"ldp x12, x13, [x28, #72]",
|
||||
"ldp x14, x15, [x28, #88]",
|
||||
"ldp x16, x17, [x28, #104]",
|
||||
"ldp x19, x29, [x28, #120]",
|
||||
"strh w23, [x28, #1008]",
|
||||
"strb w22, [x28, #747]",
|
||||
"strb w22, [x28, #744]",
|
||||
|
||||
+2260
-2158
File diff suppressed because it is too large.
Load diff
Reference in new issue
Block a user