mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 14:00:16 +02:00
OpcodeDispatcher: Match x86 overflow behaviour for F2I conversions
ARM behaviour here is to saturate on overflow or NaN inputs, whereas X86 returns a sentinel value of 2^(bitsize-1), explicitly emulate this.
This commit is contained in:
1 parent
9bdb1f4306
commit
efd6e95059
6 files changed
+95
-95
No files matched your search
@@ -39,6 +39,14 @@ namespace CPU {
|
||||
{0xC90F'DAA2'2168'C235ULL, 0x0000'0000'0000'4000ULL}, // NAMED_VECTOR_X87_PI
|
||||
{0x9A20'9A84'FBCF'F799ULL, 0x0000'0000'0000'3FFDULL}, // NAMED_VECTOR_X87_LOG10_2
|
||||
{0xB172'17F7'D1CF'79ACULL, 0x0000'0000'0000'3FFEULL}, // NAMED_VECTOR_X87_LOG_2
|
||||
{0x4F00'0000'4F00'0000ULL, 0x4F00'0000'4F00'0000ULL}, // NAMED_VECTOR_CVTMAX_F32_I32
|
||||
{0x4F00'0000'4F00'0000ULL, 0x4F00'0000'4F00'0000ULL}, // NAMED_VECTOR_CVTMAX_F32_I32_UPPER
|
||||
{0x5F00'0000'5F00'0000ULL, 0x5F00'0000'5F00'0000ULL}, // NAMED_VECTOR_CVTMAX_F32_I64
|
||||
{0x41E0'0000'0000'0000ULL, 0x41E0'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_F64_I32
|
||||
{0x41E0'0000'0000'0000ULL, 0x41E0'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_F64_I32_UPPER
|
||||
{0x43E0'0000'0000'0000ULL, 0x43E0'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_F64_I64
|
||||
{0x8000'0000'8000'0000ULL, 0x8000'0000'8000'0000ULL}, // NAMED_VECTOR_CVTMAX_I32
|
||||
{0x8000'0000'0000'0000ULL, 0x8000'0000'0000'0000ULL}, // NAMED_VECTOR_CVTMAX_I64
|
||||
};
|
||||
|
||||
constexpr static auto PSHUFLW_LUT {[]() consteval {
|
||||
|
||||
@@ -1465,7 +1465,10 @@ private:
|
||||
Ref Scalar_CVT_Float_To_FloatImpl(OpcodeArgs, IR::OpSize DstElementSize, IR::OpSize SrcElementSize,
|
||||
const X86Tables::DecodedOperand& Src1Op, const X86Tables::DecodedOperand& Src2Op);
|
||||
|
||||
Ref Vector_CVT_Float_To_IntImpl(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode);
|
||||
Ref CVTFPR_To_GPRImpl(OpcodeArgs, Ref Src, IR::OpSize SrcElementSize, bool HostRoundingMode);
|
||||
|
||||
Ref Vector_CVT_Float_To_Int32Impl(OpcodeArgs, IR::OpSize DstSize, Ref Src, IR::OpSize SrcSize, IR::OpSize SrcElementSize,
|
||||
bool HostRoundingMode, bool ZeroUpperHalf);
|
||||
|
||||
Ref Vector_CVT_Int_To_FloatImpl(OpcodeArgs, IR::OpSize SrcElementSize, bool Widen);
|
||||
|
||||
|
||||
@@ -1058,18 +1058,8 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs) {
|
||||
Src.Low = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], OpSizeFromSrc(Op), Op->Flags);
|
||||
}
|
||||
|
||||
// GPR size is determined by REX.W
|
||||
// Source Element size is determined by instruction
|
||||
const auto GPRSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Result {};
|
||||
if constexpr (HostRoundingMode) {
|
||||
Result = _Float_ToGPR_S(GPRSize, SrcElementSize, Src.Low);
|
||||
} else {
|
||||
Result = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src.Low);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, OpSize::iInvalid);
|
||||
Ref Result = CVTFPR_To_GPRImpl(Op, Src.Low, SrcElementSize, HostRoundingMode);
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) {
|
||||
@@ -1614,38 +1604,20 @@ void OpDispatchBuilder::AVX128_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
auto Src = AVX128_LoadSource_WithOpSize(Op, Op->Src[0], Op->Flags, !Is128BitSrc);
|
||||
RefPair Result {};
|
||||
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
///< Special case for VCVTPD2DQ/CVTTPD2DQ because it has weird rounding requirements.
|
||||
Result.Low = _Vector_F64ToI32(OpSize::i128Bit, Src.Low, HostRoundingMode ? Round_Host : Round_Towards_Zero, Is128BitSrc);
|
||||
|
||||
if (!Is128BitSrc) {
|
||||
// Also convert the upper 128-bit lane
|
||||
auto ResultHigh = _Vector_F64ToI32(OpSize::i128Bit, Src.High, HostRoundingMode ? Round_Host : Round_Towards_Zero, false);
|
||||
|
||||
// Zip the two halves together in to the lower 128-bits
|
||||
Result.Low = _VZip(OpSize::i128Bit, OpSize::i64Bit, Result.Low, ResultHigh);
|
||||
}
|
||||
} else {
|
||||
auto Convert = [this](Ref Src) -> Ref {
|
||||
auto ElementSize = SrcElementSize;
|
||||
|
||||
if (HostRoundingMode) {
|
||||
return _Vector_FToS(OpSize::i128Bit, ElementSize, Src);
|
||||
} else {
|
||||
return _Vector_FToZS(OpSize::i128Bit, ElementSize, Src);
|
||||
}
|
||||
};
|
||||
|
||||
Result.Low = Convert(Src.Low);
|
||||
|
||||
if (!Is128BitSrc) {
|
||||
Result.High = Convert(Src.High);
|
||||
}
|
||||
}
|
||||
|
||||
if (SrcElementSize == OpSize::i64Bit || Is128BitSrc) {
|
||||
Result.Low = Vector_CVT_Float_To_Int32Impl(Op, OpSize::i128Bit, Src.Low, OpSize::i128Bit, SrcElementSize, HostRoundingMode, Is128BitSrc);
|
||||
if (Is128BitSrc) {
|
||||
// Zero the upper 128-bit lane of the result.
|
||||
Result = AVX128_Zext(Result.Low);
|
||||
} else {
|
||||
Result.High = Vector_CVT_Float_To_Int32Impl(Op, OpSize::i128Bit, Src.High, OpSize::i128Bit, SrcElementSize, HostRoundingMode, false);
|
||||
// Also convert the upper 128-bit lane
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
// Zip the two halves together in to the lower 128-bits
|
||||
Result.Low = _VZip(OpSize::i128Bit, OpSize::i64Bit, Result.Low, Result.High);
|
||||
|
||||
// Zero the upper 128-bit lane of the result.
|
||||
Result = AVX128_Zext(Result.Low);
|
||||
}
|
||||
}
|
||||
|
||||
AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result);
|
||||
|
||||
@@ -2067,6 +2067,24 @@ void OpDispatchBuilder::AVXCVTGPR_To_FPR(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::AVXCVTGPR_To_FPR<OpSize::i32Bit>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::AVXCVTGPR_To_FPR<OpSize::i64Bit>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::CVTFPR_To_GPRImpl(OpcodeArgs, Ref Src, IR::OpSize SrcElementSize, bool HostRoundingMode) {
|
||||
// GPR size is determined by REX.W
|
||||
// Source Element size is determined by instruction
|
||||
const auto GPRSize = OpSizeFromDst(Op);
|
||||
|
||||
if (HostRoundingMode) {
|
||||
Src = _Vector_FToI(SrcElementSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
Ref Converted = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
|
||||
bool Dst32 = GPRSize == OpSize::i32Bit;
|
||||
Ref MaxI = Dst32 ? _Constant(0x80000000) : _Constant(0x8000000000000000);
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcElementSize, (SrcElementSize == OpSize::i32Bit) ?
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F32_I32 : NAMED_VECTOR_CVTMAX_F32_I64) :
|
||||
(Dst32 ? NAMED_VECTOR_CVTMAX_F64_I32 : NAMED_VECTOR_CVTMAX_F64_I64));
|
||||
return _Select(GPRSize, SrcElementSize, CondClassType {FEXCore::IR::COND_FGT}, MaxF, Src, Converted, MaxI);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs) {
|
||||
// If loading a vector, use the full size, so we don't
|
||||
@@ -2074,18 +2092,8 @@ void OpDispatchBuilder::CVTFPR_To_GPR(OpcodeArgs) {
|
||||
// memory, then we want to load the element size exactly.
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : OpSizeFromSrc(Op);
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
|
||||
// GPR size is determined by REX.W
|
||||
// Source Element size is determined by instruction
|
||||
const auto GPRSize = OpSizeFromDst(Op);
|
||||
|
||||
if constexpr (HostRoundingMode) {
|
||||
Src = _Float_ToGPR_S(GPRSize, SrcElementSize, Src);
|
||||
} else {
|
||||
Src = _Float_ToGPR_ZS(GPRSize, SrcElementSize, Src);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Src, GPRSize, OpSize::iInvalid);
|
||||
Ref Result = CVTFPR_To_GPRImpl(Op, Src, SrcElementSize, HostRoundingMode);
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::CVTFPR_To_GPR<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
@@ -2127,43 +2135,40 @@ void OpDispatchBuilder::Vector_CVT_Int_To_Float(OpcodeArgs) {
|
||||
template void OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Int_To_Float<OpSize::i32Bit, false>(OpcodeArgs);
|
||||
|
||||
Ref OpDispatchBuilder::Vector_CVT_Float_To_IntImpl(OpcodeArgs, IR::OpSize SrcElementSize, bool HostRoundingMode) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
auto ElementSize = SrcElementSize;
|
||||
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
Src = _Vector_FToF(DstSize, SrcElementSize >> 1, Src, SrcElementSize);
|
||||
ElementSize = ElementSize >> 1;
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::Vector_CVT_Float_To_Int32Impl(OpcodeArgs, IR::OpSize DstSize, Ref Src, IR::OpSize SrcSize, IR::OpSize SrcElementSize,
|
||||
bool HostRoundingMode, bool ZeroUpperHalf) {
|
||||
if (HostRoundingMode) {
|
||||
return _Vector_FToS(DstSize, ElementSize, Src);
|
||||
} else {
|
||||
return _Vector_FToZS(DstSize, ElementSize, Src);
|
||||
Src = _Vector_FToI(SrcSize, SrcElementSize, Src, Round_Host);
|
||||
}
|
||||
|
||||
OpSize OverflowConstSize = ZeroUpperHalf && SrcElementSize == OpSize::i64Bit ? DstSize / 2 : DstSize;
|
||||
Ref MaxI = LoadAndCacheNamedVectorConstant(OverflowConstSize, NAMED_VECTOR_CVTMAX_I32);
|
||||
Ref Converted {}, Cmp {};
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_CVTMAX_F64_I32);
|
||||
Converted = _Vector_F64ToI32(DstSize, Src, Round_Towards_Zero, ZeroUpperHalf);
|
||||
|
||||
Cmp = _VFCMPGT(SrcSize, OpSize::i64Bit, MaxF, Src);
|
||||
Cmp = _VUShrNI(DstSize, OpSize::i64Bit, Cmp, 32);
|
||||
} else {
|
||||
Ref MaxF = LoadAndCacheNamedVectorConstant(DstSize, NAMED_VECTOR_CVTMAX_F32_I32);
|
||||
Converted = _Vector_FToZS(DstSize, OpSize::i32Bit, Src);
|
||||
Cmp = _VFCMPGT(DstSize, OpSize::i32Bit, MaxF, Src);
|
||||
}
|
||||
return _VBSL(DstSize, Cmp, Converted, MaxI);
|
||||
}
|
||||
|
||||
template<IR::OpSize SrcElementSize, bool HostRoundingMode>
|
||||
void OpDispatchBuilder::Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
|
||||
Ref Result {};
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
///< Special case for CVTTPD2DQ because it has weird rounding requirements.
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Result = _Vector_F64ToI32(DstSize, Src, HostRoundingMode ? Round_Host : Round_Towards_Zero, true);
|
||||
} else {
|
||||
Result = Vector_CVT_Float_To_IntImpl(Op, SrcElementSize, HostRoundingMode);
|
||||
}
|
||||
|
||||
Ref Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags);
|
||||
Ref Result = Vector_CVT_Float_To_Int32Impl(Op, DstSize, Src, OpSizeFromSrc(Op), SrcElementSize, HostRoundingMode, true);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i32Bit, false>(OpcodeArgs);
|
||||
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, true>(OpcodeArgs);
|
||||
template void OpDispatchBuilder::Vector_CVT_Float_To_Int<OpSize::i64Bit, false>(OpcodeArgs);
|
||||
@@ -2257,23 +2262,10 @@ void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int(OpcodeArgs) {
|
||||
// unnecessarily zero extend the vector. Otherwise, if
|
||||
// memory, then we want to load the element size exactly.
|
||||
const auto SrcSize = Op->Src[0].IsGPR() ? OpSize::i128Bit : OpSizeFromSrc(Op);
|
||||
const auto DstSize = OpSizeFromDst(Op);
|
||||
Ref Src = LoadSource_WithOpSize(FPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
|
||||
auto ElementSize = SrcElementSize;
|
||||
const auto Size = OpSizeFromDst(Op);
|
||||
|
||||
if (SrcElementSize == OpSize::i64Bit) {
|
||||
Src = _Vector_FToF(Size, SrcElementSize >> 1, Src, SrcElementSize);
|
||||
ElementSize = ElementSize >> 1;
|
||||
}
|
||||
|
||||
if constexpr (HostRoundingMode) {
|
||||
Src = _Vector_FToS(Size, ElementSize, Src);
|
||||
} else {
|
||||
Src = _Vector_FToZS(Size, ElementSize, Src);
|
||||
}
|
||||
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Src, Size, OpSize::iInvalid);
|
||||
Ref Result = Vector_CVT_Float_To_Int32Impl(Op, DstSize, Src, SrcSize, SrcElementSize, HostRoundingMode, false /* TODO? */);
|
||||
StoreResult_WithOpSize(FPRClass, Op, Op->Dest, Result, DstSize, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
template void OpDispatchBuilder::XMM_To_MMX_Vector_CVT_Float_To_Int<OpSize::i32Bit, false>(OpcodeArgs);
|
||||
|
||||
@@ -209,6 +209,22 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView
|
||||
return "x87_log10_2";
|
||||
case NamedVectorConstant::NAMED_VECTOR_X87_LOG_2:
|
||||
return "x87_log2";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F32_I32:
|
||||
return "cvtmax_f32_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F32_I32_UPPER:
|
||||
return "cvtmax_f32_i32_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F32_I64:
|
||||
return "cvtmax_f32_i64";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F64_I32:
|
||||
return "cvtmax_f64_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F64_I32_UPPER:
|
||||
return "cvtmax_f64_i32_upper";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_F64_I64:
|
||||
return "cvtmax_f64_i64";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_I32:
|
||||
return "cvtmax_i32";
|
||||
case NamedVectorConstant::NAMED_VECTOR_CVTMAX_I64:
|
||||
return "cvtmax_i64";
|
||||
default:
|
||||
return "<Unknown Named Vector Constant>";
|
||||
}
|
||||
|
||||
@@ -71,6 +71,15 @@ enum NamedVectorConstant : uint8_t {
|
||||
NAMED_VECTOR_X87_LOG10_2,
|
||||
NAMED_VECTOR_X87_LOG_2,
|
||||
|
||||
NAMED_VECTOR_CVTMAX_F32_I32,
|
||||
NAMED_VECTOR_CVTMAX_F32_I32_UPPER,
|
||||
NAMED_VECTOR_CVTMAX_F32_I64,
|
||||
NAMED_VECTOR_CVTMAX_F64_I32,
|
||||
NAMED_VECTOR_CVTMAX_F64_I32_UPPER,
|
||||
NAMED_VECTOR_CVTMAX_F64_I64,
|
||||
NAMED_VECTOR_CVTMAX_I32,
|
||||
NAMED_VECTOR_CVTMAX_I64,
|
||||
|
||||
NAMED_VECTOR_CONST_POOL_MAX,
|
||||
// Beginning of named constants that don't have a constant pool backing.
|
||||
NAMED_VECTOR_ZERO = NAMED_VECTOR_CONST_POOL_MAX,
|
||||
|
||||
Reference in new issue
Block a user