diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp index 96f61c197..5a39e6662 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp @@ -60,15 +60,13 @@ void OpDispatchBuilder::SetX87Top(Ref Value) { // Float LoaD operation with memory operand void OpDispatchBuilder::FLD(OpcodeArgs, IR::OpSize Width) { - const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width; - Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags); Ref ConvertedData = Data; // Convert to 80bit float if (Width == OpSize::i32Bit || Width == OpSize::i64Bit) { - ConvertedData = _F80CVTTo(Data, ReadWidth); + ConvertedData = _F80CVTTo(Data, Width); } - _PushStack(ConvertedData, Data, ReadWidth); + _PushStack(ConvertedData, Data, Width); } // Float LoaD operation with memory operand @@ -80,7 +78,7 @@ void OpDispatchBuilder::FBLD(OpcodeArgs) { // Read from memory Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags); Ref ConvertedData = _F80BCDLoad(Data); - _PushStack(ConvertedData, Data, OpSize::i128Bit); + _PushStack(ConvertedData, Invalid(), OpSize::iInvalid); } void OpDispatchBuilder::FBSTP(OpcodeArgs) { @@ -92,7 +90,7 @@ void OpDispatchBuilder::FBSTP(OpcodeArgs) { void OpDispatchBuilder::FLD_Const(OpcodeArgs, NamedVectorConstant K) { // Update TOP Ref Data = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, K); - _PushStack(Data, Data, OpSize::i128Bit); + _PushStack(Data, Data, OpSize::f80Bit); } void OpDispatchBuilder::FILD(OpcodeArgs) { @@ -123,11 +121,12 @@ void OpDispatchBuilder::FILD(OpcodeArgs) { auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent); Ref ConvertedData = _VLoadTwoGPRs(shifted, upper); - _PushStack(ConvertedData, Invalid(), ReadWidth); + _PushStack(ConvertedData, Invalid(), OpSize::iInvalid); } void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) { - const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit; + LOGMAN_THROW_A_FMT(Width == OpSize::i32Bit || Width == OpSize::i64Bit || Width == OpSize::f80Bit, "Invalid store width for FST"); + const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::f80Bit; AddressMode A = DecodeAddress(Op, Op->Dest, MemoryAccessType::DEFAULT, false); A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, false, false, Width); @@ -877,8 +876,8 @@ void OpDispatchBuilder::X87FXTRACT(OpcodeArgs) { _PopStackDestroy(); auto Exp = _F80XTRACT_EXP(Top); auto Sig = _F80XTRACT_SIG(Top); - _PushStack(Exp, Invalid(), OpSize::f80Bit); - _PushStack(Sig, Invalid(), OpSize::f80Bit); + _PushStack(Exp, Invalid(), OpSize::iInvalid); + _PushStack(Sig, Invalid(), OpSize::iInvalid); } } // namespace FEXCore::IR diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87F64.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87F64.cpp index 3b92cc00a..4e1175f60 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87F64.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87F64.cpp @@ -59,7 +59,6 @@ void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) { // F64 ops // Float load op with memory operand void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) { - const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width; Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags); // Convert to 64bit float Ref ConvertedData = Data; @@ -68,7 +67,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) { } else if (Width == OpSize::f80Bit) { ConvertedData = _F80CVT(OpSize::i64Bit, Data); } - _PushStack(ConvertedData, Data, ReadWidth); + _PushStack(ConvertedData, Data, Width); } void OpDispatchBuilder::FBLDF64(OpcodeArgs) { @@ -76,7 +75,7 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) { Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags); Ref ConvertedData = _F80BCDLoad(Data); ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData); - _PushStack(ConvertedData, Data, OpSize::i64Bit); + _PushStack(ConvertedData, Invalid(), OpSize::iInvalid); } void OpDispatchBuilder::FBSTPF64(OpcodeArgs) { @@ -100,7 +99,7 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) { Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data); } auto ConvertedData = _Float_FromGPR_S(OpSize::i64Bit, ReadWidth == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, Data); - _PushStack(ConvertedData, Invalid(), ReadWidth); + _PushStack(ConvertedData, Invalid(), OpSize::iInvalid); } void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) { @@ -397,7 +396,7 @@ void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) { Ref Exp = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, ExpZV, ExpNZV); _PopStackDestroy(); - _PushStack(Exp, Invalid(), OpSize::i64Bit); - _PushStack(Sig, Invalid(), OpSize::i64Bit); + _PushStack(Exp, Invalid(), OpSize::iInvalid); + _PushStack(Sig, Invalid(), OpSize::iInvalid); } } // namespace FEXCore::IR diff --git a/FEXCore/Source/Interface/IR/Passes/x87StackOptimizationPass.cpp b/FEXCore/Source/Interface/IR/Passes/x87StackOptimizationPass.cpp index 1a88d2518..f903b225a 100644 --- a/FEXCore/Source/Interface/IR/Passes/x87StackOptimizationPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/x87StackOptimizationPass.cpp @@ -65,7 +65,7 @@ public: int8_t TopOffset = 0; FixedSizeStack() - : buffer(FixedSizeStack::size, {StackSlot::UNUSED, T()}) {} + : buffer(FixedSizeStack::size, {StackSlot::UNUSED, T::Invalid}) {} void push(const T& Value) { rotate(); @@ -84,7 +84,7 @@ public: } void pop() { - buffer.front() = {StackSlot::INVALID, T()}; + buffer.front() = {StackSlot::INVALID, T::Invalid}; rotate(false); } @@ -102,7 +102,7 @@ public: void clear() { for (auto& Elem : buffer) { - Elem = {StackSlot::UNUSED, T()}; + Elem = {StackSlot::UNUSED, T::Invalid}; } TopOffset = 0; } @@ -170,13 +170,8 @@ private: // Helpers Ref RotateRight8(uint32_t V, Ref Amount); - void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode) { - Ref AddrNode = IR->GetNode(Op->Addr); - Ref Offset = IR->GetNode(Op->Offset); - OpSize Align = Op->Align; - MemOffsetType OffsetType = Op->OffsetType; - uint8_t OffsetScale = Op->OffsetScale; - + void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode, Ref AddrNode, Ref Offset, OpSize Align, MemOffsetType OffsetType, + uint8_t OffsetScale) { IREmit->_StoreMemFPR(OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale); auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1); @@ -191,7 +186,24 @@ private: IREmit->_StoreMemGPR(OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MemOffsetType::SXTX, A.IndexScale); } + void Store80BitToMem(const IROp_StoreStackMem* Op, Ref StackNode, Ref AddrNode, Ref Offset, OpSize Align, MemOffsetType OffsetType, + uint8_t OffsetScale) { + if (Features.SupportsSVE128 || Features.SupportsSVE256) { + AddressMode A {.Base = AddrNode, + .Index = Op->Offset.IsInvalid() ? nullptr : Offset, + .IndexType = MemOffsetType::SXTX, + .IndexScale = OffsetScale, + .AddrSize = OpSize::i64Bit}; + AddrNode = LoadEffectiveAddress(IREmit, A, GPROpSize, false); + IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode); + } else { + F80SplitStore_Helper(Op, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale); + } + } + void StoreStackMem_Helper(const IROp_StoreStackMem* Op, Ref StackNode) { + LOGMAN_THROW_A_FMT(!ReducedPrecisionMode, "Full precision mode expected."); + Ref AddrNode = IR->GetNode(Op->Addr); Ref Offset = IR->GetNode(Op->Offset); OpSize Align = Op->Align; @@ -208,17 +220,7 @@ private: } case OpSize::f80Bit: { - if (Features.SupportsSVE128 || Features.SupportsSVE256) { - AddressMode A {.Base = AddrNode, - .Index = Op->Offset.IsInvalid() ? nullptr : Offset, - .IndexType = MemOffsetType::SXTX, - .IndexScale = OffsetScale, - .AddrSize = OpSize::i64Bit}; - AddrNode = LoadEffectiveAddress(IREmit, A, GPROpSize, false); - IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode); - } else { // 80bit requires split-store - F80SplitStore_Helper(Op, StackNode); - } + Store80BitToMem(Op, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale); break; } default: ERROR_AND_DIE_FMT("Unsupported x87 size"); @@ -228,6 +230,8 @@ private: // Performs a store to memory from a value the stack passed in as StackNode. // This is the version dealing with the reduced precision case. void StoreStackMem_Reduced_Helper(const IROp_StoreStackMem* Op, Ref StackNode) { + LOGMAN_THROW_A_FMT(ReducedPrecisionMode, "Reduced precision mode expected."); + Ref AddrNode = IR->GetNode(Op->Addr); Ref Offset = IR->GetNode(Op->Offset); OpSize Align = Op->Align; @@ -244,10 +248,9 @@ private: break; } - // 80bit requires split-store case OpSize::f80Bit: { StackNode = IREmit->_F80CVTTo(StackNode, OpSize::i64Bit); - F80SplitStore_Helper(Op, StackNode); + Store80BitToMem(Op, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale); break; } default: ERROR_AND_DIE_FMT("Unsupported x87 size"); @@ -289,7 +292,7 @@ private: void Reset(); struct StackMemberInfo { - StackMemberInfo() {} + StackMemberInfo() = delete; StackMemberInfo(Ref Data) : StackDataNode(Data) {} StackMemberInfo(Ref Data, Ref Source, OpSize Size) @@ -301,6 +304,9 @@ private: OpSize Size; Ref Node; }; + + static const StackMemberInfo Invalid; + // Tuple is only valid if we have information about the Source of the Stack Data Node. // In it's valid then OpSize is the original source size and Ref is the original source node. std::optional Source {}; @@ -356,6 +362,8 @@ private: IRListView* IR = nullptr; }; +inline const X87StackOptimization::StackMemberInfo X87StackOptimization::StackMemberInfo::Invalid {nullptr}; + inline void X87StackOptimization::InvalidateCaches() { InvalidateCachedRegs(); ConstantPool.fill(nullptr); @@ -995,8 +1003,16 @@ void X87StackOptimization::Run(IREmitter* Emit) { // str w2, [x1] // or similar. As long as the source size and dest size are one and the same. // This will avoid any conversions between source and stack element size and conversion back. - if (!SlowPath && Value->Source && Value->Source->Size == Op->StoreSize) { - IREmit->_StoreMemFPR(Op->StoreSize, Value->Source->Node, AddrNode, Offset, Align, OffsetType, OffsetScale); + OpSize StoreSize = Op->StoreSize; + LOGMAN_THROW_A_FMT(Op->StoreSize == OpSize::i32Bit || Op->StoreSize == OpSize::i64Bit || Op->StoreSize == OpSize::f80Bit, + "Invalid store size in x87 store stack mem"); + if (!SlowPath && Value->Source && Value->Source->Size == StoreSize) { + Ref SourceValue = Value->Source->Node; + if (Op->StoreSize == OpSize::f80Bit) { + Store80BitToMem(Op, SourceValue, AddrNode, Offset, Align, OffsetType, OffsetScale); + } else { + IREmit->_StoreMemFPR(StoreSize, SourceValue, AddrNode, Offset, Align, OffsetType, OffsetScale); + } break; }