Refactoring of storing code in x87 opt. stack pass

Enables memcpy optimization of 80bit floats on reduced precision.

Also uncovered a bug where if we had done 80bit memcpy
optimization, we wouldn't have properly stored the 80bits.
This was caught by the existing tests when we enabled the optimization.
This commit is contained in:
Paulo Matos committed 2025-11-17 10:14:29 +01:00
1 parent 3c1b0bb917
commit 39dbf46422
3 files changed
+56 -42

No files matched your search

@@ -60,15 +60,13 @@ void OpDispatchBuilder::SetX87Top(Ref Value) {
// Float LoaD operation with memory operand
void OpDispatchBuilder::FLD(OpcodeArgs, IR::OpSize Width) {
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
Ref ConvertedData = Data;
// Convert to 80bit float
if (Width == OpSize::i32Bit || Width == OpSize::i64Bit) {
ConvertedData = _F80CVTTo(Data, ReadWidth);
ConvertedData = _F80CVTTo(Data, Width);
}
_PushStack(ConvertedData, Data, ReadWidth);
_PushStack(ConvertedData, Data, Width);
}
// Float LoaD operation with memory operand
@@ -80,7 +78,7 @@ void OpDispatchBuilder::FBLD(OpcodeArgs) {
// Read from memory
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
Ref ConvertedData = _F80BCDLoad(Data);
_PushStack(ConvertedData, Data, OpSize::i128Bit);
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
}
void OpDispatchBuilder::FBSTP(OpcodeArgs) {
@@ -92,7 +90,7 @@ void OpDispatchBuilder::FBSTP(OpcodeArgs) {
void OpDispatchBuilder::FLD_Const(OpcodeArgs, NamedVectorConstant K) {
// Update TOP
Ref Data = LoadAndCacheNamedVectorConstant(OpSize::i128Bit, K);
_PushStack(Data, Data, OpSize::i128Bit);
_PushStack(Data, Data, OpSize::f80Bit);
}
void OpDispatchBuilder::FILD(OpcodeArgs) {
@@ -123,11 +121,12 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
Ref ConvertedData = _VLoadTwoGPRs(shifted, upper);
_PushStack(ConvertedData, Invalid(), ReadWidth);
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
}
void OpDispatchBuilder::FST(OpcodeArgs, IR::OpSize Width) {
const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
LOGMAN_THROW_A_FMT(Width == OpSize::i32Bit || Width == OpSize::i64Bit || Width == OpSize::f80Bit, "Invalid store width for FST");
const auto SourceSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::f80Bit;
AddressMode A = DecodeAddress(Op, Op->Dest, MemoryAccessType::DEFAULT, false);
A = SelectAddressMode(this, A, GetGPROpSize(), CTX->HostFeatures.SupportsTSOImm9, false, false, Width);
@@ -877,8 +876,8 @@ void OpDispatchBuilder::X87FXTRACT(OpcodeArgs) {
_PopStackDestroy();
auto Exp = _F80XTRACT_EXP(Top);
auto Sig = _F80XTRACT_SIG(Top);
_PushStack(Exp, Invalid(), OpSize::f80Bit);
_PushStack(Sig, Invalid(), OpSize::f80Bit);
_PushStack(Exp, Invalid(), OpSize::iInvalid);
_PushStack(Sig, Invalid(), OpSize::iInvalid);
}
} // namespace FEXCore::IR
@@ -59,7 +59,6 @@ void OpDispatchBuilder::X87FLDCWF64(OpcodeArgs) {
// F64 ops
// Float load op with memory operand
void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
const auto ReadWidth = (Width == OpSize::f80Bit) ? OpSize::i128Bit : Width;
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], Width, Op->Flags);
// Convert to 64bit float
Ref ConvertedData = Data;
@@ -68,7 +67,7 @@ void OpDispatchBuilder::FLDF64(OpcodeArgs, IR::OpSize Width) {
} else if (Width == OpSize::f80Bit) {
ConvertedData = _F80CVT(OpSize::i64Bit, Data);
}
_PushStack(ConvertedData, Data, ReadWidth);
_PushStack(ConvertedData, Data, Width);
}
void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
@@ -76,7 +75,7 @@ void OpDispatchBuilder::FBLDF64(OpcodeArgs) {
Ref Data = LoadSourceFPR_WithOpSize(Op, Op->Src[0], OpSize::f80Bit, Op->Flags);
Ref ConvertedData = _F80BCDLoad(Data);
ConvertedData = _F80CVT(OpSize::i64Bit, ConvertedData);
_PushStack(ConvertedData, Data, OpSize::i64Bit);
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
}
void OpDispatchBuilder::FBSTPF64(OpcodeArgs) {
@@ -100,7 +99,7 @@ void OpDispatchBuilder::FILDF64(OpcodeArgs) {
Data = _Sbfe(OpSize::i64Bit, IR::OpSizeAsBits(ReadWidth), 0, Data);
}
auto ConvertedData = _Float_FromGPR_S(OpSize::i64Bit, ReadWidth == OpSize::i32Bit ? OpSize::i32Bit : OpSize::i64Bit, Data);
_PushStack(ConvertedData, Invalid(), ReadWidth);
_PushStack(ConvertedData, Invalid(), OpSize::iInvalid);
}
void OpDispatchBuilder::FISTF64(OpcodeArgs, bool Truncate) {
@@ -397,7 +396,7 @@ void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) {
Ref Exp = _NZCVSelectV(OpSize::i64Bit, CondClass::EQ, ExpZV, ExpNZV);
_PopStackDestroy();
_PushStack(Exp, Invalid(), OpSize::i64Bit);
_PushStack(Sig, Invalid(), OpSize::i64Bit);
_PushStack(Exp, Invalid(), OpSize::iInvalid);
_PushStack(Sig, Invalid(), OpSize::iInvalid);
}
} // namespace FEXCore::IR
@@ -65,7 +65,7 @@ public:
int8_t TopOffset = 0;
FixedSizeStack()
: buffer(FixedSizeStack::size, {StackSlot::UNUSED, T()}) {}
: buffer(FixedSizeStack::size, {StackSlot::UNUSED, T::Invalid}) {}
void push(const T& Value) {
rotate();
@@ -84,7 +84,7 @@ public:
}
void pop() {
buffer.front() = {StackSlot::INVALID, T()};
buffer.front() = {StackSlot::INVALID, T::Invalid};
rotate(false);
}
@@ -102,7 +102,7 @@ public:
void clear() {
for (auto& Elem : buffer) {
Elem = {StackSlot::UNUSED, T()};
Elem = {StackSlot::UNUSED, T::Invalid};
}
TopOffset = 0;
}
@@ -170,13 +170,8 @@ private:
// Helpers
Ref RotateRight8(uint32_t V, Ref Amount);
void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
Ref AddrNode = IR->GetNode(Op->Addr);
Ref Offset = IR->GetNode(Op->Offset);
OpSize Align = Op->Align;
MemOffsetType OffsetType = Op->OffsetType;
uint8_t OffsetScale = Op->OffsetScale;
void F80SplitStore_Helper(const IROp_StoreStackMem* Op, Ref StackNode, Ref AddrNode, Ref Offset, OpSize Align, MemOffsetType OffsetType,
uint8_t OffsetScale) {
IREmit->_StoreMemFPR(OpSize::i64Bit, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
auto Upper = IREmit->_VExtractToGPR(OpSize::i128Bit, OpSize::i64Bit, StackNode, 1);
@@ -191,7 +186,24 @@ private:
IREmit->_StoreMemGPR(OpSize::i16Bit, Upper, A.Base, A.Index, OpSize::i64Bit, MemOffsetType::SXTX, A.IndexScale);
}
void Store80BitToMem(const IROp_StoreStackMem* Op, Ref StackNode, Ref AddrNode, Ref Offset, OpSize Align, MemOffsetType OffsetType,
uint8_t OffsetScale) {
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
AddressMode A {.Base = AddrNode,
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
.IndexType = MemOffsetType::SXTX,
.IndexScale = OffsetScale,
.AddrSize = OpSize::i64Bit};
AddrNode = LoadEffectiveAddress(IREmit, A, GPROpSize, false);
IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode);
} else {
F80SplitStore_Helper(Op, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
}
}
void StoreStackMem_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
LOGMAN_THROW_A_FMT(!ReducedPrecisionMode, "Full precision mode expected.");
Ref AddrNode = IR->GetNode(Op->Addr);
Ref Offset = IR->GetNode(Op->Offset);
OpSize Align = Op->Align;
@@ -208,17 +220,7 @@ private:
}
case OpSize::f80Bit: {
if (Features.SupportsSVE128 || Features.SupportsSVE256) {
AddressMode A {.Base = AddrNode,
.Index = Op->Offset.IsInvalid() ? nullptr : Offset,
.IndexType = MemOffsetType::SXTX,
.IndexScale = OffsetScale,
.AddrSize = OpSize::i64Bit};
AddrNode = LoadEffectiveAddress(IREmit, A, GPROpSize, false);
IREmit->_StoreMemX87SVEOptPredicate(OpSize::i128Bit, OpSize::i16Bit, StackNode, AddrNode);
} else { // 80bit requires split-store
F80SplitStore_Helper(Op, StackNode);
}
Store80BitToMem(Op, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
break;
}
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
@@ -228,6 +230,8 @@ private:
// Performs a store to memory from a value the stack passed in as StackNode.
// This is the version dealing with the reduced precision case.
void StoreStackMem_Reduced_Helper(const IROp_StoreStackMem* Op, Ref StackNode) {
LOGMAN_THROW_A_FMT(ReducedPrecisionMode, "Reduced precision mode expected.");
Ref AddrNode = IR->GetNode(Op->Addr);
Ref Offset = IR->GetNode(Op->Offset);
OpSize Align = Op->Align;
@@ -244,10 +248,9 @@ private:
break;
}
// 80bit requires split-store
case OpSize::f80Bit: {
StackNode = IREmit->_F80CVTTo(StackNode, OpSize::i64Bit);
F80SplitStore_Helper(Op, StackNode);
Store80BitToMem(Op, StackNode, AddrNode, Offset, Align, OffsetType, OffsetScale);
break;
}
default: ERROR_AND_DIE_FMT("Unsupported x87 size");
@@ -289,7 +292,7 @@ private:
void Reset();
struct StackMemberInfo {
StackMemberInfo() {}
StackMemberInfo() = delete;
StackMemberInfo(Ref Data)
: StackDataNode(Data) {}
StackMemberInfo(Ref Data, Ref Source, OpSize Size)
@@ -301,6 +304,9 @@ private:
OpSize Size;
Ref Node;
};
static const StackMemberInfo Invalid;
// Tuple is only valid if we have information about the Source of the Stack Data Node.
// In it's valid then OpSize is the original source size and Ref is the original source node.
std::optional<StackMemberData> Source {};
@@ -356,6 +362,8 @@ private:
IRListView* IR = nullptr;
};
inline const X87StackOptimization::StackMemberInfo X87StackOptimization::StackMemberInfo::Invalid {nullptr};
inline void X87StackOptimization::InvalidateCaches() {
InvalidateCachedRegs();
ConstantPool.fill(nullptr);
@@ -995,8 +1003,16 @@ void X87StackOptimization::Run(IREmitter* Emit) {
// str w2, [x1]
// or similar. As long as the source size and dest size are one and the same.
// This will avoid any conversions between source and stack element size and conversion back.
if (!SlowPath && Value->Source && Value->Source->Size == Op->StoreSize) {
IREmit->_StoreMemFPR(Op->StoreSize, Value->Source->Node, AddrNode, Offset, Align, OffsetType, OffsetScale);
OpSize StoreSize = Op->StoreSize;
LOGMAN_THROW_A_FMT(Op->StoreSize == OpSize::i32Bit || Op->StoreSize == OpSize::i64Bit || Op->StoreSize == OpSize::f80Bit,
"Invalid store size in x87 store stack mem");
if (!SlowPath && Value->Source && Value->Source->Size == StoreSize) {
Ref SourceValue = Value->Source->Node;
if (Op->StoreSize == OpSize::f80Bit) {
Store80BitToMem(Op, SourceValue, AddrNode, Offset, Align, OffsetType, OffsetScale);
} else {
IREmit->_StoreMemFPR(StoreSize, SourceValue, AddrNode, Offset, Align, OffsetType, OffsetScale);
}
break;
}