From 10a02449b1dcf8e8d1ab5188ff21bf1392c893fe Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 12:41:22 -0400 Subject: [PATCH 01/14] IR: drop RA validation There's no reasonable way to keep this around without adding significant complexity to RA. This series prefers to drop complexity from RA, lessening the need for validation in the first place. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/CMakeLists.txt | 1 - FEXCore/Source/Interface/IR/PassManager.cpp | 1 - FEXCore/Source/Interface/IR/Passes.h | 1 - .../Source/Interface/IR/Passes/IRValidation.h | 4 - .../Interface/IR/Passes/RAValidation.cpp | 197 ------------------ 5 files changed, 204 deletions(-) delete mode 100644 FEXCore/Source/Interface/IR/Passes/RAValidation.cpp diff --git a/FEXCore/Source/CMakeLists.txt b/FEXCore/Source/CMakeLists.txt index f16689e84..0ac4f4efa 100644 --- a/FEXCore/Source/CMakeLists.txt +++ b/FEXCore/Source/CMakeLists.txt @@ -69,7 +69,6 @@ set (SRCS Interface/IR/Passes/ConstProp.cpp Interface/IR/Passes/IRDumperPass.cpp Interface/IR/Passes/IRValidation.cpp - Interface/IR/Passes/RAValidation.cpp Interface/IR/Passes/RedundantFlagCalculationElimination.cpp Interface/IR/Passes/RegisterAllocationPass.cpp Interface/IR/Passes/x87StackOptimizationPass.cpp diff --git a/FEXCore/Source/Interface/IR/PassManager.cpp b/FEXCore/Source/Interface/IR/PassManager.cpp index 337de1d21..2dd3ee0f2 100644 --- a/FEXCore/Source/Interface/IR/PassManager.cpp +++ b/FEXCore/Source/Interface/IR/PassManager.cpp @@ -79,7 +79,6 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) { void PassManager::AddDefaultValidationPasses() { #if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED InsertValidationPass(Validation::CreateIRValidation(), "IRValidation"); - InsertValidationPass(Validation::CreateRAValidation()); #endif } diff --git a/FEXCore/Source/Interface/IR/Passes.h b/FEXCore/Source/Interface/IR/Passes.h index ffdcb8f10..250892794 100644 --- a/FEXCore/Source/Interface/IR/Passes.h +++ b/FEXCore/Source/Interface/IR/Passes.h @@ -24,7 +24,6 @@ fextl::unique_ptr CreateX87StackOptimizationPass(const FEXCor namespace Validation { fextl::unique_ptr CreateIRValidation(); - fextl::unique_ptr CreateRAValidation(); } // namespace Validation namespace Debug { diff --git a/FEXCore/Source/Interface/IR/Passes/IRValidation.h b/FEXCore/Source/Interface/IR/Passes/IRValidation.h index 938225d85..6a34f05ff 100644 --- a/FEXCore/Source/Interface/IR/Passes/IRValidation.h +++ b/FEXCore/Source/Interface/IR/Passes/IRValidation.h @@ -16,8 +16,6 @@ struct BlockInfo { fextl::vector Successors; }; -class RAValidation; - class IRValidation final : public FEXCore::IR::Pass { public: ~IRValidation(); @@ -29,7 +27,5 @@ private: OrderedNode* EntryBlock {}; fextl::unordered_map OffsetToBlockMap; size_t MaxNodes {}; - - friend class RAValidation; }; } // namespace FEXCore::IR::Validation diff --git a/FEXCore/Source/Interface/IR/Passes/RAValidation.cpp b/FEXCore/Source/Interface/IR/Passes/RAValidation.cpp deleted file mode 100644 index 75e4ede6f..000000000 --- a/FEXCore/Source/Interface/IR/Passes/RAValidation.cpp +++ /dev/null @@ -1,197 +0,0 @@ -// SPDX-License-Identifier: MIT - -#include "Interface/IR/IR.h" -#include "Interface/IR/IREmitter.h" -#include "Interface/IR/PassManager.h" -#include "Interface/IR/RegisterAllocationData.h" -#include "Interface/IR/Passes/IRValidation.h" -#include "Interface/IR/Passes/RegisterAllocationPass.h" - -#include -#include -#include -#include -#include -#include - -#include - -namespace FEXCore::IR::Validation { - -// Hold the mapping of physical registers to the SSA id it holds at any given point in the IR -struct RegState { - static constexpr IR::NodeID UninitializedValue {0}; - static constexpr IR::NodeID InvalidReg {0xffff'ffff}; - - // This class makes some assumptions about how the host registers are arranged and mapped to virtual registers: - // 1. There will be less than 32 GPRs and 32 FPRs - // 2. If the GPRFixed class is used, there will be 16 GPRs and 16 FixedGPRs max - // 3. Same with FPRFixed - - // These assumptions were all true for the state of the arm64 and x86 jits at the time this was written - - // Mark a physical register as containing a SSA id - bool Set(PhysicalRegister Reg, IR::NodeID ssa) { - LOGMAN_THROW_A_FMT(ssa.IsValid(), "RegState assumes ssa0 will be the block header and never assigned to a register"); - - // PhysicalRegisters aren't fully mapped until assembly emission - // We need to apply a generic mapping here to catch any aliasing - switch (Reg.Class) { - case GPRClass: GPRs[Reg.Reg] = ssa; return true; - case GPRFixedClass: - // On arm64, there are 16 Fixed and 9 normal - GPRsFixed[Reg.Reg] = ssa; - return true; - case FPRClass: FPRs[Reg.Reg] = ssa; return true; - case FPRFixedClass: - // On arm64, there are 16 Fixed and 12 normal - FPRsFixed[Reg.Reg] = ssa; - return true; - } - return false; - } - - // Get the current SSA id - // Or an error value there isn't a (sane) SSA id - IR::NodeID Get(PhysicalRegister Reg) const { - switch (Reg.Class) { - case GPRClass: return GPRs[Reg.Reg]; - case GPRFixedClass: return GPRsFixed[Reg.Reg]; - case FPRClass: return FPRs[Reg.Reg]; - case FPRFixedClass: return FPRsFixed[Reg.Reg]; - } - return InvalidReg; - } - - // Mark a spill slot as containing a SSA id - void Spill(uint32_t SpillSlot, IR::NodeID ssa) { - Spills[SpillSlot] = ssa; - } - - // Return the SSA id currently in a spill slot - IR::NodeID Unspill(uint32_t SpillSlot) { - if (Spills.contains(SpillSlot)) { - return Spills[SpillSlot]; - } else { - return UninitializedValue; - } - } - -private: - std::array GPRsFixed = {}; - std::array FPRsFixed = {}; - std::array GPRs = {}; - std::array FPRs = {}; - - fextl::unordered_map Spills; -}; - -class RAValidation final : public FEXCore::IR::Pass { -public: - ~RAValidation() {} - void Run(IREmitter* IREmit) override; -}; - - -void RAValidation::Run(IREmitter* IREmit) { - if (!Manager->HasPass("RA")) { - return; - } - - FEXCORE_PROFILE_SCOPED("PassManager::RAValidation"); - - IR::RegisterAllocationData* RAData = Manager->GetPass("RA")->GetAllocationData(); - - bool HadError = false; - fextl::ostringstream Errors; - - auto CurrentIR = IREmit->ViewIR(); - - for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) { - // We only allocate registers locally, so state is reset each block - struct RegState BlockRegState = {}; - - for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) { - const auto ID = CurrentIR.GetID(CodeNode); - - const auto CheckArg = [&](uint32_t i, OrderedNodeWrapper Arg) { - const auto ArgID = Arg.ID(); - const auto PhyReg = RAData->GetNodeRegister(ArgID); - - if (PhyReg.IsInvalid()) { - return; - } - - auto CurrentSSAAtReg = BlockRegState.Get(PhyReg); - if (CurrentSSAAtReg == RegState::InvalidReg) { - HadError |= true; - Errors << fextl::fmt::format("%{}: Arg[{}] unknown Reg: {}, class: {}\n", ID, i, PhyReg.Reg, PhyReg.Class); - } else if (CurrentSSAAtReg == RegState::UninitializedValue) { - HadError |= true; - - Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it is uninitialized\n", ID, i, PhyReg.Reg, ArgID); - } else if (CurrentSSAAtReg != ArgID) { - HadError |= true; - Errors << fextl::fmt::format("%{}: Arg[{}] expects reg{} to contain %{}, but it actually contains %{}\n", ID, i, PhyReg.Reg, - ArgID, CurrentSSAAtReg); - } - }; - - switch (IROp->Op) { - case OP_SPILLREGISTER: { - auto SpillRegister = IROp->C(); - CheckArg(0, SpillRegister->Value); - - BlockRegState.Spill(SpillRegister->Slot, SpillRegister->Value.ID()); - break; - } - - case OP_FILLREGISTER: { - auto FillRegister = IROp->C(); - const auto ExpectedValue = FillRegister->OriginalValue.ID(); - const auto Value = BlockRegState.Unspill(FillRegister->Slot); - - // TODO: This only proves that the Spill has a consistent SSA value - // In the future we need to prove it contains the correct SSA value. For - // this we need to analyze copies/swaps properly. As a hot fix, don't - // compare Value with ExpectedValue. - - if (Value == RegState::UninitializedValue) { - HadError |= true; - Errors << fextl::fmt::format("%{}: FillRegister expected %{} in Slot {}, but was undefined\n", ID, ExpectedValue, FillRegister->Slot); - } - break; - } - - default: { - // And check that all args point at the correct SSA - uint8_t NumArgs = IR::GetArgs(IROp->Op); - for (uint32_t i = 0; i < NumArgs; ++i) { - CheckArg(i, IROp->Args[i]); - } - break; - } - } - - // Update BlockState map - if (IROp->Op != OP_SPILLREGISTER) { - BlockRegState.Set(RAData->GetNodeRegister(ID), ID); - } - } - } - - if (HadError) { - fextl::stringstream IrDump; - FEXCore::IR::Dump(&IrDump, &CurrentIR, RAData); - - LogMan::Msg::EFmt("RA Validation Error\n{}\nErrors:\n{}\n", IrDump.str(), Errors.str()); - LOGMAN_MSG_A_FMT("Encountered RA validation Error"); - - Errors.clear(); - } -} - -fextl::unique_ptr CreateRAValidation() { - return fextl::make_unique(); -} -} // namespace FEXCore::IR::Validation From 7eaf5ae9e077d54e3ec379883da1636ee8e137b2 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 12:45:02 -0400 Subject: [PATCH 02/14] IR: drop FillRegister original source this is now unused, and it's problematic with future work. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/Interface/IR/IR.json | 12 +++--------- .../Interface/IR/Passes/RegisterAllocationPass.cpp | 2 +- 2 files changed, 4 insertions(+), 10 deletions(-) diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index ff3667ab4..5aec513df 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -500,19 +500,13 @@ ] }, - "SSA = FillRegister OpSize:#Size, OpSize:#ElementSize, SSA:$OriginalValue, u32:$Slot, RegisterClass:$Class": { + "SSA = FillRegister OpSize:#Size, OpSize:#ElementSize, u32:$Slot, RegisterClass:$Class": { "Desc": ["Fills a register from a spill slot", "Spill slots are register allocated and has live ranges calculated to handle slot calculation", - "```diff\n- !Don't use this op. It is for RA to handle spilling and filling!\n```", - "", - "The OriginalValue SSA arg points at the original SSA value spilled, and only exists for", - "RA validation purposes" + "```diff\n- !Don't use this op. It is for RA to handle spilling and filling!\n```" ], "DestSize": "Size", - "ElementSize": "ElementSize", - "EmitValidation": [ - "WalkFindRegClass($OriginalValue) == $Class" - ] + "ElementSize": "ElementSize" }, "GPR = LoadNZCV": { diff --git a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp index c09e3b6c1..6c8526069 100644 --- a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp @@ -163,7 +163,7 @@ private: LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Old must have been spilled"); RegisterClassType RegClass = GetRegClassFromNode(IR, IROp); - return IREmit->_FillRegister(IROp->Size, IROp->ElementSize, Old, SlotPlusOne - 1, RegClass); + return IREmit->_FillRegister(IROp->Size, IROp->ElementSize, SlotPlusOne - 1, RegClass); }; // IP of next-use of each Old source. IPs are measured from the end of the From 261ae7a195b330ba606d98add1f284a899315a4b Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 14:34:09 -0400 Subject: [PATCH 03/14] IR: add immediates into OrderedNodeWrapper this will let us encode registers directly inside OrderedNodeWrapper, rather than pointers to OrderedNode *. that will let us speed up RA & onwards. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/Interface/IR/IR.h | 32 ++++++++++++++++++++++++++++++++ 1 file changed, 32 insertions(+) diff --git a/FEXCore/Source/Interface/IR/IR.h b/FEXCore/Source/Interface/IR/IR.h index 52ec1d1a5..ef02b19a0 100644 --- a/FEXCore/Source/Interface/IR/IR.h +++ b/FEXCore/Source/Interface/IR/IR.h @@ -140,22 +140,54 @@ struct FEX_PACKED NodeWrapperBase final { return NodeOffset == 0; } + [[nodiscard]] + bool IsImmediate() const { + return NodeOffset & (1u << 31); + } + + [[nodiscard]] + bool IsPointer() const { + return !IsImmediate(); + } + [[nodiscard]] Type* GetNode(uintptr_t Base) { + LOGMAN_THROW_A_FMT(IsPointer(), "Precondition"); return reinterpret_cast(Base + NodeOffset); } [[nodiscard]] const Type* GetNode(uintptr_t Base) const { + LOGMAN_THROW_A_FMT(IsPointer(), "Precondition"); return reinterpret_cast(Base + NodeOffset); } void SetOffset(uintptr_t Base, uintptr_t Value) { NodeOffset = Value - Base; + LOGMAN_THROW_A_FMT(IsPointer(), "Offsets are within 2GiB range"); + } + + void SetImmediate(uint32_t Immediate) { + LOGMAN_THROW_A_FMT(Immediate < (1u << 31), "Bounded"); + NodeOffset = Immediate | (1u << 31); + LOGMAN_THROW_A_FMT(IsImmediate(), "Encoded above"); + } + + [[nodiscard]] + uint32_t GetImmediate() const { + LOGMAN_THROW_A_FMT(IsImmediate(), "Precondition: must be an immediate"); + return NodeOffset & ~(1u << 31); } [[nodiscard]] friend constexpr bool operator==(const NodeWrapperBase&, const NodeWrapperBase&) = default; + + [[nodiscard]] + static NodeWrapperBase FromImmediate(uint32_t Immediate) { + NodeWrapperBase A; + A.SetImmediate(Immediate); + return A; + } }; static_assert(std::is_trivially_copyable_v>); From e9a8f8a9ff0627b80eae22433a9aa08ad02e66ba Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 14:29:15 -0400 Subject: [PATCH 04/14] IR: generate builders that take OrderedNodeWrapper when we need to materialize instructions inside RA without having a corresponding OrderedNode* source, we want to just pass a OrderedNodeWrapper with an encoded register. generate appropriate builders so this is possible. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Scripts/json_ir_generator.py | 79 ++++++++++++++++++++++------ 1 file changed, 64 insertions(+), 15 deletions(-) diff --git a/FEXCore/Scripts/json_ir_generator.py b/FEXCore/Scripts/json_ir_generator.py index beba7fbee..f668209c5 100755 --- a/FEXCore/Scripts/json_ir_generator.py +++ b/FEXCore/Scripts/json_ir_generator.py @@ -587,6 +587,15 @@ def print_ir_arg_printer(): output_file.write("#undef IROP_ARGPRINTER_HELPER\n") output_file.write("#endif\n") +def print_validation(op): + if op.EmitValidation != None: + output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n") + + for Validation in op.EmitValidation: + Sanitized = Validation.replace("\"", "\\\"") + output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized)) + output_file.write("\t\t#endif\n") + # Print out IR allocator helpers def print_ir_allocator_helpers(): output_file.write("#ifdef IROP_ALLOCATE_HELPERS\n") @@ -678,7 +687,7 @@ def print_ir_allocator_helpers(): output_file.write("{} {}".format(CType, arg.Name)); elif arg.IsSSA: # SSA value - output_file.write("OrderedNode *{}".format(arg.Name)) + output_file.write("OrderedNodeWrapper {}".format(arg.Name)) else: # User defined op that is stored CType = IRTypesToCXX[arg.Type].CXXName @@ -708,15 +717,9 @@ def print_ir_allocator_helpers(): output_file.write("\t\tauto _Op = AllocateOp();\n".format(op.Name, op.Name.upper())) if op.SSAArgNum != 0: - output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n") for arg in op.Arguments: if arg.IsSSA: - output_file.write("\t\t_Op.first->{} = {}->Wrapped(ListDataBegin);\n".format(arg.Name, arg.Name)) - - if op.SSAArgNum != 0: - for arg in op.Arguments: - if arg.IsSSA: - output_file.write("\t\t{}->AddUse();\n".format(arg.Name)) + output_file.write("\t\t_Op.first->{} = {};\n".format(arg.Name, arg.Name)) if len(op.Arguments) != 0: for arg in op.Arguments: @@ -735,18 +738,64 @@ def print_ir_allocator_helpers(): else: output_file.write("\t\t_Op.first->Header.ElementSize = {};\n".format(op.ElementSize)) - # Insert validation here - if op.EmitValidation != None: - output_file.write("\t\t#if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED\n") - for Validation in op.EmitValidation: - Sanitized = Validation.replace("\"", "\\\"") - output_file.write("\tLOGMAN_THROW_A_FMT({}, \"{}\");\n".format(Validation, Sanitized)) - output_file.write("\t\t#endif\n") + # Only validate here if there's no OrderedNode * version. Else + # validation is in that version, see the comment below. + if op.SSAArgNum == 0: + print_validation(op) output_file.write("\t\treturn _Op;\n") output_file.write("\t}\n\n") + # Now do the OrderedNode * version if necessary + if op.SSAArgNum: + output_file.write("\tIRPair _{}(" .format(op.Name, op.Name)) + + for i in range(0, len(op.Arguments)): + arg = op.Arguments[i] + LastArg = len(op.Arguments) - i - 1 == 0 + + if arg.Temporary: + CType = IRTypesToCXX[arg.Type].CXXName + output_file.write("{} {}".format(CType, arg.Name)); + elif arg.IsSSA: + output_file.write("OrderedNode *{}".format(arg.Name)) + else: + CType = IRTypesToCXX[arg.Type].CXXName + output_file.write("{} {}".format(CType, arg.Name)); + + if arg.DefaultInitializer != None: + output_file.write(" = {}".format(arg.DefaultInitializer)) + + if not LastArg: + output_file.write(", ") + + output_file.write(") {\n") + output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n") + + for arg in op.Arguments: + if arg.IsSSA: + output_file.write("\t\t{}->AddUse();\n".format(arg.Name)) + + # Insert validation here. This is skipped for the + # OrderedNodeWrapper version because validation can depend on + # the OrderedNode, but that's ok in practice. Everything pre-RA + # uses the OrderedNode version, and anything RA-onwards is + # dubious to validate. + print_validation(op) + + output_file.write(f"\t\treturn _{op.Name}(") + for i in range(0, len(op.Arguments)): + arg = op.Arguments[i] + LastArg = len(op.Arguments) - i - 1 == 0 + output_file.write(arg.Name) + if arg.IsSSA: + output_file.write("->Wrapped(ListDataBegin)") + if not LastArg: + output_file.write(", ") + output_file.write(");\n"); + output_file.write("\t}\n\n"); + output_file.write("#undef IROP_ALLOCATE_HELPERS\n") output_file.write("#endif\n") From c4f00a05dfcc55acdc622b26583153522e06ad65 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Mon, 19 May 2025 11:14:29 -0400 Subject: [PATCH 05/14] IR: add space for registers right in OrderedNode * This again follows the same idea of eliminating the RAData sideband. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/Interface/IR/IR.h | 11 ++++++++++- 1 file changed, 10 insertions(+), 1 deletion(-) diff --git a/FEXCore/Source/Interface/IR/IR.h b/FEXCore/Source/Interface/IR/IR.h index ef02b19a0..1225bbef5 100644 --- a/FEXCore/Source/Interface/IR/IR.h +++ b/FEXCore/Source/Interface/IR/IR.h @@ -227,6 +227,15 @@ public: OrderedNodeHeader Header; uint32_t NumUses; + // After RA, the register allocated for the node. This is the register for the + // node at the time it is written, even if it is shuffled into other registers + // later. In other words, it is the register destination of the instruction + // represented by this OrderedNode. + // + // This is the raw value of a PhysicalRegister data structure. + uint8_t Reg; + uint8_t Pad[3]; + using value_type = OrderedNodeWrapper; OrderedNode() = default; @@ -389,7 +398,7 @@ private: static_assert(std::is_trivially_constructible_v); static_assert(std::is_trivially_copyable_v); static_assert(offsetof(OrderedNode, Header) == 0); -static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + sizeof(uint32_t))); +static_assert(sizeof(OrderedNode) == (sizeof(OrderedNodeHeader) + 2 * sizeof(uint32_t))); // This is temporary. We are transitioning away from OrderedNode's in favour of // flat Ref words. To ease porting, we have this typedef. Eventually OrderedNode From 3136a5e2f82f61bd1dbae798e1962fdcb018f07d Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 14:36:56 -0400 Subject: [PATCH 06/14] IR: add helpers to extract physical registers from IR Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/Interface/IR/RegisterAllocationData.h | 6 ++++++ 1 file changed, 6 insertions(+) diff --git a/FEXCore/Source/Interface/IR/RegisterAllocationData.h b/FEXCore/Source/Interface/IR/RegisterAllocationData.h index 205293d1c..866ed1326 100644 --- a/FEXCore/Source/Interface/IR/RegisterAllocationData.h +++ b/FEXCore/Source/Interface/IR/RegisterAllocationData.h @@ -24,6 +24,12 @@ union PhysicalRegister { : Reg(Reg) , Class(Class.Val) {} + PhysicalRegister(OrderedNodeWrapper Arg) + : Raw(Arg.GetImmediate()) {} + + PhysicalRegister(Ref Node) + : Raw(Node->Reg) {} + static const PhysicalRegister Invalid() { return PhysicalRegister(InvalidClass, InvalidReg); } From a01e29ac99c5106366031f76100b617f7dfe1f05 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 14:33:27 -0400 Subject: [PATCH 07/14] IR: extend the IR header with RA info Beyond the actual registers allocated, there are two pieces of sideband data we store in the RAData object: * # of spill slots (explicitly) * whether RA has run (implicitly by the existence of RAData) We want to get rid of RAData, so we'll move these to the header. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/Interface/Core/Core.cpp | 2 +- FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp | 2 +- FEXCore/Source/Interface/IR/IR.json | 2 +- FEXCore/Source/Interface/IR/IntrusiveIRList.h | 10 ++++++++++ 4 files changed, 13 insertions(+), 3 deletions(-) diff --git a/FEXCore/Source/Interface/Core/Core.cpp b/FEXCore/Source/Interface/Core/Core.cpp index eb1b38fe7..4e090acd7 100644 --- a/FEXCore/Source/Interface/Core/Core.cpp +++ b/FEXCore/Source/Interface/Core/Core.cpp @@ -959,7 +959,7 @@ void ContextImpl::AddThunkTrampolineIRHandler(uintptr_t Entrypoint, uintptr_t Gu auto Result = AddCustomIREntrypoint( Entrypoint, [this, GuestThunkEntrypoint](uintptr_t Entrypoint, FEXCore::IR::IREmitter* emit) { - auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0); + auto IRHeader = emit->_IRHeader(emit->Invalid(), Entrypoint, 0, 0, 0, 0); auto Block = emit->CreateCodeNode(); IRHeader.first->Blocks = emit->WrapNode(Block); emit->SetCurrentCodeBlock(Block); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index 7164d6632..a08ed2b00 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -3933,7 +3933,7 @@ void OpDispatchBuilder::CreateJumpBlocks(const fextl::vector* Blocks, uint32_t NumInstructions) { Entry = RIP; - auto IRHeader = _IRHeader(InvalidNode, RIP, 0, NumInstructions); + auto IRHeader = _IRHeader(InvalidNode, RIP, 0, NumInstructions, 0, 0); CreateJumpBlocks(Blocks); auto Block = GetNewJumpBlock(RIP); diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index 5aec513df..6b8c0bbcc 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -169,7 +169,7 @@ "SwitchGen": false, "JITDispatchOverride": "NoOp" }, - "IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions, i1:$HasX87{false}, i1:$ReadsParity{false}": { + "IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions, u32:$SpillSlots, i1:$PostRA{false}, i1:$HasX87{false}, i1:$ReadsParity{false}": { "SwitchGen": false, "JITDispatchOverride": "NoOp" }, diff --git a/FEXCore/Source/Interface/IR/IntrusiveIRList.h b/FEXCore/Source/Interface/IR/IntrusiveIRList.h index 0cc49e6fe..3158d8c32 100644 --- a/FEXCore/Source/Interface/IR/IntrusiveIRList.h +++ b/FEXCore/Source/Interface/IR/IntrusiveIRList.h @@ -234,6 +234,16 @@ public: return GetOp(GetHeaderNode()); } + [[nodiscard]] + unsigned PostRA() const { + return GetHeader()->PostRA; + } + + [[nodiscard]] + unsigned SpillSlots() const { + return GetHeader()->SpillSlots; + } + template [[nodiscard]] T* GetOp(Ref Node) const { From f66bf3811e19fdfc79876c427371a8752b413117 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 14:45:27 -0400 Subject: [PATCH 08/14] RegisterAllocationPass: set PostRA flag Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp | 2 ++ 1 file changed, 2 insertions(+) diff --git a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp index 6c8526069..099170437 100644 --- a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp @@ -670,6 +670,8 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) { SSAToReg.clear(); SpillSlots.clear(); NextUses.clear(); + + IR->GetHeader()->PostRA = true; } fextl::unique_ptr CreateRegisterAllocationPass() { From bf1597920c68f9febdc5518a5b9fbd28170f8071 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 14:31:39 -0400 Subject: [PATCH 09/14] IR: use post-RA flag rather than implicitly depending on the RA data. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/Interface/Core/Core.cpp | 2 +- FEXCore/Source/Interface/IR/Passes/IRDumperPass.cpp | 6 +++--- 2 files changed, 4 insertions(+), 4 deletions(-) diff --git a/FEXCore/Source/Interface/Core/Core.cpp b/FEXCore/Source/Interface/Core/Core.cpp index 4e090acd7..b97252134 100644 --- a/FEXCore/Source/Interface/Core/Core.cpp +++ b/FEXCore/Source/Interface/Core/Core.cpp @@ -514,7 +514,7 @@ static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* fextl::stringstream out; auto NewIR = IREmitter->ViewIR(); FEXCore::IR::Dump(&out, &NewIR, RA); - fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", GuestRIP, out.str()); + fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", NewIR.PostRA() ? "post" : "pre", GuestRIP, out.str()); }; ContextImpl::GenerateIRResult diff --git a/FEXCore/Source/Interface/IR/Passes/IRDumperPass.cpp b/FEXCore/Source/Interface/IR/Passes/IRDumperPass.cpp index 80ce1960e..4e761c659 100644 --- a/FEXCore/Source/Interface/IR/Passes/IRDumperPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/IRDumperPass.cpp @@ -57,7 +57,7 @@ void IRDumper::Run(IREmitter* IREmit) { // DumpIRStr might be no if not dumping but ShouldDump is set in OpDisp if (DumpToFile) { - const auto fileName = fextl::fmt::format("{}/{:x}{}", DumpIR(), HeaderOp->OriginalRIP, RA ? "-post.ir" : "-pre.ir"); + const auto fileName = fextl::fmt::format("{}/{:x}{}", DumpIR(), HeaderOp->OriginalRIP, IR.PostRA() ? "-post.ir" : "-pre.ir"); FD = FEXCore::File::File(fileName.c_str(), FEXCore::File::FileModes::WRITE | FEXCore::File::FileModes::CREATE | FEXCore::File::FileModes::TRUNCATE); } @@ -66,9 +66,9 @@ void IRDumper::Run(IREmitter* IREmit) { fextl::stringstream out; FEXCore::IR::Dump(&out, &IR, RA); if (FD.IsValid()) { - fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", HeaderOp->OriginalRIP, out.str()); + fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", HeaderOp->OriginalRIP, out.str()); } else { - LogMan::Msg::IFmt("IR-{} 0x{:x}:\n{}\n@@@@@\n", RA ? "post" : "pre", HeaderOp->OriginalRIP, out.str()); + LogMan::Msg::IFmt("IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", HeaderOp->OriginalRIP, out.str()); } } } From 30f3b545af9be566e46d4599d1b49fdfe6710595 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 14:47:27 -0400 Subject: [PATCH 10/14] RegisterAllocationPass: use Header spill slots instead Removes even more RAData dependence. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/Interface/Core/JIT/JIT.cpp | 2 +- .../Source/Interface/IR/Passes/RegisterAllocationPass.cpp | 5 +---- 2 files changed, 2 insertions(+), 5 deletions(-) diff --git a/FEXCore/Source/Interface/Core/JIT/JIT.cpp b/FEXCore/Source/Interface/Core/JIT/JIT.cpp index 39ccf4b75..d0886a0cf 100644 --- a/FEXCore/Source/Interface/Core/JIT/JIT.cpp +++ b/FEXCore/Source/Interface/Core/JIT/JIT.cpp @@ -748,7 +748,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size EmitInterruptChecks(CheckTF); - SpillSlots = RAData->SpillSlots(); + SpillSlots = IR->SpillSlots(); if (SpillSlots) { const auto TotalSpillSlotsSize = SpillSlots * MaxSpillSlotSize; diff --git a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp index 099170437..d3b770548 100644 --- a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp @@ -170,7 +170,6 @@ private: // block, so we don't need to size the block up-front. fextl::vector NextUses; - unsigned SpillSlotCount; bool AnySpilled; bool IsValidArg(OrderedNodeWrapper Arg) { @@ -332,7 +331,7 @@ private: } // TODO: we should colour spill slots - uint32_t Slot = SpillSlotCount++; + uint32_t Slot = IR->GetHeader()->SpillSlots++; // We must map here in case we're spilling something we shuffled. auto SpillOp = IREmit->_SpillRegister(Map(Candidate), Slot, RegisterClassType {Reg.Class}); @@ -465,7 +464,6 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) { PreferredReg.resize(IR->GetSSACount(), PhysicalRegister::Invalid()); SSAToReg.resize(IR->GetSSACount(), PhysicalRegister::Invalid()); NextUses.resize(IR->GetSSACount(), 0); - SpillSlotCount = 0; AnySpilled = false; // Next-use distance relative to the block end of each source, last first. @@ -661,7 +659,6 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) { * TODO: Rework RegisterAllocationData to remove this memcpy, it's pointless. */ AllocData = RegisterAllocationData::Create(SSAToReg.size()); - AllocData->SpillSlotCount = SpillSlotCount; memcpy(AllocData->Map, SSAToReg.data(), sizeof(PhysicalRegister) * SSAToReg.size()); PreferredReg.clear(); From afce108ed77db133995da303f004cef52f75eeae Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 15:00:44 -0400 Subject: [PATCH 11/14] JIT: make almost all the DEF_OPs common this deduplicates a bunch of #defines, letting us change the signature easier. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/Interface/Core/JIT/ALUOps.cpp | 4 ---- FEXCore/Source/Interface/Core/JIT/AtomicOps.cpp | 2 -- FEXCore/Source/Interface/Core/JIT/BranchOps.cpp | 2 -- FEXCore/Source/Interface/Core/JIT/ConversionOps.cpp | 2 -- FEXCore/Source/Interface/Core/JIT/EncryptionOps.cpp | 2 -- FEXCore/Source/Interface/Core/JIT/JITClass.h | 2 ++ FEXCore/Source/Interface/Core/JIT/MemoryOps.cpp | 2 -- FEXCore/Source/Interface/Core/JIT/MiscOps.cpp | 2 -- FEXCore/Source/Interface/Core/JIT/MoveOps.cpp | 2 -- FEXCore/Source/Interface/Core/JIT/VectorOps.cpp | 3 --- 10 files changed, 2 insertions(+), 21 deletions(-) diff --git a/FEXCore/Source/Interface/Core/JIT/ALUOps.cpp b/FEXCore/Source/Interface/Core/JIT/ALUOps.cpp index 3538f8290..dbf9d8fbd 100644 --- a/FEXCore/Source/Interface/Core/JIT/ALUOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/ALUOps.cpp @@ -16,8 +16,6 @@ namespace FEXCore::CPU { #define GRD(Node) (IROp->Size <= 4 ? GetDst(Node) : GetDst(Node)) #define GRS(Node) (IROp->Size <= 4 ? GetReg(Node) : GetReg(Node)) -#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node) - #define DEF_BINOP_WITH_CONSTANT(FEXOp, VarOp, ConstOp) \ DEF_OP(FEXOp) { \ auto Op = IROp->C(); \ @@ -1651,6 +1649,4 @@ DEF_OP(FCmp) { fcmp(EmitSubSize, Scalar1, Scalar2); } -#undef DEF_OP - } // namespace FEXCore::CPU diff --git a/FEXCore/Source/Interface/Core/JIT/AtomicOps.cpp b/FEXCore/Source/Interface/Core/JIT/AtomicOps.cpp index 6878ac6d4..aad0f5b9d 100644 --- a/FEXCore/Source/Interface/Core/JIT/AtomicOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/AtomicOps.cpp @@ -10,7 +10,6 @@ $end_info$ #include "Interface/Core/JIT/JITClass.h" namespace FEXCore::CPU { -#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node) DEF_OP(CASPair) { auto Op = IROp->C(); LOGMAN_THROW_A_FMT(IROp->ElementSize == IR::OpSize::i32Bit || IROp->ElementSize == IR::OpSize::i64Bit, "Wrong element size"); @@ -473,5 +472,4 @@ DEF_OP(TelemetrySetValue) { #endif } -#undef DEF_OP } // namespace FEXCore::CPU diff --git a/FEXCore/Source/Interface/Core/JIT/BranchOps.cpp b/FEXCore/Source/Interface/Core/JIT/BranchOps.cpp index aa7b27f41..1e2ab0742 100644 --- a/FEXCore/Source/Interface/Core/JIT/BranchOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/BranchOps.cpp @@ -18,7 +18,6 @@ $end_info$ #include namespace FEXCore::CPU { -#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node) DEF_OP(CallbackReturn) { // spill back to CTX @@ -483,5 +482,4 @@ DEF_OP(XGetBV) { ubfx(ARMEmitter::Size::i64Bit, GetReg(Op->OutEDX), TMP1, 32, 32); } -#undef DEF_OP } // namespace FEXCore::CPU diff --git a/FEXCore/Source/Interface/Core/JIT/ConversionOps.cpp b/FEXCore/Source/Interface/Core/JIT/ConversionOps.cpp index c9921f29d..2a8bcac29 100644 --- a/FEXCore/Source/Interface/Core/JIT/ConversionOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/ConversionOps.cpp @@ -9,7 +9,6 @@ $end_info$ #include "Interface/Context/Context.h" namespace FEXCore::CPU { -#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node) DEF_OP(VInsGPR) { const auto Op = IROp->C(); const auto OpSize = IROp->Size; @@ -583,5 +582,4 @@ DEF_OP(Vector_F64ToI32) { } } -#undef DEF_OP } // namespace FEXCore::CPU diff --git a/FEXCore/Source/Interface/Core/JIT/EncryptionOps.cpp b/FEXCore/Source/Interface/Core/JIT/EncryptionOps.cpp index 086f0bd74..705d8fff8 100644 --- a/FEXCore/Source/Interface/Core/JIT/EncryptionOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/EncryptionOps.cpp @@ -8,7 +8,6 @@ $end_info$ #include "Interface/Core/JIT/JITClass.h" namespace FEXCore::CPU { -#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node) DEF_OP(VAESImc) { auto Op = IROp->C(); @@ -346,5 +345,4 @@ DEF_OP(PCLMUL) { } } -#undef DEF_OP } // namespace FEXCore::CPU diff --git a/FEXCore/Source/Interface/Core/JIT/JITClass.h b/FEXCore/Source/Interface/Core/JIT/JITClass.h index 6b9da6ea5..8bf86c984 100644 --- a/FEXCore/Source/Interface/Core/JIT/JITClass.h +++ b/FEXCore/Source/Interface/Core/JIT/JITClass.h @@ -383,6 +383,8 @@ private: #undef DEF_OP }; +#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node) + [[nodiscard]] fextl::unique_ptr CreateArm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread); diff --git a/FEXCore/Source/Interface/Core/JIT/MemoryOps.cpp b/FEXCore/Source/Interface/Core/JIT/MemoryOps.cpp index c62b45f5d..075ee185f 100644 --- a/FEXCore/Source/Interface/Core/JIT/MemoryOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/MemoryOps.cpp @@ -15,7 +15,6 @@ $end_info$ #include namespace FEXCore::CPU { -#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node) DEF_OP(LoadContext) { const auto Op = IROp->C(); @@ -2587,5 +2586,4 @@ DEF_OP(VLoadNonTemporal) { } } -#undef DEF_OP } // namespace FEXCore::CPU diff --git a/FEXCore/Source/Interface/Core/JIT/MiscOps.cpp b/FEXCore/Source/Interface/Core/JIT/MiscOps.cpp index db4da9c8e..a514900ad 100644 --- a/FEXCore/Source/Interface/Core/JIT/MiscOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/MiscOps.cpp @@ -16,7 +16,6 @@ $end_info$ #include namespace FEXCore::CPU { -#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node) DEF_OP(AllocateGPR) {} DEF_OP(AllocateGPRAfter) {} @@ -271,5 +270,4 @@ DEF_OP(Yield) { yield(); } -#undef DEF_OP } // namespace FEXCore::CPU diff --git a/FEXCore/Source/Interface/Core/JIT/MoveOps.cpp b/FEXCore/Source/Interface/Core/JIT/MoveOps.cpp index 627c96b5d..aeb0c2988 100644 --- a/FEXCore/Source/Interface/Core/JIT/MoveOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/MoveOps.cpp @@ -8,7 +8,6 @@ $end_info$ #include "Interface/Core/JIT/JITClass.h" namespace FEXCore::CPU { -#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node) DEF_OP(Copy) { auto Op = IROp->C(); @@ -39,5 +38,4 @@ DEF_OP(Swap2) { // Implemented above } -#undef DEF_OP } // namespace FEXCore::CPU diff --git a/FEXCore/Source/Interface/Core/JIT/VectorOps.cpp b/FEXCore/Source/Interface/Core/JIT/VectorOps.cpp index c79d470d8..b93270983 100644 --- a/FEXCore/Source/Interface/Core/JIT/VectorOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/VectorOps.cpp @@ -11,7 +11,6 @@ $end_info$ #include namespace FEXCore::CPU { -#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node) #define DEF_UNOP(FEXOp, ARMOp, ScalarCase) \ DEF_OP(FEXOp) { \ @@ -4586,6 +4585,4 @@ DEF_OP(VFCopySign) { } } - -#undef DEF_OP } // namespace FEXCore::CPU From 49f8332c5baf6d6e610acc84f651cd0e41ffdfb5 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 14:54:38 -0400 Subject: [PATCH 12/14] JIT: use registers directly from the IR This is the flag day change from the series, using all the new shiny infrastructre we added. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Source/Interface/Core/JIT/JIT.cpp | 35 ++++------- FEXCore/Source/Interface/Core/JIT/JITClass.h | 60 ++++++++++++------- FEXCore/Source/Interface/IR/IRDumper.cpp | 57 +++++++++--------- .../Interface/IR/Passes/IRValidation.cpp | 23 ++++--- .../IR/Passes/RegisterAllocationPass.cpp | 19 ++++++ 5 files changed, 113 insertions(+), 81 deletions(-) diff --git a/FEXCore/Source/Interface/Core/JIT/JIT.cpp b/FEXCore/Source/Interface/Core/JIT/JIT.cpp index d0886a0cf..6ef2a6677 100644 --- a/FEXCore/Source/Interface/Core/JIT/JIT.cpp +++ b/FEXCore/Source/Interface/Core/JIT/JIT.cpp @@ -80,7 +80,7 @@ static void PrintVectorValue(uint64_t Value, uint64_t ValueUpper) { namespace FEXCore::CPU { -void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::NodeID Node) { +void Arm64JITCore::Op_Unhandled(const IR::IROp_Header* IROp, IR::Ref Node) { FallbackInfo Info; if (!InterpreterOps::GetFallbackHandler(IROp, &Info)) { #if defined(ASSERTIONS_ENABLED) && ASSERTIONS_ENABLED @@ -493,7 +493,7 @@ static uint64_t Arm64JITCore_ExitFunctionLink(FEXCore::Core::CpuStateFrame* Fram return HostCode; } -void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::NodeID Node) {} +void Arm64JITCore::Op_NoOp(const IR::IROp_Header* IROp, IR::Ref Node) {} Arm64JITCore::Arm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread) : CPUBackend(Thread, INITIAL_CODE_SIZE, MAX_CODE_SIZE) @@ -580,6 +580,10 @@ void Arm64JITCore::ClearCache() { Arm64JITCore::~Arm64JITCore() {} bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const { + if (WNode.IsImmediate()) { + return false; + } + auto OpHeader = IR->GetOp(WNode); if (OpHeader->Op == IR::IROps::OP_INLINECONSTANT) { @@ -594,6 +598,10 @@ bool Arm64JITCore::IsInlineConstant(const IR::OrderedNodeWrapper& WNode, uint64_ } bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, uint64_t* Value) const { + if (WNode.IsImmediate()) { + return false; + } + auto OpHeader = IR->GetOp(WNode); if (OpHeader->Op == IR::IROps::OP_INLINEENTRYPOINTOFFSET) { @@ -612,22 +620,6 @@ bool Arm64JITCore::IsInlineEntrypointOffset(const IR::OrderedNodeWrapper& WNode, } } -FEXCore::IR::RegisterClassType Arm64JITCore::GetRegClass(IR::NodeID Node) const { - return FEXCore::IR::RegisterClassType {GetPhys(Node).Class}; -} - -bool Arm64JITCore::IsFPR(IR::NodeID Node) const { - auto Class = GetRegClass(Node); - - return Class == IR::FPRClass || Class == IR::FPRFixedClass; -} - -bool Arm64JITCore::IsGPR(IR::NodeID Node) const { - auto Class = GetRegClass(Node); - - return Class == IR::GPRClass || Class == IR::GPRFixedClass; -} - void Arm64JITCore::EmitInterruptChecks(bool CheckTF) { if (CheckTF) { ARMEmitter::ForwardLabel l_TFUnset; @@ -785,18 +777,17 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size } for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) { - const auto ID = IR->GetID(CodeNode); switch (IROp->Op) { #define REGISTER_OP_RT(op, x) \ - case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, ID); break + case FEXCore::IR::IROps::OP_##op: std::invoke(RT_##x, this, IROp, CodeNode); break #define REGISTER_OP(op, x) \ - case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, ID); break + case FEXCore::IR::IROps::OP_##op: Op_##x(IROp, CodeNode); break #define IROP_DISPATCH_DISPATCH #include #undef REGISTER_OP - default: Op_Unhandled(IROp, ID); break; + default: Op_Unhandled(IROp, CodeNode); break; } } diff --git a/FEXCore/Source/Interface/Core/JIT/JITClass.h b/FEXCore/Source/Interface/Core/JIT/JITClass.h index 8bf86c984..5c2b96d4e 100644 --- a/FEXCore/Source/Interface/Core/JIT/JITClass.h +++ b/FEXCore/Source/Interface/Core/JIT/JITClass.h @@ -66,9 +66,7 @@ private: fextl::map JumpTargets; [[nodiscard]] - ARMEmitter::Register GetReg(IR::NodeID Node) const { - const auto Reg = GetPhys(Node); - + ARMEmitter::Register GetReg(IR::PhysicalRegister Reg) const { LOGMAN_THROW_A_FMT(Reg.Class == IR::GPRFixedClass.Val || Reg.Class == IR::GPRClass.Val, "Unexpected Class: {}", Reg.Class); if (Reg.Class == IR::GPRFixedClass.Val) { @@ -81,14 +79,17 @@ private: } [[nodiscard]] - ARMEmitter::Register GetReg(IR::OrderedNodeWrapper Wrap) const { - return GetReg(Wrap.ID()); + ARMEmitter::Register GetReg(IR::Ref Node) const { + return GetReg(IR::PhysicalRegister(Node)); } [[nodiscard]] - ARMEmitter::VRegister GetVReg(IR::NodeID Node) const { - const auto Reg = GetPhys(Node); + ARMEmitter::Register GetReg(IR::OrderedNodeWrapper Wrap) const { + return GetReg(IR::PhysicalRegister(Wrap)); + } + [[nodiscard]] + ARMEmitter::VRegister GetVReg(IR::PhysicalRegister Reg) const { LOGMAN_THROW_A_FMT(Reg.Class == IR::FPRFixedClass.Val || Reg.Class == IR::FPRClass.Val, "Unexpected Class: {}", Reg.Class); if (Reg.Class == IR::FPRFixedClass.Val) { @@ -101,20 +102,18 @@ private: } [[nodiscard]] - ARMEmitter::VRegister GetVReg(IR::OrderedNodeWrapper Wrap) const { - return GetVReg(Wrap.ID()); + ARMEmitter::VRegister GetVReg(IR::Ref Node) const { + return GetVReg(IR::PhysicalRegister(Node)); } [[nodiscard]] - FEXCore::IR::RegisterClassType GetRegClass(IR::NodeID Node) const; + ARMEmitter::VRegister GetVReg(IR::OrderedNodeWrapper Wrap) const { + return GetVReg(IR::PhysicalRegister(Wrap)); + } [[nodiscard]] - IR::PhysicalRegister GetPhys(IR::NodeID Node) const { - auto PhyReg = RAData->GetNodeRegister(Node); - - LOGMAN_THROW_A_FMT(!PhyReg.IsInvalid(), "Couldn't Allocate register for node: ssa{}. Class: {}", Node, PhyReg.Class); - - return PhyReg; + FEXCore::IR::RegisterClassType GetRegClass(IR::Ref Node) const { + return FEXCore::IR::RegisterClassType {IR::PhysicalRegister(Node).Class}; } [[nodiscard]] @@ -234,18 +233,33 @@ private: } [[nodiscard]] - bool IsFPR(IR::NodeID Node) const; + bool IsFPR(IR::RegisterClassType Class) const { + return Class == IR::FPRClass || Class == IR::FPRFixedClass; + } + [[nodiscard]] - bool IsGPR(IR::NodeID Node) const; + bool IsGPR(IR::RegisterClassType Class) const { + return Class == IR::GPRClass || Class == IR::GPRFixedClass; + } + + [[nodiscard]] + bool IsGPR(IR::Ref Node) { + return IsGPR(GetRegClass(Node)); + } + + [[nodiscard]] + bool IsFPR(IR::Ref Node) { + return IsFPR(GetRegClass(Node)); + } [[nodiscard]] bool IsGPR(IR::OrderedNodeWrapper Wrap) { - return IsGPR(Wrap.ID()); + return IsGPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class}); } [[nodiscard]] bool IsFPR(IR::OrderedNodeWrapper Wrap) { - return IsFPR(Wrap.ID()); + return IsFPR(IR::RegisterClassType {IR::PhysicalRegister(Wrap).Class}); } [[nodiscard]] @@ -339,7 +353,7 @@ private: /** @} */ uint32_t SpillSlots {}; - using OpType = void (Arm64JITCore::*)(const IR::IROp_Header* IROp, IR::NodeID Node); + using OpType = void (Arm64JITCore::*)(const IR::IROp_Header* IROp, IR::Ref Node); using ScalarFMAOpCaller = std::function; @@ -366,7 +380,7 @@ private: OpType RT_LoadMemTSO; OpType RT_StoreMemTSO; -#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node) +#define DEF_OP(x) void Op_##x(IR::IROp_Header const* IROp, IR::Ref Node) // Dynamic Dispatcher supporting operations DEF_OP(ParanoidLoadMemTSO); @@ -383,7 +397,7 @@ private: #undef DEF_OP }; -#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::NodeID Node) +#define DEF_OP(x) void Arm64JITCore::Op_##x(IR::IROp_Header const* IROp, IR::Ref Node) [[nodiscard]] fextl::unique_ptr CreateArm64JITCore(FEXCore::Context::ContextImpl* ctx, FEXCore::Core::InternalThreadState* Thread); diff --git a/FEXCore/Source/Interface/IR/IRDumper.cpp b/FEXCore/Source/Interface/IR/IRDumper.cpp index 7e55159a8..cc11a860b 100644 --- a/FEXCore/Source/Interface/IR/IRDumper.cpp +++ b/FEXCore/Source/Interface/IR/IRDumper.cpp @@ -83,6 +83,26 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView } static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg, const IR::RegisterAllocationData* RAData) { + if (Arg.IsImmediate()) { + auto PhyReg = PhysicalRegister(Arg); + + switch (PhyReg.Class) { + case FEXCore::IR::GPRClass.Val: *out << "r"; break; + case FEXCore::IR::GPRFixedClass.Val: *out << "R"; break; + case FEXCore::IR::FPRClass.Val: *out << "v"; break; + case FEXCore::IR::FPRFixedClass.Val: *out << "V"; break; + case FEXCore::IR::ComplexClass.Val: *out << "c"; break; + case FEXCore::IR::InvalidClass.Val: *out << "invalid"; break; + default: *out << "unknown"; break; + } + + if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) { + *out << std::dec << (uint32_t)PhyReg.Reg; + } + + return; + } + auto [CodeNode, IROp] = IR->at(Arg)(); const auto ArgID = Arg.ID(); @@ -90,25 +110,6 @@ static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNode *out << "%Invalid"; } else { *out << "%" << std::dec << ArgID; - if (RAData) { - auto PhyReg = RAData->GetNodeRegister(ArgID); - - switch (PhyReg.Class) { - case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break; - case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break; - case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break; - case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break; - case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break; - case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break; - default: *out << "(Unknown"; break; - } - - if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) { - *out << std::dec << (uint32_t)PhyReg.Reg << ")"; - } else { - *out << ")"; - } - } } if (GetHasDest(IROp->Op)) { @@ -323,16 +324,16 @@ void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllo *out << "%" << std::dec << ID; - if (RAData) { - auto PhyReg = RAData->GetNodeRegister(ID); + auto PhyReg = PhysicalRegister(CodeNode); + if (!PhyReg.IsInvalid()) { switch (PhyReg.Class) { - case FEXCore::IR::GPRClass.Val: *out << "(GPR"; break; - case FEXCore::IR::GPRFixedClass.Val: *out << "(GPRFixed"; break; - case FEXCore::IR::FPRClass.Val: *out << "(FPR"; break; - case FEXCore::IR::FPRFixedClass.Val: *out << "(FPRFixed"; break; - case FEXCore::IR::ComplexClass.Val: *out << "(Complex"; break; - case FEXCore::IR::InvalidClass.Val: *out << "(Invalid"; break; - default: *out << "(Unknown"; break; + case FEXCore::IR::GPRClass.Val: *out << "(r"; break; + case FEXCore::IR::GPRFixedClass.Val: *out << "(R"; break; + case FEXCore::IR::FPRClass.Val: *out << "(v"; break; + case FEXCore::IR::FPRFixedClass.Val: *out << "(V"; break; + case FEXCore::IR::ComplexClass.Val: *out << "(complex"; break; + case FEXCore::IR::InvalidClass.Val: *out << "(invalid"; break; + default: *out << "(unknown"; break; } if (PhyReg.Class != FEXCore::IR::InvalidClass.Val) { *out << std::dec << (uint32_t)PhyReg.Reg << ")"; diff --git a/FEXCore/Source/Interface/IR/Passes/IRValidation.cpp b/FEXCore/Source/Interface/IR/Passes/IRValidation.cpp index dfdfc87e6..3879b98ab 100644 --- a/FEXCore/Source/Interface/IR/Passes/IRValidation.cpp +++ b/FEXCore/Source/Interface/IR/Passes/IRValidation.cpp @@ -94,9 +94,9 @@ void IRValidation::Run(IREmitter* IREmit) { Warnings << "%" << ID << ": Destination created but had no uses" << std::endl; } - if (RAData) { - // If we have a register allocator then the destination needs to be assigned a register and class - auto PhyReg = RAData->GetNodeRegister(ID); + if (CurrentIR.PostRA()) { + // After RA, the destination needs to be assigned a register and class + auto PhyReg = PhysicalRegister(CodeNode); FEXCore::IR::RegisterClassType ExpectedClass = IR::GetRegClass(IROp->Op); FEXCore::IR::RegisterClassType AssignedClass = FEXCore::IR::RegisterClassType {PhyReg.Class}; @@ -127,6 +127,10 @@ void IRValidation::Run(IREmitter* IREmit) { for (uint32_t i = 0; i < NumArgs; ++i) { OrderedNodeWrapper Arg = IROp->Args[i]; const auto ArgID = Arg.ID(); + if (Arg.IsImmediate()) { + continue; + } + IROps Op = CurrentIR.GetOp(Arg)->Op; if (ArgID.IsValid()) { @@ -239,11 +243,14 @@ void IRValidation::Run(IREmitter* IREmit) { } } - for (uint32_t i = 0; i < CurrentIR.GetSSACount(); i++) { - auto [Node, IROp] = CurrentIR.at(IR::NodeID {i})(); - if (Node->NumUses != Uses[i] && IROp->Op != OP_CODEBLOCK && IROp->Op != OP_IRHEADER) { - HadError |= true; - Errors << "%" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl; + // Use counts are only relevant pre-RA. + if (!CurrentIR.PostRA()) { + for (uint32_t i = 0; i < CurrentIR.GetSSACount(); i++) { + auto [Node, IROp] = CurrentIR.at(IR::NodeID {i})(); + if (Node->NumUses != Uses[i] && IROp->Op != OP_CODEBLOCK && IROp->Op != OP_IRHEADER) { + HadError |= true; + Errors << "%" << i << " Has " << Uses[i] << " Uses, but reports " << Node->NumUses << std::endl; + } } } diff --git a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp index d3b770548..4d22ac820 100644 --- a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp @@ -652,6 +652,25 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) { } LOGMAN_THROW_A_FMT(SourceIndex == 0, "Consistent source count in block"); + + // Finalize results for the block. This will go away. + for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) { + for (auto s = 0; s < IR::GetRAArgs(IROp->Op); ++s) { + if (IROp->Args[s].IsInvalid()) { + continue; + } + + auto Reg = SSAToReg[IROp->Args[s].ID().Value]; + + if (!Reg.IsInvalid()) { + IROp->Args[s].SetImmediate(Reg.Raw); + } + } + + if (GetHasDest(IROp->Op)) { + CodeNode->Reg = SSAToReg[IR->GetID(CodeNode).Value].Raw; + } + } } /* Now that we're done growing things, we can finalize our results. From 65ee1fafa8a694c747371a15981b92a5ce5cabdf Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 14:32:17 -0400 Subject: [PATCH 13/14] IR: drop RegisterAllocationData This sideband is now unused, registers are encoded directly in the IR. So we can garbage collect all this code for quite some savings. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Scripts/json_ir_generator.py | 2 +- FEXCore/Source/Interface/Context/Context.h | 2 - FEXCore/Source/Interface/Core/CPUBackend.h | 3 +- FEXCore/Source/Interface/Core/Core.cpp | 18 +++--- FEXCore/Source/Interface/Core/JIT/JIT.cpp | 4 +- FEXCore/Source/Interface/Core/JIT/JITClass.h | 6 +- FEXCore/Source/Interface/IR/AOTIR.cpp | 20 ++----- FEXCore/Source/Interface/IR/AOTIR.h | 7 +-- FEXCore/Source/Interface/IR/IR.h | 3 +- FEXCore/Source/Interface/IR/IRDumper.cpp | 4 +- FEXCore/Source/Interface/IR/IntrusiveIRList.h | 3 - FEXCore/Source/Interface/IR/Passes.h | 1 - .../Interface/IR/Passes/IRDumperPass.cpp | 8 +-- .../Interface/IR/Passes/IRValidation.cpp | 7 +-- .../IR/Passes/RegisterAllocationPass.cpp | 19 ------- .../IR/Passes/RegisterAllocationPass.h | 12 ---- .../Interface/IR/RegisterAllocationData.h | 57 ------------------- FEXCore/include/FEXCore/IR/IR.h | 1 - 18 files changed, 23 insertions(+), 154 deletions(-) diff --git a/FEXCore/Scripts/json_ir_generator.py b/FEXCore/Scripts/json_ir_generator.py index f668209c5..73dbd5577 100755 --- a/FEXCore/Scripts/json_ir_generator.py +++ b/FEXCore/Scripts/json_ir_generator.py @@ -575,7 +575,7 @@ def print_ir_arg_printer(): if arg.IsSSA: # SSA value - output_file.write("\tPrintArg(out, IR, Op->Header.Args[{}], RAData);\n".format(SSAArgNum)) + output_file.write("\tPrintArg(out, IR, Op->Header.Args[{}]);\n".format(SSAArgNum)) SSAArgNum = SSAArgNum + 1 else: # User defined op that is stored diff --git a/FEXCore/Source/Interface/Context/Context.h b/FEXCore/Source/Interface/Context/Context.h index 9c2de7d96..374aacd41 100644 --- a/FEXCore/Source/Interface/Context/Context.h +++ b/FEXCore/Source/Interface/Context/Context.h @@ -50,7 +50,6 @@ namespace HLE { } // namespace FEXCore namespace FEXCore::IR { -class RegisterAllocationData; struct IRListCopy; class IRListView; namespace Validation { @@ -271,7 +270,6 @@ public: struct GenerateIRResult { std::optional IRView; - IR::RegisterAllocationData* RAData; uint64_t TotalInstructions; uint64_t TotalInstructionsLength; uint64_t StartAddr; diff --git a/FEXCore/Source/Interface/Core/CPUBackend.h b/FEXCore/Source/Interface/Core/CPUBackend.h index 98b0bf3f8..15a60e7ba 100644 --- a/FEXCore/Source/Interface/Core/CPUBackend.h +++ b/FEXCore/Source/Interface/Core/CPUBackend.h @@ -18,7 +18,6 @@ namespace FEXCore { namespace IR { class IRListView; - class RegisterAllocationData; } // namespace IR namespace Core { @@ -119,7 +118,7 @@ namespace CPU { */ [[nodiscard]] virtual CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR, - FEXCore::Core::DebugData* DebugData, const FEXCore::IR::RegisterAllocationData* RAData, bool CheckTF) = 0; + FEXCore::Core::DebugData* DebugData, bool CheckTF) = 0; /** * @brief Relocates a block of code from the JIT code object cache diff --git a/FEXCore/Source/Interface/Core/Core.cpp b/FEXCore/Source/Interface/Core/Core.cpp index b97252134..36d0bd07f 100644 --- a/FEXCore/Source/Interface/Core/Core.cpp +++ b/FEXCore/Source/Interface/Core/Core.cpp @@ -509,11 +509,11 @@ void ContextImpl::ClearCodeCache(FEXCore::Core::InternalThreadState* Thread) { Thread->CPUBackend->ClearCache(); } -static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP, IR::RegisterAllocationData* RA) { +static void IRDumper(FEXCore::Core::InternalThreadState* Thread, IR::IREmitter* IREmitter, uint64_t GuestRIP) { FEXCore::File::File FD = FEXCore::File::File::GetStdERR(); fextl::stringstream out; auto NewIR = IREmitter->ViewIR(); - FEXCore::IR::Dump(&out, &NewIR, RA); + FEXCore::IR::Dump(&out, &NewIR); fextl::fmt::print(FD, "IR-ShouldDump-{} 0x{:x}:\n{}\n@@@@@\n", NewIR.PostRA() ? "post" : "pre", GuestRIP, out.str()); }; @@ -686,7 +686,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue if (HadDispatchError && TotalInstructions == 0) { // Couldn't handle any instruction in op dispatcher Thread->OpDispatcher->ResetWorkingList(); - return {{}, nullptr, 0, 0, 0, 0}; + return {{}, 0, 0, 0, 0}; } if (NeedsBlockEnd) { @@ -712,22 +712,19 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue auto ShouldDump = Thread->OpDispatcher->ShouldDumpIR(); // Debug if (ShouldDump) { - IRDumper(Thread, IREmitter, GuestRIP, nullptr); + IRDumper(Thread, IREmitter, GuestRIP); } // Run the passmanager over the IR from the dispatcher Thread->PassManager->Run(IREmitter); - auto RAData = Thread->PassManager->HasPass("RA") ? Thread->PassManager->GetPass("RA")->GetAllocationData() : nullptr; - // Debug if (ShouldDump) { - IRDumper(Thread, IREmitter, GuestRIP, RAData); + IRDumper(Thread, IREmitter, GuestRIP); } return { .IRView = IREmitter->ViewIR(), - .RAData = RAData, .TotalInstructions = TotalInstructions, .TotalInstructionsLength = TotalInstructionsLength, .StartAddr = Thread->FrontendDecoder->DecodedMinAddress, @@ -760,8 +757,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT } // Generate IR + Meta Info - auto [IRView, RAData, TotalInstructions, TotalInstructionsLength, StartAddr, Length] = - GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst); + auto [IRView, TotalInstructions, TotalInstructionsLength, StartAddr, Length] = GenerateIR(Thread, GuestRIP, Config.GDBSymbols(), MaxInst); if (!IRView) { return {nullptr, nullptr, 0, 0}; } @@ -772,7 +768,7 @@ ContextImpl::CompileCodeResult ContextImpl::CompileCode(FEXCore::Core::InternalT // Attempt to get the CPU backend to compile this code - auto CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &*IRView, DebugData.get(), RAData, TFSet); + auto CompiledCode = Thread->CPUBackend->CompileCode(GuestRIP, Length, TotalInstructions == 1, &*IRView, DebugData.get(), TFSet); // Release the IR Thread->OpDispatcher->DelayedDisownBuffer(); diff --git a/FEXCore/Source/Interface/Core/JIT/JIT.cpp b/FEXCore/Source/Interface/Core/JIT/JIT.cpp index 6ef2a6677..aa2513ecb 100644 --- a/FEXCore/Source/Interface/Core/JIT/JIT.cpp +++ b/FEXCore/Source/Interface/Core/JIT/JIT.cpp @@ -681,15 +681,13 @@ void Arm64JITCore::EmitInterruptChecks(bool CheckTF) { } CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR, - FEXCore::Core::DebugData* DebugData, const FEXCore::IR::RegisterAllocationData* RAData, - bool CheckTF) { + FEXCore::Core::DebugData* DebugData, bool CheckTF) { FEXCORE_PROFILE_SCOPED("Arm64::CompileCode"); JumpTargets.clear(); uint32_t SSACount = IR->GetSSACount(); this->Entry = Entry; - this->RAData = RAData; this->DebugData = DebugData; this->IR = IR; diff --git a/FEXCore/Source/Interface/Core/JIT/JITClass.h b/FEXCore/Source/Interface/Core/JIT/JITClass.h index 5c2b96d4e..2ebb8a08b 100644 --- a/FEXCore/Source/Interface/Core/JIT/JITClass.h +++ b/FEXCore/Source/Interface/Core/JIT/JITClass.h @@ -38,9 +38,8 @@ public: ~Arm64JITCore() override; [[nodiscard]] - CPUBackend::CompiledCode - CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR, FEXCore::Core::DebugData* DebugData, - const FEXCore::IR::RegisterAllocationData* RAData, bool CheckTF) override; + CPUBackend::CompiledCode CompileCode(uint64_t Entry, uint64_t Size, bool SingleInst, const FEXCore::IR::IRListView* IR, + FEXCore::Core::DebugData* DebugData, bool CheckTF) override; void ClearCache() override; @@ -292,7 +291,6 @@ private: // This is purely a debugging aid for developers to see if they are in JIT code space when inspecting raw memory void EmitDetectionString(); IR::RegisterAllocationPass* RAPass {}; - const IR::RegisterAllocationData* RAData {}; FEXCore::Core::DebugData* DebugData {}; void ResetStack(); diff --git a/FEXCore/Source/Interface/IR/AOTIR.cpp b/FEXCore/Source/Interface/IR/AOTIR.cpp index cdb9dbd1a..ce6c1e90f 100644 --- a/FEXCore/Source/Interface/IR/AOTIR.cpp +++ b/FEXCore/Source/Interface/IR/AOTIR.cpp @@ -47,19 +47,12 @@ AOTIRInlineEntry* AOTIRInlineIndex::Find(uint64_t GuestStart) { return nullptr; } -IR::RegisterAllocationData* AOTIRInlineEntry::GetRAData() { - return (IR::RegisterAllocationData*)InlineData; -} - IR::IRListView* AOTIRInlineEntry::GetIRData() { - auto RAData = GetRAData(); - auto Offset = RAData->Size(RAData->MapCount); - - return (IR::IRListView*)&InlineData[Offset]; + return (IR::IRListView*)InlineData; } void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, - const FEXCore::IR::IRListView& IRList, const FEXCore::IR::RegisterAllocationData* RAData) { + const FEXCore::IR::IRListView& IRList) { auto Inserted = Index.emplace(GuestRIP, Stream->Offset()); if (Inserted.second) { @@ -69,8 +62,6 @@ void AOTIRCaptureCacheEntry::AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t }; Stream->Write((const char*)&entry, sizeof(entry)); - RAData->Serialize(*Stream); - // IRData (inline) IRList.Serialize(*Stream); } @@ -265,9 +256,6 @@ class IRInlineStorage : public IRStorageBase { public: IRInlineStorage(AOTIRInlineEntry& entry) : entry(entry) {} - const RegisterAllocationData* RAData() override { - return entry.GetRAData(); - } IRListView GetIRView() override { return entry.GetIRData(); } @@ -332,7 +320,7 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre } // Add to AOT cache if aot generation is enabled - if (GeneratedIR && IR->RAData() && (CTX->Config.AOTIRCapture() || CTX->Config.AOTIRGenerate())) { + if (GeneratedIR && (CTX->Config.AOTIRCapture() || CTX->Config.AOTIRGenerate())) { auto hash = XXH3_64bits((void*)StartAddr, Length); @@ -356,7 +344,7 @@ bool AOTIRCaptureCache::PostCompileCode(FEXCore::Core::InternalThreadState* Thre uint64_t tag = FEXCore::IR::AOTIR_COOKIE; AotFile->Stream->Write(&tag, sizeof(tag)); } - AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IR->GetIRView(), IR->RAData()); + AotFile->AppendAOTIRCaptureCache(LocalRIP, LocalStartAddr, Length, hash, IR->GetIRView()); }); if (CTX->Config.AOTIRGenerate()) { diff --git a/FEXCore/Source/Interface/IR/AOTIR.h b/FEXCore/Source/Interface/IR/AOTIR.h index 3edab6e60..1fc7fd0ac 100644 --- a/FEXCore/Source/Interface/IR/AOTIR.h +++ b/FEXCore/Source/Interface/IR/AOTIR.h @@ -50,7 +50,6 @@ class ContextImpl; } namespace FEXCore::IR { -class RegisterAllocationData; class IRListView; constexpr auto COOKIE_VERSION = [](const char CookieText[4], uint32_t Version) { @@ -75,10 +74,9 @@ struct AOTIRInlineEntry { uint64_t GuestHash; uint64_t GuestLength; - /* RAData followed by IRData */ + /* IRData */ uint8_t InlineData[0]; - IR::RegisterAllocationData* GetRAData(); IR::IRListView* GetIRData(); }; @@ -100,8 +98,7 @@ struct AOTIRCaptureCacheEntry { fextl::unique_ptr Stream; fextl::map Index; - void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, const FEXCore::IR::IRListView& IRList, - const FEXCore::IR::RegisterAllocationData* RAData); + void AppendAOTIRCaptureCache(uint64_t GuestRIP, uint64_t Start, uint64_t Length, uint64_t Hash, const FEXCore::IR::IRListView& IRList); }; struct AOTIRCacheEntry { diff --git a/FEXCore/Source/Interface/IR/IR.h b/FEXCore/Source/Interface/IR/IR.h index 1225bbef5..5177a1bcc 100644 --- a/FEXCore/Source/Interface/IR/IR.h +++ b/FEXCore/Source/Interface/IR/IR.h @@ -11,7 +11,6 @@ namespace FEXCore::IR { class OrderedNode; class RegisterAllocationPass; -class RegisterAllocationData; /** * @brief The IROp_Header is an dynamically sized array @@ -766,7 +765,7 @@ inline NodeID NodeWrapperBase::ID() const { bool IsFragmentExit(FEXCore::IR::IROps Op); bool IsBlockExit(FEXCore::IR::IROps Op); -void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllocationData* RAData); +void Dump(fextl::stringstream* out, const IRListView* IR); } // namespace FEXCore::IR template<> diff --git a/FEXCore/Source/Interface/IR/IRDumper.cpp b/FEXCore/Source/Interface/IR/IRDumper.cpp index cc11a860b..a5c77c515 100644 --- a/FEXCore/Source/Interface/IR/IRDumper.cpp +++ b/FEXCore/Source/Interface/IR/IRDumper.cpp @@ -82,7 +82,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView } } -static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg, const IR::RegisterAllocationData* RAData) { +static void PrintArg(fextl::stringstream* out, const IRListView* IR, OrderedNodeWrapper Arg) { if (Arg.IsImmediate()) { auto PhyReg = PhysicalRegister(Arg); @@ -272,7 +272,7 @@ static void PrintArg(fextl::stringstream* out, [[maybe_unused]] const IRListView } } -void Dump(fextl::stringstream* out, const IRListView* IR, const IR::RegisterAllocationData* RAData) { +void Dump(fextl::stringstream* out, const IRListView* IR) { auto HeaderOp = IR->GetHeader(); int8_t CurrentIndent = 0; diff --git a/FEXCore/Source/Interface/IR/IntrusiveIRList.h b/FEXCore/Source/Interface/IR/IntrusiveIRList.h index 3158d8c32..af793cc6c 100644 --- a/FEXCore/Source/Interface/IR/IntrusiveIRList.h +++ b/FEXCore/Source/Interface/IR/IntrusiveIRList.h @@ -419,9 +419,6 @@ class IRStorageBase { public: virtual ~IRStorageBase() = default; - // Optional RA data. Returns nullptr if none present - virtual const RegisterAllocationData* RAData() = 0; - virtual IRListView GetIRView() = 0; }; diff --git a/FEXCore/Source/Interface/IR/Passes.h b/FEXCore/Source/Interface/IR/Passes.h index 250892794..94e22794a 100644 --- a/FEXCore/Source/Interface/IR/Passes.h +++ b/FEXCore/Source/Interface/IR/Passes.h @@ -15,7 +15,6 @@ class IntrusivePooledAllocator; namespace FEXCore::IR { class Pass; class RegisterAllocationPass; -class RegisterAllocationData; fextl::unique_ptr CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID); fextl::unique_ptr CreateDeadFlagCalculationEliminination(); diff --git a/FEXCore/Source/Interface/IR/Passes/IRDumperPass.cpp b/FEXCore/Source/Interface/IR/Passes/IRDumperPass.cpp index 4e761c659..0ac7da06a 100644 --- a/FEXCore/Source/Interface/IR/Passes/IRDumperPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/IRDumperPass.cpp @@ -38,12 +38,6 @@ IRDumper::IRDumper() { } void IRDumper::Run(IREmitter* IREmit) { - auto RAPass = Manager->GetPass("RA"); - IR::RegisterAllocationData* RA {}; - if (RAPass) { - RA = RAPass->GetAllocationData(); - } - FEXCore::File::File FD {}; if (DumpIR() == "stderr") { FD = FEXCore::File::File::GetStdERR(); @@ -64,7 +58,7 @@ void IRDumper::Run(IREmitter* IREmit) { if (FD.IsValid() || DumpToLog) { fextl::stringstream out; - FEXCore::IR::Dump(&out, &IR, RA); + FEXCore::IR::Dump(&out, &IR); if (FD.IsValid()) { fextl::fmt::print(FD, "IR-{} 0x{:x}:\n{}\n@@@@@\n", IR.PostRA() ? "post" : "pre", HeaderOp->OriginalRIP, out.str()); } else { diff --git a/FEXCore/Source/Interface/IR/Passes/IRValidation.cpp b/FEXCore/Source/Interface/IR/Passes/IRValidation.cpp index 3879b98ab..c201b1d8f 100644 --- a/FEXCore/Source/Interface/IR/Passes/IRValidation.cpp +++ b/FEXCore/Source/Interface/IR/Passes/IRValidation.cpp @@ -58,11 +58,6 @@ void IRValidation::Run(IREmitter* IREmit) { LOGMAN_THROW_A_FMT(HeaderOp->Header.Op == OP_IRHEADER, "First op wasn't IRHeader"); #endif - IR::RegisterAllocationData* RAData {}; - if (Manager->HasPass("RA")) { - RAData = Manager->GetPass("RA")->GetAllocationData(); - } - for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) { auto BlockIROp = BlockHeader->CW(); LOGMAN_THROW_A_FMT(BlockIROp->Header.Op == OP_CODEBLOCK, "IR type failed to be a code block"); @@ -257,7 +252,7 @@ void IRValidation::Run(IREmitter* IREmit) { HadWarning = false; if (HadError || HadWarning) { fextl::stringstream Out; - FEXCore::IR::Dump(&Out, &CurrentIR, RAData); + FEXCore::IR::Dump(&Out, &CurrentIR); if (HadError) { Out << "Errors:" << std::endl << Errors.str() << std::endl; diff --git a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp index 4d22ac820..37218284f 100644 --- a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp @@ -58,11 +58,7 @@ public: void Run(IREmitter* IREmit) override; void AddRegisters(IR::RegisterClassType Class, uint32_t RegisterCount) override; - RegisterAllocationData* GetAllocationData() override; - RegisterAllocationData::UniquePtr PullAllocationData() override; - private: - IR::RegisterAllocationData::UniquePtr AllocData; RegisterClass Classes[INVALID_CLASS]; IREmitter* IREmit; @@ -445,14 +441,6 @@ void ConstrainedRAPass::AddRegisters(IR::RegisterClassType Class, uint32_t Regis Classes[Class].Count = RegisterCount; } -RegisterAllocationData* ConstrainedRAPass::GetAllocationData() { - return AllocData.get(); -} - -RegisterAllocationData::UniquePtr ConstrainedRAPass::PullAllocationData() { - return std::move(AllocData); -} - void ConstrainedRAPass::Run(IREmitter* IREmit_) { FEXCORE_PROFILE_SCOPED("PassManager::RA"); @@ -673,13 +661,6 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) { } } - /* Now that we're done growing things, we can finalize our results. - * - * TODO: Rework RegisterAllocationData to remove this memcpy, it's pointless. - */ - AllocData = RegisterAllocationData::Create(SSAToReg.size()); - memcpy(AllocData->Map, SSAToReg.data(), sizeof(PhysicalRegister) * SSAToReg.size()); - PreferredReg.clear(); SSAToNewSSA.clear(); NewSSAToSSA.clear(); diff --git a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.h b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.h index d5e2ce798..a5f29de13 100644 --- a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.h +++ b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.h @@ -12,8 +12,6 @@ $end_info$ #include namespace FEXCore::IR { -class RegisterAllocationData; -struct RegisterAllocationDataDeleter; struct RegisterClassType; class RegisterAllocationPass : public FEXCore::IR::Pass { @@ -22,16 +20,6 @@ public: // Number of GPRs usable for pairs at start of GPR set. Must be even. uint32_t PairRegs; - - /** - * @brief Returns the register and class map array - */ - virtual RegisterAllocationData* GetAllocationData() = 0; - - /** - * @brief Returns and transfers ownership of the register and class map array - */ - virtual std::unique_ptr PullAllocationData() = 0; }; } // namespace FEXCore::IR diff --git a/FEXCore/Source/Interface/IR/RegisterAllocationData.h b/FEXCore/Source/Interface/IR/RegisterAllocationData.h index 866ed1326..bf11d78f5 100644 --- a/FEXCore/Source/Interface/IR/RegisterAllocationData.h +++ b/FEXCore/Source/Interface/IR/RegisterAllocationData.h @@ -41,61 +41,4 @@ union PhysicalRegister { static_assert(sizeof(PhysicalRegister) == 1); -struct RegisterAllocationDataDeleter; - -// This class is serialized, can't have any holes in the structure -// otherwise ASAN complains about reading uninitialized memory -class FEX_PACKED RegisterAllocationData { -public: - uint32_t SpillSlotCount {}; - uint32_t MapCount {}; - PhysicalRegister Map[0]; - - PhysicalRegister GetNodeRegister(NodeID Node) const { - return Map[Node.Value]; - } - uint32_t SpillSlots() const { - return SpillSlotCount; - } - - static size_t Size(uint32_t NodeCount) { - return sizeof(RegisterAllocationData) + NodeCount * sizeof(Map[0]); - } - - using UniquePtr = std::unique_ptr; - - static UniquePtr Create(uint32_t NodeCount); - - UniquePtr CreateCopy() const; - - void Serialize(FEXCore::Context::AOTIRWriter& stream) const { - stream.Write((const char*)&SpillSlotCount, sizeof(SpillSlotCount)); - stream.Write((const char*)&MapCount, sizeof(MapCount)); - // RAData (inline) - stream.Write((const char*)&Map[0], sizeof(Map[0]) * MapCount); - } -}; - -struct RegisterAllocationDataDeleter { - void operator()(RegisterAllocationData* r) const { - FEXCore::Allocator::free(r); - } -}; - -inline auto RegisterAllocationData::Create(uint32_t NodeCount) -> UniquePtr { - auto Ret = (RegisterAllocationData*)FEXCore::Allocator::malloc(Size(NodeCount)); - memset(&Ret->Map[0], PhysicalRegister::Invalid().Raw, NodeCount); - Ret->SpillSlotCount = 0; - Ret->MapCount = NodeCount; - return UniquePtr {Ret}; -} - -inline auto RegisterAllocationData::CreateCopy() const -> UniquePtr { - auto copy = (RegisterAllocationData*)FEXCore::Allocator::malloc(Size(MapCount)); - memcpy((void*)©->Map[0], (void*)&Map[0], MapCount * sizeof(Map[0])); - copy->SpillSlotCount = SpillSlotCount; - copy->MapCount = MapCount; - return UniquePtr {copy}; -} - } // namespace FEXCore::IR diff --git a/FEXCore/include/FEXCore/IR/IR.h b/FEXCore/include/FEXCore/IR/IR.h index 71daf1ed7..31d11fbf5 100644 --- a/FEXCore/include/FEXCore/IR/IR.h +++ b/FEXCore/include/FEXCore/IR/IR.h @@ -17,7 +17,6 @@ namespace FEXCore::IR { class OrderedNode; class RegisterAllocationPass; -class RegisterAllocationData; enum class SyscallFlags : uint8_t { DEFAULT = 0, From feb67658e14d0992a164f9c9c31de73665f0a99c Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Fri, 16 May 2025 14:56:44 -0400 Subject: [PATCH 14/14] RegisterAllocationPass: exploit new IR Now that we can just set registers directly, we can simplify RA a lot. All the Map/Unmap nonsense - it all goes away. We just assign registers as we go and everything clicks into place naturally. Signed-off-by: Alyssa Rosenzweig --- .../IR/Passes/RegisterAllocationPass.cpp | 211 +++++------------- 1 file changed, 55 insertions(+), 156 deletions(-) diff --git a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp index 37218284f..ea9633b45 100644 --- a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp @@ -28,9 +28,9 @@ namespace { uint32_t Available; uint32_t Count; - // If bit R of Available is 0, then RegToSSA[R] is the Old node - // currently allocated to R. Else, RegToSSA[R] is UNDEFINED, no need to - // clear this when freeing registers. + // If bit R of Available is 0, then RegToSSA[R] is the node currently + // allocated to R. Else, RegToSSA[R] is UNDEFINED, no need to clear this + // when freeing registers. Ref RegToSSA[32]; }; @@ -64,89 +64,21 @@ private: IREmitter* IREmit; IRListView* IR; - // Map of Old nodes to their preferred register, to coalesce load/store reg. + // Map of nodes to their preferred register, to coalesce load/store reg. fextl::vector PreferredReg; - // FEX's original RA could only assign a single register to a given def for - // its entire live range, and this limitation is baked deep into the IR. - // However, we split live ranges to implement register pairs and spilling. - // - // To reconcile, we generate new SSA nodes when we split live ranges, and - // remap SSA sources accordingly. This means SSAToReg can grow. - // - // We define "Old" nodes as nodes present in the original IR, and "New" nodes - // as nodes added to split live ranges. Helpful properties: - // - // - A node is Old <===> it is not New - // - A node is Old <===> its ID < IR.GetSSACount() at the start - // - All sources are Old before remapping an instruction - // - // SSAToNewSSA tracks the current remapping. nullptr indicates no remapping. - // - // Since its indexed by Old nodes, SSAToNewSSA does not grow after allocation. - fextl::vector SSAToNewSSA; - - // Inverse of SSAToNewSSA. Since it's indexed by new nodes, it grows. - fextl::vector NewSSAToSSA; - - // Map of assigned registers. Grows. + // Map of assigned registers. Does not grow beyond the initial set. fextl::vector SSAToReg; - bool IsOld(Ref Node) { - return IR->GetID(Node).Value < PreferredReg.size(); - }; - - // Return the New node (if it exists) for an Old node, else the Old node. - Ref Map(Ref Old) { - LOGMAN_THROW_A_FMT(IsOld(Old), "Pre-condition"); - - if (SSAToNewSSA.empty()) { - return Old; - } else { - return SSAToNewSSA[IR->GetID(Old).Value] ?: Old; - } - }; - - // Return the Old node for a possibly-remapped node. - Ref Unmap(Ref Node) { - if (NewSSAToSSA.empty()) { - return Node; - } else { - return NewSSAToSSA[IR->GetID(Node).Value] ?: Node; - } - }; - - // Record a remapping of Old to New. - void Remap(Ref Old, Ref New) { - LOGMAN_THROW_A_FMT(IsOld(Old) && !IsOld(New), "Pre-condition"); - - uint32_t OldID = IR->GetID(Old).Value; - uint32_t NewID = IR->GetID(New).Value; - - LOGMAN_THROW_A_FMT(NewID >= NewSSAToSSA.size(), "Brand new SSA def"); - NewSSAToSSA.resize(NewID + 1, 0); - - if (SSAToNewSSA.empty()) { - SSAToNewSSA.resize(PreferredReg.size(), nullptr); - } - - SSAToNewSSA[OldID] = New; - NewSSAToSSA[NewID] = Old; - - LOGMAN_THROW_A_FMT(Map(Old) == New && Unmap(New) == Old, "Post-condition"); - LOGMAN_THROW_A_FMT(Unmap(Old) == Old, "Invariant1"); - }; - - // Maps Old defs to their assigned spill slot + 1, or 0 if not spilled. + // Maps defs to their assigned spill slot + 1, or 0 if not spilled. fextl::vector SpillSlots; bool Rematerializable(IROp_Header* IROp) { return IROp->Op == OP_CONSTANT; } - Ref InsertFill(Ref Old) { - LOGMAN_THROW_A_FMT(IsOld(Old), "Precondition"); - IROp_Header* IROp = IR->GetOp(Old); + Ref InsertFill(Ref Node) { + IROp_Header* IROp = IR->GetOp(Node); // Remat if we can if (Rematerializable(IROp)) { @@ -155,14 +87,14 @@ private: } // Otherwise fill from stack - uint32_t SlotPlusOne = SpillSlots[IR->GetID(Old).Value]; - LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Old must have been spilled"); + uint32_t SlotPlusOne = SpillSlots[IR->GetID(Node).Value]; + LOGMAN_THROW_A_FMT(SlotPlusOne >= 1, "Node must have been spilled"); RegisterClassType RegClass = GetRegClassFromNode(IR, IROp); return IREmit->_FillRegister(IROp->Size, IROp->ElementSize, SlotPlusOne - 1, RegClass); }; - // IP of next-use of each Old source. IPs are measured from the end of the + // IP of next-use of each source. IPs are measured from the end of the // block, so we don't need to size the block up-front. fextl::vector NextUses; @@ -192,13 +124,14 @@ private: return 1 << Reg.Reg; }; - bool IsInRegisterFile(Ref Old) { - LOGMAN_THROW_A_FMT(IsOld(Old), "Precondition"); + bool IsInRegisterFile(Ref Node) { + auto ID = IR->GetID(Node).Value; + LOGMAN_THROW_A_FMT(ID < SSAToReg.size(), "Only old nodes looked up"); - PhysicalRegister Reg = SSAToReg[IR->GetID(Map(Old)).Value]; + PhysicalRegister Reg = SSAToReg[ID]; RegisterClass* Class = GetClass(Reg); - return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Old; + return (Class->Available & GetRegBits(Reg)) == 0 && Class->RegToSSA[Reg.Reg] == Node; }; void FreeReg(PhysicalRegister Reg) { @@ -210,14 +143,9 @@ private: Class->Available |= RegBits; }; - bool HasSource(IROp_Header* I, Ref Old) { - LOGMAN_THROW_A_FMT(IsOld(Old), "Invariant2"); - + bool HasSource(IROp_Header* I, PhysicalRegister Reg) { for (auto s = 0; s < IR::GetRAArgs(I->Op); ++s) { - Ref Node = IR->GetNode(I->Args[s]); - LOGMAN_THROW_A_FMT(IsOld(Node), "not yet mapped"); - - if (Node == Old) { + if (I->Args[s].IsImmediate() && PhysicalRegister(I->Args[s]) == Reg) { return true; } } @@ -283,33 +211,33 @@ private: uint32_t Allocated = ((1u << Class->Count) - 1) & ~Class->Available; foreach_bit(i, Allocated) { - Ref Old = Class->RegToSSA[i]; + Ref Node = Class->RegToSSA[i]; + auto Reg = SSAToReg[IR->GetID(Node).Value]; - LOGMAN_THROW_A_FMT(Old != nullptr, "Invariant3"); - LOGMAN_THROW_A_FMT(SSAToReg[IR->GetID(Map(Old)).Value].Reg == i, "Invariant4"); + LOGMAN_THROW_A_FMT(Node != nullptr, "Invariant3"); + LOGMAN_THROW_A_FMT(Reg.Reg == i, "Invariant4"); // Skip any source used by the current instruction, it is unspillable. - if (!HasSource(Exclude, Old)) { - uint32_t NextUse = NextUses[IR->GetID(Old).Value]; + if (!HasSource(Exclude, Reg)) { + uint32_t NextUse = NextUses[IR->GetID(Node).Value]; // Prioritize remat over spilling. It is typically cheaper to remat a // constant multiple times than to spill a single value. - if (!Rematerializable(IR->GetOp(Old))) { + if (!Rematerializable(IR->GetOp(Node))) { NextUse += 100000; } if (NextUse < BestDistance) { BestDistance = NextUse; BestReg = i; - Candidate = Old; + Candidate = Node; } } } LOGMAN_THROW_A_FMT(Candidate != nullptr, "must've found something.."); - LOGMAN_THROW_A_FMT(IsOld(Candidate), "Invariant5"); - PhysicalRegister Reg = SSAToReg[IR->GetID(Map(Candidate)).Value]; + PhysicalRegister Reg = SSAToReg[IR->GetID(Candidate).Value]; LOGMAN_THROW_A_FMT(Reg.Reg == BestReg, "Invariant6"); IROp_Header* Header = IR->GetOp(Candidate); @@ -330,7 +258,7 @@ private: uint32_t Slot = IR->GetHeader()->SpillSlots++; // We must map here in case we're spilling something we shuffled. - auto SpillOp = IREmit->_SpillRegister(Map(Candidate), Slot, RegisterClassType {Reg.Class}); + auto SpillOp = IREmit->_SpillRegister(OrderedNodeWrapper::FromImmediate(Reg.Raw), Slot, RegisterClassType {Reg.Class}); SpillOp.first->Header.Size = Header->Size; SpillOp.first->Header.ElementSize = Header->ElementSize; SpillSlots[Value] = Slot + 1; @@ -341,22 +269,27 @@ private: AnySpilled = true; }; + void RemapReg(Ref Node, PhysicalRegister Reg) { + RegisterClass* Class = GetClass(Reg); + Class->RegToSSA[Reg.Reg] = Node; + + uint32_t Index = IR->GetID(Node).Value; + if (Index < SSAToReg.size()) { + SSAToReg[Index] = Reg; + } + }; + // Record a given assignment of register Reg to Node. void SetReg(Ref Node, PhysicalRegister Reg) { - uint32_t Index = IR->GetID(Node).Value; RegisterClass* Class = GetClass(Reg); uint32_t RegBits = GetRegBits(Reg); LOGMAN_THROW_A_FMT((Class->Available & RegBits) == RegBits, "Precondition"); Class->Available &= ~RegBits; - Class->RegToSSA[Reg.Reg] = Unmap(Node); - if (Index >= SSAToReg.size()) { - SSAToReg.resize(Index + 1, PhysicalRegister::Invalid()); - } - - SSAToReg[Index] = Reg; + RemapReg(Node, Reg); + Node->Reg = Reg.Raw; }; // Assign a register for a given Node, spilling if necessary. @@ -378,7 +311,7 @@ private: // Try to handle tied registers. This can fail, the JIT will insert moves. if (int TiedIdx = IR::TiedSource(IROp->Op); TiedIdx >= 0) { - PhysicalRegister Reg = SSAToReg[IROp->Args[TiedIdx].ID().Value]; + auto Reg = PhysicalRegister(IROp->Args[TiedIdx]); RegisterClass* Class = GetClass(Reg); uint32_t RegBits = GetRegBits(Reg); @@ -408,7 +341,7 @@ private: } } else if (IROp->Op == OP_ALLOCATEGPRAFTER) { uint32_t Available = Classes[GPRClass].Available; - auto After = SSAToReg[IR->GetID(IR->GetNode(IROp->Args[0])).Value]; + auto After = PhysicalRegister(IROp->Args[0]); if ((After.Reg & 1) == 0 && Available & (1ull << (After.Reg + 1))) { SetReg(CodeNode, PhysicalRegister(GPRClass, After.Reg + 1)); return; @@ -448,7 +381,6 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) { auto IR_ = IREmit->ViewIR(); IR = &IR_; - // SSAToNewSSA, NewSSAToSSA allocated on first-use PreferredReg.resize(IR->GetSSACount(), PhysicalRegister::Invalid()); SSAToReg.resize(IR->GetSSACount(), PhysicalRegister::Invalid()); NextUses.resize(IR->GetSSACount(), 0); @@ -554,23 +486,20 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) { if (!(Class->Available & (1u << Reg.Reg))) { Ref Old = Class->RegToSSA[Reg.Reg]; - LOGMAN_THROW_A_FMT(IsOld(Old), "RegToSSA invariant"); - LOGMAN_THROW_A_FMT(IsOld(Node), "Haven't remapped this instruction"); - if (Old != Node) { IREmit->SetWriteCursorBefore(CodeNode); Ref Copy; if (Reg.Class == FPRFixedClass) { IROp_Header* Header = IR->GetOp(Old); - Copy = IREmit->_VMov(Header->Size, Map(Old)); + Copy = IREmit->_VMov(Header->Size, OrderedNodeWrapper::FromImmediate(Reg.Raw)); } else { - Copy = IREmit->_Copy(Map(Old)); + Copy = IREmit->_Copy(OrderedNodeWrapper::FromImmediate(Reg.Raw)); } - Remap(Old, Copy); FreeReg(Reg); AssignReg(IR->GetOp(Copy), Copy, IROp); + RemapReg(Old, PhysicalRegister(Copy)); } } } @@ -586,14 +515,13 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) { } Ref Old = IR->GetNode(IROp->Args[s]); - LOGMAN_THROW_A_FMT(IsOld(Old), "before remapping"); if (!IsInRegisterFile(Old)) { IREmit->SetWriteCursorBefore(CodeNode); Ref Fill = InsertFill(Old); - Remap(Old, Fill); AssignReg(IR->GetOp(Fill), Fill, IROp); + RemapReg(Old, PhysicalRegister(Fill)); } } } @@ -603,20 +531,23 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) { continue; } + Ref Node = IR->GetNode(IROp->Args[s]); + auto ID = IR->GetID(Node).Value; + auto Reg = SSAToReg[ID]; + SourceIndex--; LOGMAN_THROW_A_FMT(SourceIndex >= 0, "Consistent source count"); - if (!SourcesNextUses[SourceIndex]) { - Ref Old = IR->GetNode(IROp->Args[s]); - auto Reg = SSAToReg[IR->GetID(Map(Old)).Value]; + if (!Reg.IsInvalid()) { + IROp->Args[s].SetImmediate(Reg.Raw); - if (!Reg.IsInvalid()) { - LOGMAN_THROW_A_FMT(IsInRegisterFile(Old), "sources in file"); + if (!SourcesNextUses[SourceIndex]) { + LOGMAN_THROW_A_FMT(IsInRegisterFile(Node), "sources in file"); FreeReg(Reg); } } - NextUses[IROp->Args[s].ID().Value] = SourcesNextUses[SourceIndex]; + NextUses[ID] = SourcesNextUses[SourceIndex]; } // Assign destinations. @@ -624,46 +555,14 @@ void ConstrainedRAPass::Run(IREmitter* IREmit_) { AssignReg(IROp, CodeNode, IROp); } - // Remap sources last, since AssignReg can shuffle. - if (!SSAToNewSSA.empty()) { - for (auto s = 0; s < IR::GetRAArgs(IROp->Op); ++s) { - Ref Remapped = SSAToNewSSA[IROp->Args[s].ID().Value]; - - if (Remapped != nullptr) { - IREmit->ReplaceNodeArgument(CodeNode, s, Remapped); - } - } - } - LOGMAN_THROW_A_FMT(IP >= 1, "IP relative to end of block, iterating forward"); --IP; } LOGMAN_THROW_A_FMT(SourceIndex == 0, "Consistent source count in block"); - - // Finalize results for the block. This will go away. - for (auto [CodeNode, IROp] : IR->GetCode(BlockNode)) { - for (auto s = 0; s < IR::GetRAArgs(IROp->Op); ++s) { - if (IROp->Args[s].IsInvalid()) { - continue; - } - - auto Reg = SSAToReg[IROp->Args[s].ID().Value]; - - if (!Reg.IsInvalid()) { - IROp->Args[s].SetImmediate(Reg.Raw); - } - } - - if (GetHasDest(IROp->Op)) { - CodeNode->Reg = SSAToReg[IR->GetID(CodeNode).Value].Raw; - } - } } PreferredReg.clear(); - SSAToNewSSA.clear(); - NewSSAToSSA.clear(); SSAToReg.clear(); SpillSlots.clear(); NextUses.clear();