From f0a434c1677f15ac04e986b992fe39f185cb0217 Mon Sep 17 00:00:00 2001 From: Alyssa Rosenzweig Date: Wed, 6 Aug 2025 14:51:53 -0400 Subject: [PATCH] IR: inline as we go Augment IREmitter to inline constants as we generate code, rather than needing a later clean up pass. This replaces the last function of ConstProp. Signed-off-by: Alyssa Rosenzweig --- FEXCore/Scripts/json_ir_generator.py | 29 +++++++++- .../Interface/Core/OpcodeDispatcher.cpp | 2 +- FEXCore/Source/Interface/IR/IREmitter.h | 56 ++++++++++++++++++- 3 files changed, 83 insertions(+), 4 deletions(-) diff --git a/FEXCore/Scripts/json_ir_generator.py b/FEXCore/Scripts/json_ir_generator.py index 73dbd5577..51d98f5fa 100755 --- a/FEXCore/Scripts/json_ir_generator.py +++ b/FEXCore/Scripts/json_ir_generator.py @@ -58,6 +58,7 @@ class OpDefinition: JITDispatch: bool JITDispatchOverride: str TiedSource: int + Inline: list Arguments: list EmitValidation: list Desc: list @@ -278,6 +279,12 @@ def parse_ops(ops): if "TiedSource" in op_val: OpDef.TiedSource = op_val["TiedSource"] + # Pad Inline out to the argument count + OpDef.Inline = [''] * len(OpDef.Arguments) + if "Inline" in op_val: + Value = op_val["Inline"] + OpDef.Inline[0:len(Value)] = Value + # Do some fixups of the data here if len(OpDef.EmitValidation) != 0: for i in range(len(OpDef.EmitValidation)): @@ -773,9 +780,29 @@ def print_ir_allocator_helpers(): output_file.write(") {\n") output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n") + idx = 0 for arg in op.Arguments: if arg.IsSSA: - output_file.write("\t\t{}->AddUse();\n".format(arg.Name)) + # Inline an immediate if we can + inline = op.Inline[idx] + idx += 1 + + if inline != '': + Sized = "Size" in [x.Name for x in op.Arguments] + P = ["Size" if Sized else "OpSize::i64Bit", arg.Name] + + # A few cases need extra info plumbed. + if inline == "SubtractZero": + P += ["Src2"] + elif inline == "Mem": + P += ["OffsetType", "OffsetScale"] + elif inline == "Memtso": + P += ["OffsetType", "OffsetScale", "true /* TSO */"] + inline = "Mem" + + output_file.write(f"\t\t{arg.Name} = Inline{inline}({', '.join(P)});\n") + + output_file.write(f"\t\t{arg.Name}->AddUse();\n") # Insert validation here. This is skipped for the # OrderedNodeWrapper version because validation can depend on diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index b6cbb5746..7eab787c8 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -4316,7 +4316,7 @@ void OpDispatchBuilder::StoreResult(FEXCore::IR::RegisterClassType Class, FEXCor } OpDispatchBuilder::OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx) - : IREmitter {ctx->OpDispatcherAllocator} + : IREmitter {ctx->OpDispatcherAllocator, ctx->HostFeatures.SupportsTSOImm9} , CTX {ctx} { ResetWorkingList(); InstallHostSpecificOpcodeHandlers(); diff --git a/FEXCore/Source/Interface/IR/IREmitter.h b/FEXCore/Source/Interface/IR/IREmitter.h index 4a6c769c6..093f67356 100644 --- a/FEXCore/Source/Interface/IR/IREmitter.h +++ b/FEXCore/Source/Interface/IR/IREmitter.h @@ -24,8 +24,9 @@ class IREmitter { friend class FEXCore::IR::PassManager; public: - IREmitter(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator) - : DualListData {ThreadAllocator, 8 * 1024 * 1024} { + IREmitter(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator, bool SupportsTSOImm9) + : DualListData {ThreadAllocator, 8 * 1024 * 1024} + , SupportsTSOImm9(SupportsTSOImm9) { ReownOrClaimBuffer(); ResetWorkingList(); } @@ -52,6 +53,56 @@ public: FEXCore::IR::RegisterClassType WalkFindRegClass(Ref Node); + // These inlining helpers are used by IRDefines.inc so define first. + Ref InlineMem(OpSize Size, Ref Offset, MemOffsetType OffsetType, uint8_t& OffsetScale, bool TSO = false) { + uint64_t Imm {}; + if (OffsetType != MEM_OFFSET_SXTX || !IsValueConstant(WrapNode(Offset), &Imm)) { + return Offset; + } + + // The immediate may be scaled in the IR, we need to correct for that. + Imm *= OffsetScale; + + // Signed immediate unscaled 9-bit range for both regular and LRCPC2 ops. + bool IsSIMM9 = ((int64_t)Imm >= -256) && ((int64_t)Imm <= 255); + IsSIMM9 &= (SupportsTSOImm9 || !TSO); + + // Extended offsets for regular loadstore only. + LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i256Bit, "Must be sized"); + + bool IsExtended = (Imm & (IR::OpSizeToSize(Size) - 1)) == 0 && Imm / IR::OpSizeToSize(Size) <= 4095; + IsExtended &= !TSO; + + if (IsSIMM9 || IsExtended) { + OffsetScale = 1; + return _InlineConstant(Imm); + } else { + return Offset; + } + } + +#define DEF_INLINE(Type, Variable, Filter) \ + Ref Inline##Type(OpSize Size, Ref Source) { \ + uint64_t Variable; \ + if (IsValueConstant(WrapNode(Source), &Variable) && (Filter)) { \ + return _InlineConstant(Variable); \ + } else { \ + return Source; \ + } \ + } + + DEF_INLINE(Any, _, true) + DEF_INLINE(Zero, X, X == 0) + DEF_INLINE(AddSub, X, ARMEmitter::IsImmAddSub(X)) + DEF_INLINE(LargeAddSub, X, ARMEmitter::IsImmAddSub(X) && Size >= OpSize::i32Bit); + DEF_INLINE(Logical, X, ARMEmitter::Emitter::IsImmLogical(X, std::max((int)IR::OpSizeAsBits(Size), 32))); + + Ref InlineSubtractZero(OpSize Size, Ref Src1, Ref Src2) { + // Only inline a zero if we won't inline the other source. + return IsValueConstant(WrapNode(Src2)) ? Src1 : InlineZero(Size, Src1); + } +#undef DEF_INLINE + // These handlers add cost to the constructor and destructor // If it becomes an issue then blow them away // GCC also generates some pretty atrocious code around these @@ -409,6 +460,7 @@ protected: Ref CurrentCodeBlock {}; fextl::vector CodeBlocks; uint64_t Entry {}; + bool SupportsTSOImm9 {}; }; } // namespace FEXCore::IR