/* $info$ tags: ir|opts desc: ConstProp, ZExt elim, addressgen coalesce, const pooling, fcmp reduction, const inlining $end_info$ */ #if JIT_ARM64 //aarch64 heuristics #include "aarch64/assembler-aarch64.h" #include "aarch64/cpu-aarch64.h" #include "aarch64/disasm-aarch64.h" #include "aarch64/assembler-aarch64.h" #endif #include "Interface/IR/PassManager.h" #include #include #include #include #include #include #include #include #include #include #include #include #include #include namespace FEXCore::IR { template uint64_t getMask(T Op) { uint64_t NumBits = Op->Header.Size * 8; return (~0ULL) >> (64 - NumBits); } template<> uint64_t getMask(IROp_Header* Op) { uint64_t NumBits = Op->Size * 8; return (~0ULL) >> (64 - NumBits); } // Returns true if the number bits from [0:width) contain the same bit. // Ensuring that the consecutive bits in the range are entirely 0 or 1. static bool HasConsecutiveBits(uint64_t imm, unsigned width) { if (width == 0) { return true; } // Credit to https://github.com/dougallj for this implementation. return ((imm ^ (imm >> 1)) & ((1ULL << (width - 1)) - 1)) == 0; } #if JIT_ARM64 //aarch64 heuristics static bool IsImmLogical(uint64_t imm, unsigned width) { if (width < 32) width = 32; return vixl::aarch64::Assembler::IsImmLogical(imm, width); } static bool IsImmAddSub(uint64_t imm) { return vixl::aarch64::Assembler::IsImmAddSub(imm); } static bool IsMemoryScale(uint64_t Scale, uint8_t AccessSize) { return Scale == AccessSize; } #elif JIT_X86_64 // very lazy heuristics static bool IsImmLogical(uint64_t imm, unsigned width) { return imm < 0x8000'0000; } static bool IsImmAddSub(uint64_t imm) { return imm < 0x8000'0000; } static bool IsMemoryScale(uint64_t Scale, uint8_t AccessSize) { return Scale == 1 || Scale == 2 || Scale == 4 || Scale == 8; } #else #error No inline constant heuristics for this target #endif static bool IsImmMemory(uint64_t imm, uint8_t AccessSize) { if ( ((int64_t)imm >= -255) && ((int64_t)imm <= 256) ) return true; else if ( (imm & (AccessSize-1)) == 0 && imm/AccessSize <= 4095 ) return true; else { return false; } } static bool IsTSOImm9(uint64_t imm) { // RCPC2 only has a 9-bit signed offset if ( ((int64_t)imm >= -256) && ((int64_t)imm <= 255) ) return true; else { return false; } } static std::tuple MemExtendedAddressing(IREmitter *IREmit, uint8_t AccessSize, IROp_Header* AddressHeader) { auto Src0Header = IREmit->GetOpHeader(AddressHeader->Args[0]); if (Src0Header->Size == 8) { //Try to optimize: Base + MUL(Offset, Scale) if (Src0Header->Op == OP_MUL) { uint64_t Scale; if (IREmit->IsValueConstant(Src0Header->Args[1], &Scale)) { if (IsMemoryScale(Scale, AccessSize)) { // remove mul as it can be folded to the mem op return { MEM_OFFSET_SXTX, (uint8_t)Scale, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) }; } else if (Scale == 1) { // remove nop mul return { MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) }; } } } //Try to optimize: Base + LSHL(Offset, Scale) else if (Src0Header->Op == OP_LSHL) { uint64_t Constant2; if (IREmit->IsValueConstant(Src0Header->Args[1], &Constant2)) { uint64_t Scale = 1<UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) }; } else if (Scale == 1) { // remove nop shift return { MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) }; } } } #if defined(_M_ARM_64) // x86 can't sext or zext on mem ops //Try to optimize: Base + (u32)Offset else if (Src0Header->Op == OP_BFE) { auto Bfe = Src0Header->C(); if (Bfe->lsb == 0 && Bfe->Width == 32) { //todo: arm can also scale here return { MEM_OFFSET_UXTW, 1, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) }; } } //Try to optimize: Base + (s32)Offset else if (Src0Header->Op == OP_SBFE) { auto Sbfe = Src0Header->C(); if (Sbfe->lsb == 0 && Sbfe->Width == 32) { //todo: arm can also scale here return { MEM_OFFSET_SXTW, 1, IREmit->UnwrapNode(AddressHeader->Args[1]), IREmit->UnwrapNode(Src0Header->Args[0]) }; } } #endif } // no match anywhere, just add return { MEM_OFFSET_SXTX, 1, IREmit->UnwrapNode(AddressHeader->Args[0]), IREmit->UnwrapNode(AddressHeader->Args[1]) }; } static OrderedNodeWrapper RemoveUselessMasking(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t mask) { #if 1 // HOTFIX: We need to clear up the meaning of opsize and dest size. See #594 return src; #else auto IROp = IREmit->GetOpHeader(src); if (IROp->Op == OP_AND) { auto Op = IROp->C(); uint64_t imm; if (IREmit->IsValueConstant(IROp->Args[1], &imm) && ((imm & mask) == mask)) { return RemoveUselessMasking(IREmit, IROp->Args[0], mask); } } else if (IROp->Op == OP_BFE) { auto Op = IROp->C(); if (Op->lsb == 0) { uint64_t imm = 1ULL << (Op->Width-1); imm = (imm-1) *2 + 1; if ((imm & mask) == mask) { return RemoveUselessMasking(IREmit, IROp->Args[0], mask); } } } return src; #endif } static bool IsBfeAlreadyDone(IREmitter *IREmit, OrderedNodeWrapper src, uint64_t Width) { auto IROp = IREmit->GetOpHeader(src); if (IROp->Op == OP_BFE) { auto Op = IROp->C(); if (Width >= Op->Width) { return true; } } return false; } class ConstProp final : public FEXCore::IR::Pass { public: explicit ConstProp(bool DoInlineConstants, bool SupportsTSOImm9) : InlineConstants(DoInlineConstants) , SupportsTSOImm9 {SupportsTSOImm9} { } bool Run(IREmitter *IREmit) override; bool InlineConstants; private: bool HandleConstantPools(IREmitter *IREmit, const IRListView& CurrentIR); void CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& CurrentIR); void FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR); void LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR); bool ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& CurrentIR, OrderedNode* CodeNode, IROp_Header* IROp); bool ConstantPropagation(IREmitter *IREmit, const IRListView& CurrentIR, OrderedNode* CodeNode, IROp_Header* IROp); bool ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR); struct ConstPoolData { OrderedNode *Node; IR::NodeID NodeID; }; fextl::unordered_map ConstPool; fextl::map AddressgenConsts; // Pool inline constant generation. These are typically very small and pool efficiently. fextl::robin_map InlineConstantGen; OrderedNode *CreateInlineConstant(IREmitter *IREmit, uint64_t Constant) { const auto it = InlineConstantGen.find(Constant); if (it != InlineConstantGen.end()) { return it->second; } auto Result = InlineConstantGen.insert_or_assign(Constant, IREmit->_InlineConstant(Constant)); return Result.first->second; } bool SupportsTSOImm9{}; // This is a heuristic to limit constant pool live ranges to reduce RA interference pressure. // If the range is unbounded then RA interference pressure seems to increase to the point // that long blocks of constant usage can slow to a crawl. // See https://github.com/FEX-Emu/FEX/issues/2688 for more information. constexpr static uint32_t CONSTANT_POOL_RANGE_LIMIT = 200; }; bool ConstProp::HandleConstantPools(IREmitter *IREmit, const IRListView& CurrentIR) { bool Changed = false; // constants are pooled per block for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) { for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) { if (IROp->Op == OP_CONSTANT) { auto Op = IROp->C(); const auto NewNodeID = CurrentIR.GetID(CodeNode); auto it = ConstPool.find(Op->Constant); if (it != ConstPool.end()) { const auto OldNodeID = it->second.NodeID; if ((NewNodeID.Value - OldNodeID.Value) > CONSTANT_POOL_RANGE_LIMIT) { // Don't reuse if the live range is beyond the heurstic range. // Update the tracked value to this new constant. it->second.Node = CodeNode; it->second.NodeID = NewNodeID; continue; } auto CodeIter = CurrentIR.at(CodeNode); IREmit->ReplaceUsesWithAfter(CodeNode, it->second.Node, CodeIter); Changed = true; } else { ConstPool[Op->Constant] = ConstPoolData { .Node = CodeNode, .NodeID = NewNodeID, }; } } } ConstPool.clear(); } return Changed; } // Code motion around selects // Moves unary ops that depend on a select before the select, if both inputs are constants // assumes that unary ops without side effects on constants will be constprop'd void ConstProp::CodeMotionAroundSelects(IREmitter *IREmit, const IRListView& CurrentIR) { // Code motion around selects // Moves unary ops that depend on a select before the select, if both inputs are constants // assumes that unary ops without side effects on constants will be constprop'd for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) { auto BlockOp = BlockIROp->CW(); for (auto [UnaryOpNode, UnaryOpHdr] : CurrentIR.GetCode(BlockNode)) { if (IR::GetArgs(UnaryOpHdr->Op) == 1 && !HasSideEffects(UnaryOpHdr->Op)) { // could be moved auto SelectOpNode = IREmit->UnwrapNode(UnaryOpHdr->Args[0]); auto SelectOpHdr = IREmit->GetOpHeader(UnaryOpHdr->Args[0]); auto SelectOp = SelectOpHdr->CW(); // the value isn't used after the select otherwise // make sure the sizes match if (SelectOpHdr->Size == UnaryOpHdr->Size && SelectOpHdr->Op == OP_SELECT && SelectOpNode->NumUses == 1 && IREmit->IsValueConstant(SelectOp->TrueVal) && IREmit->IsValueConstant(SelectOp->FalseVal)) { IREmit->SetWriteCursor(IREmit->UnwrapNode(SelectOpNode->Header.Previous)); size_t OpSize = FEXCore::IR::GetSize(UnaryOpHdr->Op); /// copy for TrueVal /// auto NewUnaryOp1 = IREmit->AllocateRawOp(OpSize); // Copy over the op memcpy(NewUnaryOp1.first, UnaryOpHdr, OpSize); for (int i = 0; i < IR::GetArgs(NewUnaryOp1.first->Op); i++) { NewUnaryOp1.first->Args[i] = IREmit->WrapNode(IREmit->Invalid()); } // Set New Op to operate on the constant IREmit->ReplaceNodeArgument(NewUnaryOp1, 0, IREmit->UnwrapNode(SelectOp->TrueVal)); // Make select use the operated constant IREmit->ReplaceNodeArgument(SelectOpNode, 2, NewUnaryOp1); /// copy for FalseVal /// auto NewUnaryOp2 = IREmit->AllocateRawOp(OpSize); // Copy over the op memcpy(NewUnaryOp2.first, UnaryOpHdr, OpSize); for (int i = 0; i < IR::GetArgs(NewUnaryOp2.first->Op); i++) { NewUnaryOp2.first->Args[i] = IREmit->WrapNode(IREmit->Invalid()); } // Set New Op to operate on the constant IREmit->ReplaceNodeArgument(NewUnaryOp2, 0, IREmit->UnwrapNode(SelectOp->FalseVal)); // Make select use the operated constant IREmit->ReplaceNodeArgument(SelectOpNode, 3, NewUnaryOp2); // Replace uses of the defuct unary op w/ select IREmit->ReplaceAllUsesWithRange(UnaryOpNode, SelectOpNode, IREmit->GetIterator(IREmit->WrapNode(UnaryOpNode)), IREmit->GetIterator(BlockOp->Last)); } } } } } void ConstProp::FCMPOptimization(IREmitter *IREmit, const IRListView& CurrentIR) { // Make all FCMPs set no flags for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) { if (IROp->Op == OP_FCMP) { auto fcmp = IROp->CW(); fcmp->Flags = 0; } } // Set needed flags for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) { if (IROp->Op == OP_GETHOSTFLAG) { auto ghf = IROp->CW(); auto fcmp = IREmit->GetOpHeader(ghf->Value)->CW(); LOGMAN_THROW_AA_FMT(fcmp->Header.Op == OP_FCMP || fcmp->Header.Op == OP_F80CMP, "Unexpected OP_GETHOSTFLAG source"); if(fcmp->Header.Op == OP_FCMP) { fcmp->Flags |= 1 << ghf->Flag; } } } } // LoadMem / StoreMem imm pooling // If imms are close by, use address gen to generate the values instead of using a new imm void ConstProp::LoadMemStoreMemImmediatePooling(IREmitter *IREmit, const IRListView& CurrentIR) { for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) { for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) { if (IROp->Op == OP_LOADMEM || IROp->Op == OP_STOREMEM) { size_t AddrIndex = 0; size_t OffsetIndex = 0; if (IROp->Op == OP_LOADMEM) { AddrIndex = IR::IROp_LoadMem::Addr_Index; OffsetIndex = IR::IROp_LoadMem::Offset_Index; } else { AddrIndex = IR::IROp_StoreMem::Addr_Index; OffsetIndex = IR::IROp_StoreMem::Offset_Index; } uint64_t Addr; if (IREmit->IsValueConstant(IROp->Args[AddrIndex], &Addr) && IROp->Args[OffsetIndex].IsInvalid()) { for (auto& Const: AddressgenConsts) { if ((Addr - Const.second) < 65536) { IREmit->ReplaceNodeArgument(CodeNode, AddrIndex, Const.first); IREmit->ReplaceNodeArgument(CodeNode, OffsetIndex, IREmit->_Constant(Addr - Const.second)); goto doneOp; } } AddressgenConsts[IREmit->UnwrapNode(IROp->Args[AddrIndex])] = Addr; } doneOp: ; } IREmit->SetWriteCursor(CodeNode); } AddressgenConsts.clear(); } } bool ConstProp::ZextAndMaskingElimination(IREmitter *IREmit, const IRListView& CurrentIR, OrderedNode* CodeNode, IROp_Header* IROp) { bool Changed = false; switch (IROp->Op) { // Generic handling case OP_OR: case OP_XOR: case OP_NOT: case OP_ADD: case OP_SUB: case OP_MUL: case OP_UMUL: case OP_DIV: case OP_UDIV: case OP_LSHR: case OP_ASHR: case OP_LSHL: case OP_ROR: { for (int i = 0; i < IR::GetArgs(IROp->Op); i++) { auto newArg = RemoveUselessMasking(IREmit, IROp->Args[i], getMask(IROp)); if (newArg.ID() != IROp->Args[i].ID()) { IREmit->ReplaceNodeArgument(CodeNode, i, IREmit->UnwrapNode(newArg)); Changed = true; } } break; } case OP_AND: { // if AND's arguments are imms, they are masking for (int i = 0; i < IR::GetArgs(IROp->Op); i++) { uint64_t imm = 0; if (!IREmit->IsValueConstant(IROp->Args[i^1], &imm)) continue; auto newArg = RemoveUselessMasking(IREmit, IROp->Args[i], imm); if (newArg.ID() != IROp->Args[i].ID()) { IREmit->ReplaceNodeArgument(CodeNode, i, IREmit->UnwrapNode(newArg)); Changed = true; } } break; } case OP_BFE: { auto Op = IROp->C(); // Is this value already BFE'd? if (IsBfeAlreadyDone(IREmit, Op->Src, Op->Width)) { IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Src)); //printf("Removed BFE once \n"); break; } // Is this value already ZEXT'd? if (Op->lsb == 0) { //LoadMem, LoadMemTSO & LoadContext ZExt auto source = Op->Src; auto sourceHeader = IREmit->GetOpHeader(source); if (Op->Width >= (sourceHeader->Size*8) && (sourceHeader->Op == OP_LOADMEM || sourceHeader->Op == OP_LOADMEMTSO || sourceHeader->Op == OP_LOADCONTEXT) ) { //printf("Eliminated needless zext bfe\n"); // Load mem / load ctx zexts, no need to vmem IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(source)); break; } } // BFE does implicit masking, remove any masks leading to this, if possible uint64_t imm = 1ULL << (Op->Width-1); imm = (imm-1) *2 + 1; imm <<= Op->lsb; auto newArg = RemoveUselessMasking(IREmit, Op->Src, imm); if (newArg.ID() != Op->Src.ID()) { IREmit->ReplaceNodeArgument(CodeNode, Op->Src_Index, IREmit->UnwrapNode(newArg)); Changed = true; } break; } case OP_SBFE: { auto Op = IROp->C(); // BFE does implicit masking uint64_t imm = 1ULL << (Op->Width-1); imm = (imm-1) *2 + 1; imm <<= Op->lsb; auto newArg = RemoveUselessMasking(IREmit, Op->Src, imm); if (newArg.ID() != Op->Src.ID()) { IREmit->ReplaceNodeArgument(CodeNode, Op->Src_Index, IREmit->UnwrapNode(newArg)); Changed = true; } break; } case OP_VFADD: case OP_VFSUB: case OP_VFMUL: case OP_VFDIV: case OP_FCMP: { auto flopSize = IROp->Size; for (int i = 0; i < IR::GetArgs(IROp->Op); i++) { auto argHeader = IREmit->GetOpHeader(IROp->Args[i]); if (argHeader->Op == OP_VMOV) { auto source = argHeader->Args[0]; auto sourceHeader = IREmit->GetOpHeader(source); if (sourceHeader->Size >= flopSize) { IREmit->ReplaceNodeArgument(CodeNode, i, IREmit->UnwrapNode(source)); //printf("VMOV bypassed\n"); } } } break; } case OP_VMOV: { // elim from load mem auto source = IROp->Args[0]; auto sourceHeader = IREmit->GetOpHeader(source); if (IROp->Size >= sourceHeader->Size && (sourceHeader->Op == OP_LOADMEM || sourceHeader->Op == OP_LOADMEMTSO || sourceHeader->Op == OP_LOADCONTEXT) ) { //printf("Eliminated needless zext VMOV\n"); // Load mem / load ctx zexts, no need to vmem IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(source)); } else if (IROp->Size == sourceHeader->Size) { // VMOV of same size // XXX: This is unsafe of an optimization since in some cases we can't see through garbage data in the upper bits of a vector // RCLSE generates VMOV instructions which are being used as a zero extension //printf("printf vmov of same size?!\n"); //IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(source)); } break; } default: break; } return Changed; } // constprop + some more per instruction logic bool ConstProp::ConstantPropagation(IREmitter *IREmit, const IRListView& CurrentIR, OrderedNode* CodeNode, IROp_Header* IROp) { bool Changed = false; switch (IROp->Op) { /* case OP_UMUL: case OP_DIV: case OP_UDIV: case OP_REM: case OP_UREM: case OP_MULH: case OP_UMULH: case OP_LSHR: case OP_ASHR: case OP_ROL: case OP_ROR: case OP_LDIV: case OP_LUDIV: case OP_LREM: case OP_LUREM: case OP_BFI: { uint64_t Constant1; uint64_t Constant2; if (IREmit->IsValueConstant(IROp->Args[0], &Constant1) && IREmit->IsValueConstant(IROp->Args[1], &Constant2)) { LOGMAN_MSG_A_FMT("Could const prop op: {}", IR::GetName(IROp->Op)); } break; } case OP_SEXT: case OP_NEG: case OP_POPCOUNT: case OP_FINDLSB: case OP_FINDMSB: case OP_REV: case OP_SBFE: { uint64_t Constant1; if (IREmit->IsValueConstant(IROp->Args[0], &Constant1)) { LOGMAN_MSG_A_FMT("Could const prop op: {}", IR::GetName(IROp->Op)); } break; } */ case OP_LOADMEMTSO: { auto Op = IROp->CW(); auto AddressHeader = IREmit->GetOpHeader(Op->Addr); if (Op->Class == FEXCore::IR::FPRClass && AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) { // TODO: LRCPC3 supports a vector unscaled offset like LRCPC2. // Support once hardware is available to use this. auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader); Op->OffsetType = OffsetType; Op->OffsetScale = OffsetScale; IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset Changed = true; } break; } case OP_STOREMEMTSO: { auto Op = IROp->CW(); auto AddressHeader = IREmit->GetOpHeader(Op->Addr); if (Op->Class == FEXCore::IR::FPRClass && AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) { // TODO: LRCPC3 supports a vector unscaled offset like LRCPC2. // Support once hardware is available to use this. auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader); Op->OffsetType = OffsetType; Op->OffsetScale = OffsetScale; IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset Changed = true; } break; } case OP_LOADMEM: { auto Op = IROp->CW(); auto AddressHeader = IREmit->GetOpHeader(Op->Addr); if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) { auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader); Op->OffsetType = OffsetType; Op->OffsetScale = OffsetScale; IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset Changed = true; } break; } case OP_STOREMEM: { auto Op = IROp->CW(); auto AddressHeader = IREmit->GetOpHeader(Op->Addr); if (AddressHeader->Op == OP_ADD && AddressHeader->Size == 8) { auto [OffsetType, OffsetScale, Arg0, Arg1] = MemExtendedAddressing(IREmit, IROp->Size, AddressHeader); Op->OffsetType = OffsetType; Op->OffsetScale = OffsetScale; IREmit->ReplaceNodeArgument(CodeNode, Op->Addr_Index, Arg0); // Addr IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, Arg1); // Offset Changed = true; } break; } case OP_ADD: { auto Op = IROp->C(); uint64_t Constant1{}; uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { uint64_t NewConstant = (Constant1 + Constant2) & getMask(Op) ; IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } break; } case OP_SUB: { auto Op = IROp->C(); uint64_t Constant1{}; uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { uint64_t NewConstant = (Constant1 - Constant2) & getMask(Op) ; IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } break; } case OP_AND: { auto Op = IROp->CW(); uint64_t Constant1{}; uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { uint64_t NewConstant = (Constant1 & Constant2) & getMask(Op) ; IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } else if (Constant2 == 1) { // happens from flag calcs auto val = IREmit->GetOpHeader(Op->Header.Args[0]); uint64_t Constant3; if (val->Op == OP_SELECT && IREmit->IsValueConstant(val->Args[2], &Constant2) && IREmit->IsValueConstant(val->Args[3], &Constant3) && Constant2 == 1 && Constant3 == 0) { IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Header.Args[0])); Changed = true; } } else if (Op->Header.Args[0].ID() == Op->Header.Args[1].ID()) { // AND with same value results in original value IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Header.Args[0])); Changed = true; } break; } case OP_TESTNZ: { auto Op = IROp->CW(); uint64_t Constant1{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) { bool N = Constant1 & (1ull << ((Op->Size * 8) - 1)); bool Z = Constant1 == 0; uint32_t NZVC = (N ? (1u << 31) : 0) | (Z ? (1u << 30) : 0); IREmit->ReplaceWithConstant(CodeNode, NZVC); Changed = true; } break; } case OP_OR: { auto Op = IROp->CW(); uint64_t Constant1{}; uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { uint64_t NewConstant = Constant1 | Constant2; IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } else if (Op->Header.Args[0].ID() == Op->Header.Args[1].ID()) { // OR with same value results in original value IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Header.Args[0])); Changed = true; } break; } case OP_ORLSHL: { auto Op = IROp->CW(); uint64_t Constant1{}; uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { uint64_t NewConstant = Constant1 | (Constant2 << Op->BitShift); IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } break; } case OP_ORLSHR: { auto Op = IROp->CW(); uint64_t Constant1{}; uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { uint64_t NewConstant = Constant1 | (Constant2 >> Op->BitShift); IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } break; } case OP_XOR: { auto Op = IROp->C(); uint64_t Constant1{}; uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { uint64_t NewConstant = Constant1 ^ Constant2; IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } else if (Op->Header.Args[0].ID() == Op->Header.Args[1].ID()) { // XOR with same value results to zero IREmit->SetWriteCursor(CodeNode); IREmit->ReplaceAllUsesWith(CodeNode, IREmit->_Constant(0)); Changed = true; } else { // XOR with zero results in the nonzero source for (unsigned i = 0; i < 2; ++i) { if (!IREmit->IsValueConstant(Op->Header.Args[i], &Constant1)) continue; if (Constant1 != 0) continue; IREmit->SetWriteCursor(CodeNode); OrderedNode *Arg = CurrentIR.GetNode(Op->Header.Args[1 - i]); IREmit->ReplaceAllUsesWith(CodeNode, Arg); Changed = true; break; } } break; } case OP_LSHL: { auto Op = IROp->CW(); uint64_t Constant1{}; uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { // Shifts mask the shift amount by 63 or 31 depending on operating size; uint64_t ShiftMask = IROp->Size == 8 ? 63 : 31; uint64_t NewConstant = (Constant1 << (Constant2 & ShiftMask)) & getMask(Op); IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } else if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2) && Constant2 == 0) { IREmit->SetWriteCursor(CodeNode); OrderedNode *Arg = CurrentIR.GetNode(Op->Header.Args[0]); IREmit->ReplaceAllUsesWith(CodeNode, Arg); Changed = true; } else { auto newArg = RemoveUselessMasking(IREmit, Op->Header.Args[1], IROp->Size * 8 - 1); if (newArg.ID() != Op->Header.Args[1].ID()) { IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->UnwrapNode(newArg)); Changed = true; } } break; } case OP_LSHR: { auto Op = IROp->CW(); uint64_t Constant1{}; uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { // Shifts mask the shift amount by 63 or 31 depending on operating size; uint64_t ShiftMask = IROp->Size == 8 ? 63 : 31; uint64_t NewConstant = (Constant1 >> (Constant2 & ShiftMask)) & getMask(Op); IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } else if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2) && Constant2 == 0) { IREmit->SetWriteCursor(CodeNode); OrderedNode *Arg = CurrentIR.GetNode(Op->Header.Args[0]); IREmit->ReplaceAllUsesWith(CodeNode, Arg); Changed = true; } else { auto newArg = RemoveUselessMasking(IREmit, Op->Header.Args[1], IROp->Size * 8 - 1); if (newArg.ID() != Op->Header.Args[1].ID()) { IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->UnwrapNode(newArg)); Changed = true; } } break; } case OP_BFE: { auto Op = IROp->C(); uint64_t Constant; if (IROp->Size <= 8 && IREmit->IsValueConstant(Op->Src, &Constant)) { uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1); SourceMask <<= Op->lsb; uint64_t NewConstant = (Constant & SourceMask) >> Op->lsb; IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } else if (IROp->Size == CurrentIR.GetOp(Op->Header.Args[0])->Size && Op->Width == (IROp->Size * 8) && Op->lsb == 0 ) { // A BFE that extracts all bits results in original value // XXX - This is broken for now - see https://github.com/FEX-Emu/FEX/issues/351 // IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Header.Args[0])); // Changed = true; } else if (Op->Width == 1 && Op->lsb == 0) { // common from flag codegen auto val = IREmit->GetOpHeader(Op->Header.Args[0]); uint64_t Constant2{}; uint64_t Constant3{}; if (val->Op == OP_SELECT && IREmit->IsValueConstant(val->Args[2], &Constant2) && IREmit->IsValueConstant(val->Args[3], &Constant3) && Constant2 == 1 && Constant3 == 0) { IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Header.Args[0])); Changed = true; } } break; } case OP_SBFE: { auto Op = IROp->C(); uint64_t Constant; if (IREmit->IsValueConstant(Op->Src, &Constant)) { // SBFE of a constant can be converted to a constant. uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1); SourceMask <<= Op->lsb; int64_t NewConstant = (Constant & SourceMask) >> Op->lsb; NewConstant <<= 64 - Op->Width; NewConstant >>= 64 - Op->Width; IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } break; } case OP_BFI: { auto Op = IROp->C(); uint64_t ConstantDest{}; uint64_t ConstantSrc{}; bool DestIsConstant = IREmit->IsValueConstant(Op->Header.Args[0], &ConstantDest); bool SrcIsConstant = IREmit->IsValueConstant(Op->Header.Args[1], &ConstantSrc); if (DestIsConstant && SrcIsConstant) { uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1); uint64_t NewConstant = ConstantDest & ~(SourceMask << Op->lsb); NewConstant |= (ConstantSrc & SourceMask) << Op->lsb; IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } else if (SrcIsConstant && HasConsecutiveBits(ConstantSrc, Op->Width)) { // We are trying to insert constant, if it is a bitfield of only set bits then we can orr or and it. IREmit->SetWriteCursor(CodeNode); uint64_t SourceMask = Op->Width == 64 ? ~0ULL : ((1ULL << Op->Width) - 1); uint64_t NewConstant = SourceMask << Op->lsb; if (ConstantSrc & 1) { auto orr = IREmit->_Or(IR::SizeToOpSize(IROp->Size), CurrentIR.GetNode(Op->Header.Args[0]), IREmit->_Constant(NewConstant)); IREmit->ReplaceAllUsesWith(CodeNode, orr); Changed = true; } else { // We are wanting to clear the bitfield. auto andn = IREmit->_Andn(IR::SizeToOpSize(IROp->Size), CurrentIR.GetNode(Op->Header.Args[0]), IREmit->_Constant(NewConstant)); IREmit->ReplaceAllUsesWith(CodeNode, andn); Changed = true; } } break; } case OP_MUL: { auto Op = IROp->C(); uint64_t Constant1{}; uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { uint64_t NewConstant = (Constant1 * Constant2) & getMask(Op); IREmit->ReplaceWithConstant(CodeNode, NewConstant); Changed = true; } else if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2) && std::popcount(Constant2) == 1) { if (IROp->Size == 4 || IROp->Size == 8) { uint64_t amt = std::countr_zero(Constant2); IREmit->SetWriteCursor(CodeNode); auto shift = IREmit->_Lshl(IR::SizeToOpSize(IROp->Size), CurrentIR.GetNode(Op->Header.Args[0]), IREmit->_Constant(amt)); IREmit->ReplaceAllUsesWith(CodeNode, shift); Changed = true; } } break; } case OP_SELECT: { auto Op = IROp->C(); uint64_t Constant1{}; uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1) && IREmit->IsValueConstant(Op->Header.Args[1], &Constant2) && Op->Cond == COND_EQ) { Constant1 &= getMask(Op); Constant2 &= getMask(Op); bool is_true = Constant1 == Constant2; IREmit->ReplaceAllUsesWith(CodeNode, CurrentIR.GetNode(Op->Header.Args[is_true ? 2 : 3])); Changed = true; } break; } case OP_CONDJUMP: { auto Op = IROp->CW(); auto Select = IREmit->GetOpHeader(Op->Header.Args[0]); uint64_t Constant; // Fold the select into the CondJump if possible. Could handle more complex cases, too. if (Op->Cond.Val == COND_NEQ && IREmit->IsValueConstant(Op->Cmp2, &Constant) && Constant == 0 && Select->Op == OP_SELECT) { const auto SelectCmpClass = IREmit->WalkFindRegClass(Select->Args[0]); if (SelectCmpClass == GPRPairClass) { // If the comparison class is a GPRPair then don't fold the select since it isn't free. break; } uint64_t Constant1{}; uint64_t Constant2{}; if (IREmit->IsValueConstant(Select->Args[2], &Constant1) && IREmit->IsValueConstant(Select->Args[3], &Constant2)) { if (Constant1 == 1 && Constant2 == 0) { auto slc = Select->C(); IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->UnwrapNode(Select->Args[0])); IREmit->ReplaceNodeArgument(CodeNode, 1, IREmit->UnwrapNode(Select->Args[1])); Op->Cond = slc->Cond; Op->CompareSize = slc->CompareSize; Changed = true; } } } break; } default: break; } return Changed; } bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR) { InlineConstantGen.clear(); bool Changed = false; for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) { switch(IROp->Op) { case OP_LSHR: case OP_ASHR: case OP_ROR: case OP_LSHL: { auto Op = IROp->C(); uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1])); // this shouldn't be here, but rather on the emitter themselves or the constprop transformation? if (IROp->Size <=4) Constant2 &= 31; else Constant2 &= 63; IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2)); Changed = true; } break; } case OP_ADD: case OP_SUB: case OP_ADDNZCV: case OP_SUBNZCV: { auto Op = IROp->C(); uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { if (IsImmAddSub(Constant2)) { IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1])); IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2)); Changed = true; } } else if (IROp->Op == OP_SUBNZCV) { // If the first source is zero, we can use a NEGS instruction. uint64_t Constant1{}; if (IREmit->IsValueConstant(Op->Header.Args[0], &Constant1)) { if (Constant1 == 0) { IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[0])); IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, 0)); Changed = true; } } } break; } case OP_SELECT: { auto Op = IROp->C(); uint64_t Constant1{}; if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant1)) { if (IsImmAddSub(Constant1)) { IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1])); IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant1)); Changed = true; } } uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull; #ifdef JIT_ARM64 bool SupportsAllOnes = true; #else bool SupportsAllOnes = false; #endif uint64_t Constant2{}; uint64_t Constant3{}; if (IREmit->IsValueConstant(Op->Header.Args[2], &Constant2) && IREmit->IsValueConstant(Op->Header.Args[3], &Constant3) && (Constant2 == 1 || (SupportsAllOnes && Constant2 == AllOnes)) && Constant3 == 0) { IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[2])); IREmit->ReplaceNodeArgument(CodeNode, 2, CreateInlineConstant(IREmit, Constant2)); IREmit->ReplaceNodeArgument(CodeNode, 3, CreateInlineConstant(IREmit, Constant3)); } break; } case OP_CONDJUMP: { auto Op = IROp->C(); uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { if (IsImmAddSub(Constant2)) { IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1])); IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2)); Changed = true; } } break; } case OP_EXITFUNCTION: { auto Op = IROp->C(); uint64_t Constant{}; if (IREmit->IsValueConstant(Op->NewRIP, &Constant)) { IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP)); IREmit->ReplaceNodeArgument(CodeNode, 0, CreateInlineConstant(IREmit, Constant)); Changed = true; } else { auto NewRIP = IREmit->GetOpHeader(Op->NewRIP); if (NewRIP->Op == OP_ENTRYPOINTOFFSET) { auto EO = NewRIP->C(); IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP)); IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineEntrypointOffset(IR::SizeToOpSize(EO->Header.Size), EO->Offset)); Changed = true; } } break; } case OP_OR: case OP_XOR: case OP_AND: case OP_ANDN: { auto Op = IROp->CW(); uint64_t Constant2{}; if (IREmit->IsValueConstant(Op->Header.Args[1], &Constant2)) { if (IsImmLogical(Constant2, IROp->Size * 8)) { IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[1])); IREmit->ReplaceNodeArgument(CodeNode, 1, CreateInlineConstant(IREmit, Constant2)); Changed = true; } } break; } case OP_LOADMEM: { auto Op = IROp->CW(); uint64_t Constant2{}; if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Offset, &Constant2)) { if (IsImmMemory(Constant2, IROp->Size)) { IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Offset)); IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, CreateInlineConstant(IREmit, Constant2)); Changed = true; } } break; } case OP_STOREMEM: { auto Op = IROp->CW(); uint64_t Constant2{}; if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Offset, &Constant2)) { if (IsImmMemory(Constant2, IROp->Size)) { IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Offset)); IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, CreateInlineConstant(IREmit, Constant2)); Changed = true; } } break; } case OP_LOADMEMTSO: { auto Op = IROp->CW(); uint64_t Constant2{}; if (SupportsTSOImm9) { if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Offset, &Constant2)) { if (IsTSOImm9(Constant2)) { IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Offset)); IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, CreateInlineConstant(IREmit, Constant2)); Changed = true; } } } break; } case OP_STOREMEMTSO: { auto Op = IROp->CW(); uint64_t Constant2{}; if (SupportsTSOImm9) { if (Op->OffsetType == MEM_OFFSET_SXTX && IREmit->IsValueConstant(Op->Offset, &Constant2)) { if (IsTSOImm9(Constant2)) { IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Offset)); IREmit->ReplaceNodeArgument(CodeNode, Op->Offset_Index, CreateInlineConstant(IREmit, Constant2)); Changed = true; } } } break; } default: break; } } return Changed; } bool ConstProp::Run(IREmitter *IREmit) { FEXCORE_PROFILE_SCOPED("PassManager::ConstProp"); bool Changed = false; auto CurrentIR = IREmit->ViewIR(); auto OriginalWriteCursor = IREmit->GetWriteCursor(); if (HandleConstantPools(IREmit, CurrentIR)) { Changed = true; } CodeMotionAroundSelects(IREmit, CurrentIR); FCMPOptimization(IREmit, CurrentIR); LoadMemStoreMemImmediatePooling(IREmit, CurrentIR); for (auto [CodeNode, IROp] : CurrentIR.GetAllCode()) { if (ZextAndMaskingElimination(IREmit, CurrentIR, CodeNode, IROp)) { Changed = true; } if (ConstantPropagation(IREmit, CurrentIR, CodeNode, IROp)) { Changed = true; } } if (InlineConstants && ConstantInlining(IREmit, CurrentIR)) { Changed = true; } IREmit->SetWriteCursor(OriginalWriteCursor); return Changed; } fextl::unique_ptr CreateConstProp(bool InlineConstants, bool SupportsTSOImm9) { return fextl::make_unique(InlineConstants, SupportsTSOImm9); } }