diff --git a/FEXCore/Source/Interface/Core/Interpreter/ALUOps.cpp b/FEXCore/Source/Interface/Core/Interpreter/ALUOps.cpp index 753896705..7092a32e4 100644 --- a/FEXCore/Source/Interface/Core/Interpreter/ALUOps.cpp +++ b/FEXCore/Source/Interface/Core/Interpreter/ALUOps.cpp @@ -856,6 +856,18 @@ DEF_OP(Bfi) { GD = Res; } +DEF_OP(Bfxil) { + auto Op = IROp->C(); + uint64_t SourceMask = (1ULL << Op->Width) - 1; + if (Op->Width == 64) { + SourceMask = ~0ULL; + } + const uint64_t Src1 = *GetSrc(Data->SSAData, Op->Dest); + const uint64_t Src2 = *GetSrc(Data->SSAData, Op->Src); + const uint64_t Res = (Src1 & ~SourceMask) | ((Src2 >> Op->lsb) & SourceMask); + GD = Res; +} + DEF_OP(Bfe) { auto Op = IROp->C(); diff --git a/FEXCore/Source/Interface/Core/Interpreter/InterpreterOps.cpp b/FEXCore/Source/Interface/Core/Interpreter/InterpreterOps.cpp index c803ee306..8fb5e1924 100644 --- a/FEXCore/Source/Interface/Core/Interpreter/InterpreterOps.cpp +++ b/FEXCore/Source/Interface/Core/Interpreter/InterpreterOps.cpp @@ -84,6 +84,7 @@ constexpr OpHandlerArray InterpreterOpHandlers = [] { REGISTER_OP(COUNTLEADINGZEROES, CountLeadingZeroes); REGISTER_OP(REV, Rev); REGISTER_OP(BFI, Bfi); + REGISTER_OP(BFXIL, Bfxil); REGISTER_OP(BFE, Bfe); REGISTER_OP(SBFE, Sbfe); REGISTER_OP(SELECT, Select); diff --git a/FEXCore/Source/Interface/Core/Interpreter/InterpreterOps.h b/FEXCore/Source/Interface/Core/Interpreter/InterpreterOps.h index 571339fa9..5d80541cd 100644 --- a/FEXCore/Source/Interface/Core/Interpreter/InterpreterOps.h +++ b/FEXCore/Source/Interface/Core/Interpreter/InterpreterOps.h @@ -121,6 +121,7 @@ namespace FEXCore::CPU { DEF_OP(CountLeadingZeroes); DEF_OP(Rev); DEF_OP(Bfi); + DEF_OP(Bfxil); DEF_OP(Bfe); DEF_OP(Sbfe); DEF_OP(Select); diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp b/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp index 1469ad803..380a98474 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp @@ -1177,6 +1177,29 @@ DEF_OP(Bfi) { } } +DEF_OP(Bfxil) { + auto Op = IROp->C(); + const uint8_t OpSize = IROp->Size; + + const auto EmitSize = OpSize == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit; + + const auto Dst = GetReg(Node); + const auto SrcDst = GetReg(Op->Dest.ID()); + const auto Src = GetReg(Op->Src.ID()); + + if (Dst == SrcDst) { + // If Dst and SrcDst match then this turns in to a single instruction. + bfxil(EmitSize, Dst, Src, Op->lsb, Op->Width); + } + else { + // Destination didn't match the dst source register. + // TODO: Inefficient until FEX can have RA constraints here. + mov(EmitSize, TMP1, SrcDst); + bfxil(EmitSize, TMP1, Src, Op->lsb, Op->Width); + mov(EmitSize, Dst, TMP1.R()); + } +} + DEF_OP(Bfe) { auto Op = IROp->C(); LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size); diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/JIT.cpp b/FEXCore/Source/Interface/Core/JIT/Arm64/JIT.cpp index 2ec159fce..39418e836 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/JIT.cpp +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/JIT.cpp @@ -894,6 +894,7 @@ CPUBackend::CompiledCode Arm64JITCore::CompileCode(uint64_t Entry, REGISTER_OP(COUNTLEADINGZEROES, CountLeadingZeroes); REGISTER_OP(REV, Rev); REGISTER_OP(BFI, Bfi); + REGISTER_OP(BFXIL, Bfxil); REGISTER_OP(BFE, Bfe); REGISTER_OP(SBFE, Sbfe); REGISTER_OP(SELECT, Select); diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h b/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h index 1446ac453..d980df372 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/JITClass.h @@ -272,6 +272,7 @@ private: DEF_OP(CountLeadingZeroes); DEF_OP(Rev); DEF_OP(Bfi); + DEF_OP(Bfxil); DEF_OP(Bfe); DEF_OP(Sbfe); DEF_OP(Select); diff --git a/FEXCore/Source/Interface/Core/JIT/x86_64/ALUOps.cpp b/FEXCore/Source/Interface/Core/JIT/x86_64/ALUOps.cpp index e135ce780..919d604f3 100644 --- a/FEXCore/Source/Interface/Core/JIT/x86_64/ALUOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/x86_64/ALUOps.cpp @@ -1056,6 +1056,34 @@ DEF_OP(Bfi) { } } +DEF_OP(Bfxil) { + auto Op = IROp->C(); + const uint8_t OpSize = IROp->Size; + + auto Dst = GetDst(Node); + + uint64_t SourceMask = (1ULL << Op->Width) - 1; + if (Op->Width == 64) { + SourceMask = ~0ULL; + } + const uint64_t DestMask = ~SourceMask; + + mov(TMP1, GetSrc(Op->Src.ID())); + shr(TMP1, Op->lsb); + mov(Dst, GetSrc(Op->Dest.ID())); + + mov(TMP2, DestMask); + and_(Dst, TMP2); + mov(TMP2, SourceMask); + and_(TMP1, TMP2); + or_(Dst, TMP1); + + if (OpSize != 8) { + mov(rcx, uint64_t((1ULL << (OpSize * 8)) - 1)); + and_(Dst, rcx); + } +} + DEF_OP(Bfe) { auto Op = IROp->C(); LOGMAN_THROW_AA_FMT(IROp->Size <= 8, "OpSize is too large for BFE: {}", IROp->Size); @@ -1375,6 +1403,7 @@ void X86JITCore::RegisterALUHandlers() { REGISTER_OP(COUNTLEADINGZEROES, CountLeadingZeroes); REGISTER_OP(REV, Rev); REGISTER_OP(BFI, Bfi); + REGISTER_OP(BFXIL, Bfxil); REGISTER_OP(BFE, Bfe); REGISTER_OP(SBFE, Sbfe); REGISTER_OP(SELECT, Select); diff --git a/FEXCore/Source/Interface/Core/JIT/x86_64/JITClass.h b/FEXCore/Source/Interface/Core/JIT/x86_64/JITClass.h index 9c7bd8d03..b19bb97cf 100644 --- a/FEXCore/Source/Interface/Core/JIT/x86_64/JITClass.h +++ b/FEXCore/Source/Interface/Core/JIT/x86_64/JITClass.h @@ -280,6 +280,7 @@ private: DEF_OP(CountLeadingZeroes); DEF_OP(Rev); DEF_OP(Bfi); + DEF_OP(Bfxil); DEF_OP(Bfe); DEF_OP(Sbfe); DEF_OP(Select); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index ec455a5e1..f32c0a3a4 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -5647,9 +5647,24 @@ void OpDispatchBuilder::LZCNT(OpcodeArgs) { } void OpDispatchBuilder::MOVBEOp(OpcodeArgs) { + const uint8_t GPRSize = CTX->GetGPRSize(); + const auto SrcSize = GetSrcSize(Op); + OrderedNode *Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, 1); - Src = _Rev(OpSizeFromSrc(Op), Src); - StoreResult(GPRClass, Op, Src, 1); + Src = _Rev(IR::SizeToOpSize(std::max(4u, SrcSize)), Src); + + if (SrcSize == 2) { + // 16-bit does an insert. + // Rev of 16-bit value as 32-bit replaces the result in the upper 16-bits of the result. + // bfxil the 16-bit result in to the GPR. + OrderedNode *Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags, -1); + auto Result = _Bfxil(IR::SizeToOpSize(GPRSize), 16, 16, Dest, Src); + StoreResult_WithOpSize(GPRClass, Op, Op->Dest, Result, GPRSize, -1); + } + else { + // 32-bit does regular zext + StoreResult(GPRClass, Op, Op->Dest, Src, -1); + } } void OpDispatchBuilder::CLWB(OpcodeArgs) { diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index e24adcd41..c56e8344c 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -1073,6 +1073,17 @@ "Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit" ] }, + "GPR = Bfxil OpSize:#Size, u8:$Width, u8:$lsb, GPR:$Dest, GPR:$Src": { + "Desc": ["Copies a bitfield from one GPR to another", + "Inserting in to the low bits of the destination", + "The source bitfield is from Src[(Width + lsb):lsb]", + "The bitfield is copied in to Dest[Width:0]" + ], + "DestSize": "Size", + "EmitValidation": [ + "Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit" + ] + }, "GPR = Bfe OpSize:#Size, u8:$Width, u8:$lsb, GPR:$Src": { "Desc": ["Extracts a bitfield from one GPR with zext", "The source bitfield is from Src[Width:0]", diff --git a/unittests/InstructionCountCI/H0F38.json b/unittests/InstructionCountCI/H0F38.json index 30769c8f1..2092337cd 100644 --- a/unittests/InstructionCountCI/H0F38.json +++ b/unittests/InstructionCountCI/H0F38.json @@ -1012,16 +1012,15 @@ ] }, "movbe ax, word [rbx]": { - "ExpectedInstructionCount": 4, - "Optimal": "No", + "ExpectedInstructionCount": 3, + "Optimal": "Yes", "Comment": [ "0x66 0x0f 0x38 0xf0" ], "ExpectedArm64ASM": [ "ldrh w20, [x7]", "rev w20, w20", - "lsr w20, w20, #16", - "bfxil x4, x20, #0, #16" + "bfxil x4, x20, #16, #16" ] }, "movbe eax, dword [rbx]": {