diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp b/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp index eb2d50169..4b6d29a4d 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/ALUOps.cpp @@ -690,6 +690,50 @@ DEF_OP(ShiftFlags) { } } +DEF_OP(RotateFlags) { + auto Op = IROp->C(); + const auto Result = GetReg(Op->Result.ID()); + const auto Shift = GetReg(Op->Shift.ID()); + const bool Left = Op->Left; + const auto EmitSize = Op->Size == 8 ? ARMEmitter::Size::i64Bit : ARMEmitter::Size::i32Bit; + + // If shift=0, flags are unaffected. Wrap the whole implementation in a cbz. + ARMEmitter::SingleUseForwardLabel Done; + cbz(EmitSize, Shift, &Done); + { + // Extract the last bit shifted in to CF + const auto BitSize = Op->Size * 8; + unsigned CFBit = Left ? 0 : BitSize - 1; + + // For ROR, OF is the XOR of the new CF bit and the most significant bit of the result. + // For ROL, OF is the LSB and MSB XOR'd together. + // OF is architecturally only defined for 1-bit rotate. + eor(ARMEmitter::Size::i64Bit, TMP1, Result, Result, ARMEmitter::ShiftType::LSR, Left ? BitSize - 1 : 1); + unsigned OFBit = Left ? 0 : BitSize - 2; + + // Invert result so we get inverted carry. + mvn(ARMEmitter::Size::i64Bit, TMP2, Result); + + if (CTX->HostFeatures.SupportsFlagM) { + rmif(TMP2, (CFBit - 1) % 64, 1 << 1 /* nzCv */); + rmif(TMP1, OFBit, 1 << 0 /* nzcV */); + } else { + if (OFBit != 0) { + lsr(EmitSize, TMP1, TMP1, OFBit); + } + if (CFBit != 0) { + lsr(EmitSize, TMP2, TMP2, CFBit); + } + + mrs(TMP3, ARMEmitter::SystemRegister::NZCV); + bfi(ARMEmitter::Size::i32Bit, TMP3, TMP1, 28 /* V */, 1); + bfi(ARMEmitter::Size::i32Bit, TMP3, TMP2, 29 /* C */, 1); + msr(ARMEmitter::SystemRegister::NZCV, TMP3); + } + } + Bind(&Done); +} + DEF_OP(Extr) { auto Op = IROp->C(); const auto Dst = GetReg(Node); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp index c49dbd08c..3f5fc9d8e 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.cpp @@ -1574,62 +1574,61 @@ void OpDispatchBuilder::ASHROp(OpcodeArgs) { void OpDispatchBuilder::RotateOp(OpcodeArgs, bool Left, bool IsImmediate, bool Is1Bit) { CalculateDeferredFlags(); - auto LoadShift = [=, this](bool MustMask) -> Ref { - if (Is1Bit || IsImmediate) { - return _Constant(LoadConstantShift(Op, Is1Bit)); - } else { - // x86 masks the shift by 0x3F or 0x1F depending on size of op - const uint32_t Size = GetSrcBitSize(Op); - uint64_t Mask = Size == 64 ? 0x3F : 0x1F; + const uint32_t Size = GetSrcBitSize(Op); + const auto OpSize = Size == 64 ? OpSize::i64Bit : OpSize::i32Bit; - auto Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); - return MustMask ? _And(OpSize::i64Bit, Src, _Constant(Mask)) : Src; - } - }; - - Calculate_ShiftVariable( - Op, LoadShift(true), - [this, LoadShift, Op, Left]() { + Ref Src; + if (Is1Bit || IsImmediate) { + Src = _Constant(LoadConstantShift(Op, Is1Bit)); + } else { + // x86 masks the shift by 0x3F or 0x1F depending on size of op const uint32_t Size = GetSrcBitSize(Op); - const auto OpSize = Size == 64 ? OpSize::i64Bit : OpSize::i32Bit; + uint64_t Mask = Size == 64 ? 0x3F : 0x1F; - // We don't need to mask when we rematerialize since the Ror aborbs. - auto Src = LoadShift(false); + Src = LoadSource(GPRClass, Op, Op->Src[1], Op->Flags, {.AllowUpperGarbage = true}); + Src = _And(OpSize::i64Bit, Src, _Constant(Mask)); + } - uint64_t Const; - bool IsConst = IsValueConstant(WrapNode(Src), &Const); + uint64_t Const; + bool IsConst = IsValueConstant(WrapNode(Src), &Const); - // We fill the upper bits so we allow garbage on load. - auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); + // We fill the upper bits so we allow garbage on load. + auto Dest = LoadSource(GPRClass, Op, Op->Dest, Op->Flags, {.AllowUpperGarbage = true}); - if (Size < 32) { - // ARM doesn't support 8/16bit rotates. Emulate with an insert - // StoreResult truncates back to a 8/16 bit value - Dest = _Bfi(OpSize, Size, Size, Dest, Dest); + if (Size < 32) { + // ARM doesn't support 8/16bit rotates. Emulate with an insert + // StoreResult truncates back to a 8/16 bit value + Dest = _Bfi(OpSize, Size, Size, Dest, Dest); - if (Size == 8 && !(IsConst && Const < 8 && !Left)) { - // And because the shift size isn't masked to 8 bits, we need to fill the - // the full 32bits to get the correct result. - Dest = _Bfi(OpSize, 16, 16, Dest, Dest); + if (Size == 8 && !(IsConst && Const < 8 && !Left)) { + // And because the shift size isn't masked to 8 bits, we need to fill the + // the full 32bits to get the correct result. + Dest = _Bfi(OpSize, 16, 16, Dest, Dest); + } + } + + // To rotate 64-bits left, right-rotate by (64 - Shift) = -Shift mod 64. + auto Res = _Ror(OpSize, Dest, Left ? _Neg(OpSize, Src) : Src); + StoreResult(GPRClass, Op, Res, -1); + + if (IsConst) { + if (Const) { + // Extract the last bit shifted in to CF + SetCFDirect(Res, Left ? 0 : Size - 1, true); + + // For ROR, OF is the XOR of the new CF bit and the most significant bit of the result. + // For ROL, OF is the LSB and MSB XOR'd together. + // OF is architecturally only defined for 1-bit rotate. + if (Const == 1) { + auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, Left ? Size - 1 : 1); + SetRFLAG(NewOF, Left ? 0 : Size - 2, true); } } - - // To rotate 64-bits left, right-rotate by (64 - Shift) = -Shift mod 64. - auto Res = _Ror(OpSize, Dest, Left ? _Neg(OpSize, Src) : Src); - StoreResult(GPRClass, Op, Res, -1); - - // Extract the last bit shifted in to CF - SetCFDirect(Res, Left ? 0 : Size - 1, true); - - // For ROR, OF is the XOR of the new CF bit and the most significant bit of the result. - // For ROL, OF is the LSB and MSB XOR'd together. - // OF is architecturally only defined for 1-bit rotate. - if (!IsConst || Const == 1) { - auto NewOF = _XorShift(OpSize, Res, Res, ShiftType::LSR, Left ? Size - 1 : 1); - SetRFLAG(NewOF, Left ? 0 : Size - 2, true); - } - }, - GetSrcSize(Op) == OpSize::i32Bit ? std::make_optional(&OpDispatchBuilder::ZeroShiftResult) : std::nullopt); + } else { + HandleNZCVWrite(); + RectifyCarryInvert(true); + _RotateFlags(OpSizeFromSrc(Op), Res, Src, Left); + } } template diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index 63ac5b982..5233db745 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -1291,6 +1291,10 @@ "HasSideEffects": true, "DestSize": "8" }, + "RotateFlags OpSize:$Size, GPR:$Result, GPR:$Shift, i1:$Left": { + "Desc": ["Set NZCV flags for specified variable integer rotate with given result."], + "HasSideEffects": true + }, "GPR = Ror OpSize:#Size, GPR:$Src1, GPR:$Src2": { "Desc": ["Integer rotate right" ], diff --git a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp index c616f0e07..3f2db4b29 100644 --- a/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp @@ -165,6 +165,14 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) { .CanEliminate = true, }; + case OP_ROTATEFLAGS: + // _RotateFlags conditionally sets CV, again modeled as RMW. + return { + .Read = FLAG_C | FLAG_V, + .Write = FLAG_C | FLAG_V, + .CanEliminate = true, + }; + case OP_RDRAND: return {.Write = FLAG_NZCV}; case OP_ADDNZCV: