diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp index 9dc629a9d..796437fad 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/AVX_128.cpp @@ -603,7 +603,7 @@ void OpDispatchBuilder::AVX128_CVTFPR_To_GPR(OpcodeArgs, IR::OpSize SrcElementSi void OpDispatchBuilder::AVX128_VANDN(OpcodeArgs) { AVX128_VectorBinaryImpl(Op, OpSizeFromSrc(Op), OpSize::i128Bit, - [this](IR::OpSize _ElementSize, Ref Src1, Ref Src2) { return _VAndn(OpSize::i128Bit, _ElementSize, Src2, Src1); }); + [this](IR::OpSize, Ref Src1, Ref Src2) { return _VAndn(OpSize::i128Bit, Src2, Src1); }); } void OpDispatchBuilder::AVX128_VPACKSS(OpcodeArgs, IR::OpSize ElementSize) { @@ -885,7 +885,7 @@ void OpDispatchBuilder::AVX128_MOVMSKB(OpcodeArgs) { auto Mask1Byte = [this](Ref Src, Ref VMask) { auto VCMP = _VCMPLTZ(OpSize::i128Bit, OpSize::i8Bit, Src); - auto VAnd = _VAnd(OpSize::i128Bit, OpSize::i8Bit, VCMP, VMask); + auto VAnd = _VAnd(OpSize::i128Bit, VCMP, VMask); auto VAdd1 = _VAddP(OpSize::i128Bit, OpSize::i8Bit, VAnd, VAnd); auto VAdd2 = _VAddP(OpSize::i128Bit, OpSize::i8Bit, VAdd1, VAdd1); @@ -1729,8 +1729,8 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs, IR::OpSize ElementSize) { { // Calculate ZF first. - auto AndLow = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src2.Low, Src1.Low); - auto AndHigh = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src2.High, Src1.High); + auto AndLow = _VAnd(OpSize::i128Bit, Src2.Low, Src1.Low); + auto AndHigh = _VAnd(OpSize::i128Bit, Src2.High, Src1.High); auto ShiftLow = _VUShrI(OpSize::i128Bit, ElementSize, AndLow, ElementSizeInBits - 1); auto ShiftHigh = _VUShrI(OpSize::i128Bit, ElementSize, AndHigh, ElementSizeInBits - 1); @@ -1749,8 +1749,8 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs, IR::OpSize ElementSize) { { // Calculate CF Second - auto AndLow = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.Low, Src1.Low); - auto AndHigh = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.High, Src1.High); + auto AndLow = _VAndn(OpSize::i128Bit, Src2.Low, Src1.Low); + auto AndHigh = _VAndn(OpSize::i128Bit, Src2.High, Src1.High); auto ShiftLow = _VUShrI(OpSize::i128Bit, ElementSize, AndLow, ElementSizeInBits - 1); auto ShiftHigh = _VUShrI(OpSize::i128Bit, ElementSize, AndHigh, ElementSizeInBits - 1); @@ -1788,11 +1788,11 @@ void OpDispatchBuilder::AVX128_PTest(OpcodeArgs) { } // For 256-bit, we need to unroll. This is nontrivial. - Ref Test1Low = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src1.Low, Src2.Low); - Ref Test2Low = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.Low, Src1.Low); + Ref Test1Low = _VAnd(OpSize::i128Bit, Src1.Low, Src2.Low); + Ref Test2Low = _VAndn(OpSize::i128Bit, Src2.Low, Src1.Low); - Ref Test1High = _VAnd(OpSize::i128Bit, OpSize::i8Bit, Src1.High, Src2.High); - Ref Test2High = _VAndn(OpSize::i128Bit, OpSize::i8Bit, Src2.High, Src1.High); + Ref Test1High = _VAnd(OpSize::i128Bit, Src1.High, Src2.High); + Ref Test2High = _VAndn(OpSize::i128Bit, Src2.High, Src1.High); // Element size must be less than 32-bit for the sign bit tricks. Ref Test1Max = _VUMax(OpSize::i128Bit, OpSize::i16Bit, Test1Low, Test1High); @@ -2009,13 +2009,13 @@ void OpDispatchBuilder::AVX128_VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Sr ConstantEOR = LoadAndCacheNamedVectorConstant( OpSize::i128Bit, ElementSize == OpSize::i32Bit ? NAMED_VECTOR_PSUBADDPS_INVERT : NAMED_VECTOR_PSUBADDPD_INVERT); } - auto InvertedSourceLow = _VXor(OpSize::i128Bit, ElementSize, Sources[AddendIdx - 1].Low, ConstantEOR); + auto InvertedSourceLow = _VXor(OpSize::i128Bit, Sources[AddendIdx - 1].Low, ConstantEOR); Result.Low = _VFMLA(OpSize::i128Bit, ElementSize, Sources[Src1Idx - 1].Low, Sources[Src2Idx - 1].Low, InvertedSourceLow); if (Is128Bit) { Result.High = LoadZeroVector(OpSize::i128Bit); } else { - auto InvertedSourceHigh = _VXor(OpSize::i128Bit, ElementSize, Sources[AddendIdx - 1].High, ConstantEOR); + auto InvertedSourceHigh = _VXor(OpSize::i128Bit, Sources[AddendIdx - 1].High, ConstantEOR); Result.High = _VFMLA(OpSize::i128Bit, ElementSize, Sources[Src1Idx - 1].High, Sources[Src2Idx - 1].High, InvertedSourceHigh); } AVX128_StoreResult_WithOpSize(Op, Op->Dest, Result); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp index d6346ada6..57e7ea64b 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Crypto.cpp @@ -50,7 +50,7 @@ void OpDispatchBuilder::SHA1MSG1Op(OpcodeArgs) { Ref NewVec = _VExtr(OpSize::i128Bit, OpSize::i64Bit, Dest, Src, 1); // [W0, W1, W2, W3] ^ [W2, W3, W4, W5] - Ref Result = _VXor(OpSize::i128Bit, OpSize::i8Bit, Dest, NewVec); + Ref Result = _VXor(OpSize::i128Bit, Dest, NewVec); StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result); } diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp index f5cffe898..7ff915dea 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp @@ -523,14 +523,14 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize, case VectorCompareType::NLT_US: // NGT(Swapped operand) case VectorCompareType::NLT_UQ: { Ref Result = _VFCMPLT(ElementSize, ElementSize, Src1, Src2); - Result = _VNot(ElementSize, ElementSize, Result); + Result = _VNot(ElementSize, Result); // Insert the lower bits return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result); } case VectorCompareType::NLE_US: // NGE(Swapped operand) case VectorCompareType::NLE_UQ: { Ref Result = _VFCMPLE(ElementSize, ElementSize, Src1, Src2); - Result = _VNot(ElementSize, ElementSize, Result); + Result = _VNot(ElementSize, Result); // Insert the lower bits return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result); } @@ -539,14 +539,14 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize, case VectorCompareType::NGT_UQ: case VectorCompareType::NGT_US: { Ref Result = _VFCMPLT(ElementSize, ElementSize, Src2, Src1); - Result = _VNot(ElementSize, ElementSize, Result); + Result = _VNot(ElementSize, Result); // Insert the lower bits return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result); } case VectorCompareType::NGE_UQ: case VectorCompareType::NGE_US: { Ref Result = _VFCMPLE(ElementSize, ElementSize, Src2, Src1); - Result = _VNot(ElementSize, ElementSize, Result); + Result = _VNot(ElementSize, Result); // Insert the lower bits return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result); } @@ -567,10 +567,10 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize, // If either of the sources are unordered, then returns true. Ref Src1_U = _VFCMPEQ(Size, ElementSize, Src1, Src1); Ref Src2_U = _VFCMPEQ(Size, ElementSize, Src2, Src2); - auto Ordered = _VAnd(Size, ElementSize, Src1_U, Src2_U); + auto Ordered = _VAnd(Size, Src1_U, Src2_U); Ref Compare_Ordered = _VFCMPEQ(Size, ElementSize, Src1, Src2); - Ref Result = _VOrn(Size, ElementSize, Compare_Ordered, Ordered); + Ref Result = _VOrn(Size, Compare_Ordered, Ordered); // Insert the lower bits return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result); @@ -582,8 +582,8 @@ Ref OpDispatchBuilder::InsertScalarFCMPOpImpl(OpSize Size, IR::OpSize OpDstSize, Ref Src2_U = _VFCMPEQ(Size, ElementSize, Src2, Src2); Ref Compare_Ordered = _VFCMPEQ(Size, ElementSize, Src1, Src2); - Ref Result = _VAndn(Size, ElementSize, Src1_U, Compare_Ordered); - Result = _VAnd(Size, ElementSize, Result, Src2_U); + Ref Result = _VAndn(Size, Src1_U, Compare_Ordered); + Result = _VAnd(Size, Result, Src2_U); // Insert the lower bits return _VInsElement(OpDstSize, ElementSize, 0, 0, Src1, Result); @@ -789,7 +789,7 @@ void OpDispatchBuilder::MOVMSKOpOne(OpcodeArgs) { Ref VMask = LoadAndCacheNamedVectorConstant(SrcSize, NAMED_VECTOR_MOVMASKB); auto VCMP = _VCMPLTZ(SrcSize, OpSize::i8Bit, Src); - auto VAnd = _VAnd(SrcSize, OpSize::i8Bit, VCMP, VMask); + auto VAnd = _VAnd(SrcSize, VCMP, VMask); // Since we also handle the MM MOVMSKB here too, // we need to clamp the lower bound. @@ -884,7 +884,7 @@ Ref OpDispatchBuilder::PSHUFBOpImpl(IR::OpSize SrcSize, Ref Src1, Ref Src2, Ref // the lane splitting behavior, so cap the maximum size at 16. const auto SanitizedSrcSize = std::min(SrcSize, OpSize::i128Bit); - Ref MaskedIndices = _VAnd(SrcSize, SrcSize, Src2, MaskVector); + Ref MaskedIndices = _VAnd(SrcSize, Src2, MaskVector); Ref Low = _VTBL1(SanitizedSrcSize, Src1, MaskedIndices); if (!Is256Bit) { @@ -1999,7 +1999,7 @@ void OpDispatchBuilder::VANDNOp(OpcodeArgs) { Ref Src1 = LoadSourceFPR(Op, Op->Src[0], Op->Flags); Ref Src2 = LoadSourceFPR(Op, Op->Src[1], Op->Flags); - Ref Dest = _VAndn(SrcSize, SrcSize, Src2, Src1); + Ref Dest = _VAndn(SrcSize, Src2, Src1); StoreResultFPR(Op, Dest); } @@ -2878,24 +2878,24 @@ Ref OpDispatchBuilder::VFCMPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1 case VectorCompareType::NLT_US: // NGT(Swapped operand) case VectorCompareType::NLT_UQ: { Ref Result = _VFCMPLT(Size, ElementSize, Src1, Src2); - return _VNot(Size, ElementSize, Result); + return _VNot(Size, Result); } case VectorCompareType::NLE_US: // NGE(Swapped operand) case VectorCompareType::NLE_UQ: { Ref Result = _VFCMPLE(Size, ElementSize, Src1, Src2); - return _VNot(Size, ElementSize, Result); + return _VNot(Size, Result); } case VectorCompareType::ORD_Q: case VectorCompareType::ORD_S: return _VFCMPORD(Size, ElementSize, Src1, Src2); case VectorCompareType::NGT_UQ: case VectorCompareType::NGT_US: { Ref Result = _VFCMPLT(Size, ElementSize, Src2, Src1); - return _VNot(Size, ElementSize, Result); + return _VNot(Size, Result); } case VectorCompareType::NGE_UQ: case VectorCompareType::NGE_US: { Ref Result = _VFCMPLE(Size, ElementSize, Src2, Src1); - return _VNot(Size, ElementSize, Result); + return _VNot(Size, Result); } case VectorCompareType::GT_OQ: case VectorCompareType::GT_OS: return _VFCMPLT(Size, ElementSize, Src2, Src1); @@ -2906,10 +2906,10 @@ Ref OpDispatchBuilder::VFCMPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1 // If either of the sources are unordered, then returns true. Ref Src1_U = _VFCMPEQ(Size, ElementSize, Src1, Src1); Ref Src2_U = _VFCMPEQ(Size, ElementSize, Src2, Src2); - auto Ordered = _VAnd(Size, ElementSize, Src1_U, Src2_U); + auto Ordered = _VAnd(Size, Src1_U, Src2_U); Ref Compare_Ordered = _VFCMPEQ(Size, ElementSize, Src1, Src2); - return _VOrn(Size, ElementSize, Compare_Ordered, Ordered); + return _VOrn(Size, Compare_Ordered, Ordered); } case VectorCompareType::NEQ_OQ: case VectorCompareType::NEQ_OS: { @@ -2918,8 +2918,8 @@ Ref OpDispatchBuilder::VFCMPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Src1 Ref Src2_U = _VFCMPEQ(Size, ElementSize, Src2, Src2); Ref Compare_Ordered = _VFCMPEQ(Size, ElementSize, Src1, Src2); - Ref Result = _VAndn(Size, ElementSize, Src1_U, Compare_Ordered); - return _VAnd(Size, ElementSize, Result, Src2_U); + Ref Result = _VAndn(Size, Src1_U, Compare_Ordered); + return _VAnd(Size, Result, Src2_U); } case VectorCompareType::FALSE_OQ: case VectorCompareType::FALSE_OS: return LoadZeroVector(Size); @@ -3485,7 +3485,7 @@ Ref OpDispatchBuilder::ADDSUBPOpImpl(OpSize Size, IR::OpSize ElementSize, Ref Sr } else { auto ConstantEOR = LoadAndCacheNamedVectorConstant(Size, ElementSize == OpSize::i32Bit ? NAMED_VECTOR_PADDSUBPS_INVERT : NAMED_VECTOR_PADDSUBPD_INVERT); - auto InvertedSource = _VXor(Size, ElementSize, Src2, ConstantEOR); + auto InvertedSource = _VXor(Size, Src2, ConstantEOR); return _VFAdd(Size, ElementSize, Src1, InvertedSource); } } @@ -4349,8 +4349,8 @@ void OpDispatchBuilder::AVXVectorVariableBlend(OpcodeArgs, IR::OpSize ElementSiz } void OpDispatchBuilder::PTestOpImpl(OpSize Size, Ref Dest, Ref Src) { - Ref Test1 = _VAnd(Size, OpSize::i8Bit, Dest, Src); - Ref Test2 = _VAndn(Size, OpSize::i8Bit, Src, Dest); + Ref Test1 = _VAnd(Size, Dest, Src); + Ref Test2 = _VAndn(Size, Src, Dest); // Element size must be less than 32-bit for the sign bit tricks. Test1 = _VUMaxV(Size, OpSize::i16Bit, Test1); @@ -4384,11 +4384,11 @@ void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref Ref Mask = _VDupFromGPR(SrcSize, ElementSize, Constant(MaskConstant)); - Ref AndTest = _VAnd(SrcSize, OpSize::i8Bit, Src2, Src1); - Ref AndNotTest = _VAndn(SrcSize, OpSize::i8Bit, Src2, Src1); + Ref AndTest = _VAnd(SrcSize, Src2, Src1); + Ref AndNotTest = _VAndn(SrcSize, Src2, Src1); - Ref MaskedAnd = _VAnd(SrcSize, OpSize::i8Bit, AndTest, Mask); - Ref MaskedAndNot = _VAnd(SrcSize, OpSize::i8Bit, AndNotTest, Mask); + Ref MaskedAnd = _VAnd(SrcSize, AndTest, Mask); + Ref MaskedAndNot = _VAnd(SrcSize, AndNotTest, Mask); Ref MaxAnd = _VUMaxV(SrcSize, OpSize::i16Bit, MaskedAnd); Ref MaxAndNot = _VUMaxV(SrcSize, OpSize::i16Bit, MaskedAndNot); @@ -4491,7 +4491,7 @@ Ref OpDispatchBuilder::DPPOpImpl(IR::OpSize DstSize, Ref Src1, Ref Src2, uint8_t // Now mask results based on IndexMask. if (SrcMask != SizeMask) { auto InputMask = LoadAndCacheIndexedNamedVectorConstant(DstSize, NamedIndexMask, SrcMask * 16); - Temp = _VAnd(DstSize, ElementSize, Temp, InputMask); + Temp = _VAnd(DstSize, Temp, InputMask); } // Now due a float reduction @@ -4913,7 +4913,7 @@ void OpDispatchBuilder::VPERM2Op(OpcodeArgs) { Ref OpDispatchBuilder::VPERMDIndices(OpSize DstSize, Ref Indices, Ref IndexMask, Ref Repeating3210) { // Get rid of any junk unrelated to the relevant selector index bits (bits [2:0]) - Ref SanitizedIndices = _VAnd(DstSize, OpSize::i8Bit, Indices, IndexMask); + Ref SanitizedIndices = _VAnd(DstSize, Indices, IndexMask); // Build up the broadcasted index mask. e.g. On x86-64, the selector index // is always in the lower 3 bits of a 32-bit element. However, in order to @@ -5313,7 +5313,7 @@ Ref OpDispatchBuilder::VPERMILRegOpImpl(OpSize DstSize, IR::OpSize ElementSize, // Sanitize indices first const auto ShiftAmount = 0b11 >> static_cast(IsPD); Ref IndexMask = _VectorImm(DstSize, ElementSize, ShiftAmount); - Ref SanitizedIndices = _VAnd(DstSize, OpSize::i8Bit, Indices, IndexMask); + Ref SanitizedIndices = _VAnd(DstSize, Indices, IndexMask); Ref IndexTrn1 = _VTrn(DstSize, OpSize::i8Bit, SanitizedIndices, SanitizedIndices); Ref IndexTrn2 = _VTrn(DstSize, OpSize::i16Bit, IndexTrn1, IndexTrn1); @@ -5515,7 +5515,7 @@ void OpDispatchBuilder::VFMAddSubImpl(OpcodeArgs, bool AddSub, uint8_t Src1Idx, LoadAndCacheNamedVectorConstant(Size, ElementSize == OpSize::i32Bit ? NAMED_VECTOR_PSUBADDPS_INVERT : NAMED_VECTOR_PSUBADDPD_INVERT); } - auto InvertedSourc = _VXor(Size, ElementSize, Sources[AddendIdx - 1], ConstantEOR); + auto InvertedSourc = _VXor(Size, Sources[AddendIdx - 1], ConstantEOR); Ref Result = _VFMLA(Size, ElementSize, Sources[Src1Idx - 1], Sources[Src2Idx - 1], InvertedSourc); if (!Is256Bit) { @@ -5662,7 +5662,7 @@ void OpDispatchBuilder::Extrq_imm(OpcodeArgs) { const uint64_t Mask = ~0ULL >> (MaskWidth == 0 ? 0 : (64 - MaskWidth)); const Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask)); - Result = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Result, MaskVector); + Result = _VAnd(OpSize::i128Bit, Result, MaskVector); StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result); } @@ -5678,7 +5678,7 @@ void OpDispatchBuilder::Insertq_imm(OpcodeArgs) { Ref MaskVector = _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _Constant(Mask)); // Mask incoming source. - Src = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Src, MaskVector); + Src = _VAnd(OpSize::i64Bit, Src, MaskVector); // If shifting then shift source and mask in to the correct location. if (Shift) { @@ -5687,10 +5687,10 @@ void OpDispatchBuilder::Insertq_imm(OpcodeArgs) { } // Negate the mask. - MaskVector = _VNot(OpSize::i64Bit, OpSize::i64Bit, MaskVector); + MaskVector = _VNot(OpSize::i64Bit, MaskVector); - Dest = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Dest, MaskVector); - const Ref Result = _VOr(OpSize::i64Bit, OpSize::i64Bit, Dest, Src); + Dest = _VAnd(OpSize::i64Bit, Dest, MaskVector); + const Ref Result = _VOr(OpSize::i64Bit, Dest, Src); StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result); } @@ -5707,15 +5707,15 @@ void OpDispatchBuilder::Extrq(OpcodeArgs) { }; // Bits[5:0] = Mask width in bits - const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Src, ElementMask); + const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, Src, ElementMask); // Bits[13:8] = Shift right in bits - const Ref ShiftBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, Src, 8), ElementMask); + const Ref ShiftBits = _VAnd(OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, Src, 8), ElementMask); // First shift in to the correct position. Ref Result = _VUShr(OpSize::i64Bit, OpSize::i64Bit, Dest, ShiftBits, false); - Result = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Result, GenerateMask(MaskWidthBits)); + Result = _VAnd(OpSize::i128Bit, Result, GenerateMask(MaskWidthBits)); StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result); } @@ -5734,21 +5734,21 @@ void OpDispatchBuilder::Insertq(OpcodeArgs) { }; // Bits[5:0] = Mask width in bits - const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, SelectorBits, ElementMask); + const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, SelectorBits, ElementMask); // Bits[13:8] = Shift right in bits - const Ref ShiftBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, SelectorBits, 8), ElementMask); + const Ref ShiftBits = _VAnd(OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, SelectorBits, 8), ElementMask); // Extract the source data and put in to the correct location const Ref SrcMask = GenerateMask(MaskWidthBits); - Ref SrcData = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Src, SrcMask); + Ref SrcData = _VAnd(OpSize::i128Bit, Src, SrcMask); SrcData = _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcData, ShiftBits, false); // Generate a destination mask - const Ref DstMask = _VNot(OpSize::i64Bit, OpSize::i64Bit, _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcMask, ShiftBits, false)); + const Ref DstMask = _VNot(OpSize::i64Bit, _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcMask, ShiftBits, false)); - Ref Result = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Dest, DstMask); - Result = _VOr(OpSize::i64Bit, OpSize::i64Bit, Result, SrcData); + Ref Result = _VAnd(OpSize::i64Bit, Dest, DstMask); + Result = _VOr(OpSize::i64Bit, Result, SrcData); StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result); } diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp index ca9cb9ff8..2adcf1c94 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp @@ -575,7 +575,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) { for (int i = 0; i < 7; ++i) { Ref Reg = _LoadMemFPR(OpSize::i128Bit, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MemOffsetType::SXTX, 1); // Mask off the top bits - Reg = _VAnd(OpSize::i128Bit, OpSize::i128Bit, Reg, Mask); + Reg = _VAnd(OpSize::i128Bit, Reg, Mask); if (ReducedPrecisionMode) { // Convert to double precision Reg = _F80CVT(OpSize::i64Bit, Reg); diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index 3327ac022..665a87593 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -1864,9 +1864,9 @@ "DestSize": "RegisterSize", "ElementSize": "ElementSize" }, - "FPR = VNot OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": { + "FPR = VNot OpSize:#RegisterSize, FPR:$Vector": { "DestSize": "RegisterSize", - "ElementSize": "ElementSize" + "ElementSize": "OpSize::i8Bit" }, "FPR = VAbs OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector": { @@ -2153,41 +2153,41 @@ "ElementSize": "ElementSize" }, - "FPR = VAnd OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": { + "FPR = VAnd OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": { "DestSize": "RegisterSize", - "ElementSize": "ElementSize", + "ElementSize": "OpSize::i8Bit", "EmitValidation": [ "RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit" ] }, - "FPR = VAndn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": { + "FPR = VAndn OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": { "DestSize": "RegisterSize", - "ElementSize": "ElementSize", + "ElementSize": "OpSize::i8Bit", "EmitValidation": [ "RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit" ] }, - "FPR = VOrn OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": { + "FPR = VOrn OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": { "DestSize": "RegisterSize", - "ElementSize": "ElementSize", + "ElementSize": "OpSize::i8Bit", "EmitValidation": [ "RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit" ] }, - "FPR = VOr OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": { + "FPR = VOr OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": { "DestSize": "RegisterSize", - "ElementSize": "ElementSize", + "ElementSize": "OpSize::i8Bit", "EmitValidation": [ "RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit" ] }, - "FPR = VXor OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Vector1, FPR:$Vector2": { + "FPR = VXor OpSize:#RegisterSize, FPR:$Vector1, FPR:$Vector2": { "DestSize": "RegisterSize", - "ElementSize": "ElementSize", + "ElementSize": "OpSize::i8Bit", "EmitValidation": [ "RegisterSize == FEXCore::IR::OpSize::i256Bit || RegisterSize == FEXCore::IR::OpSize::i128Bit || RegisterSize == FEXCore::IR::OpSize::i64Bit" ] diff --git a/FEXCore/Source/Interface/IR/Passes/x87StackOptimizationPass.cpp b/FEXCore/Source/Interface/IR/Passes/x87StackOptimizationPass.cpp index 24e49a0aa..ea6cac3f5 100644 --- a/FEXCore/Source/Interface/IR/Passes/x87StackOptimizationPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/x87StackOptimizationPass.cpp @@ -1087,7 +1087,7 @@ void X87StackOptimization::Run(IREmitter* Emit) { ResultNode = IREmit->_VFNeg(OpSize::i64Bit, OpSize::i64Bit, Value); } else { Ref HelperNode = IREmit->_LoadNamedVectorConstant(OpSize::i128Bit, IR::NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK); - ResultNode = IREmit->_VXor(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode); + ResultNode = IREmit->_VXor(OpSize::i128Bit, Value, HelperNode); } StoreStackValue(ResultNode); break; @@ -1102,7 +1102,7 @@ void X87StackOptimization::Run(IREmitter* Emit) { } else { // Intermediate insts Ref HelperNode = IREmit->_LoadNamedVectorConstant(OpSize::i128Bit, IR::NamedVectorConstant::NAMED_VECTOR_F80_SIGN_MASK); - ResultNode = IREmit->_VAndn(OpSize::i128Bit, OpSize::i8Bit, Value, HelperNode); + ResultNode = IREmit->_VAndn(OpSize::i128Bit, Value, HelperNode); } StoreStackValue(ResultNode); break;