diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp index 7ff915dea..8d2ced7ac 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp @@ -5686,10 +5686,7 @@ void OpDispatchBuilder::Insertq_imm(OpcodeArgs) { MaskVector = _VShlI(OpSize::i128Bit, OpSize::i64Bit, MaskVector, Shift); } - // Negate the mask. - MaskVector = _VNot(OpSize::i64Bit, MaskVector); - - Dest = _VAnd(OpSize::i64Bit, Dest, MaskVector); + Dest = _VAndn(OpSize::i64Bit, Dest, MaskVector); const Ref Result = _VOr(OpSize::i64Bit, Dest, Src); StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result); @@ -5745,9 +5742,7 @@ void OpDispatchBuilder::Insertq(OpcodeArgs) { SrcData = _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcData, ShiftBits, false); // Generate a destination mask - const Ref DstMask = _VNot(OpSize::i64Bit, _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcMask, ShiftBits, false)); - - Ref Result = _VAnd(OpSize::i64Bit, Dest, DstMask); + Ref Result = _VAndn(OpSize::i64Bit, Dest, _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcMask, ShiftBits, false)); Result = _VOr(OpSize::i64Bit, Result, SrcData); StoreResult_WithAVXInsert(VectorOpType::SSE, RegClass::FPR, Op, Result); } diff --git a/unittests/InstructionCountCI/Secondary_REPNE.json b/unittests/InstructionCountCI/Secondary_REPNE.json index 7a211f2ce..1f281b829 100644 --- a/unittests/InstructionCountCI/Secondary_REPNE.json +++ b/unittests/InstructionCountCI/Secondary_REPNE.json @@ -320,7 +320,7 @@ ] }, "insertq xmm0, xmm1, 0, 0": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 5, "Comment": [ "SSE4a", "0xf2 0x0f 0x78" @@ -329,13 +329,12 @@ "mov x20, #0xffffffffffffffff", "fmov d2, x20", "and v3.8b, v17.8b, v2.8b", - "mvn v2.16b, v2.16b", - "and v2.8b, v16.8b, v2.8b", + "bic v2.8b, v16.8b, v2.8b", "orr v16.8b, v2.8b, v3.8b" ] }, "insertq xmm0, xmm1, 64, 0": { - "ExpectedInstructionCount": 6, + "ExpectedInstructionCount": 5, "Comment": [ "SSE4a", "0xf2 0x0f 0x78" @@ -344,13 +343,12 @@ "mov x20, #0xffffffffffffffff", "fmov d2, x20", "and v3.8b, v17.8b, v2.8b", - "mvn v2.16b, v2.16b", - "and v2.8b, v16.8b, v2.8b", + "bic v2.8b, v16.8b, v2.8b", "orr v16.8b, v2.8b, v3.8b" ] }, "insertq xmm0, xmm1, 32, 32": { - "ExpectedInstructionCount": 8, + "ExpectedInstructionCount": 7, "Comment": [ "SSE4a", "0xf2 0x0f 0x78" @@ -361,13 +359,12 @@ "and v3.8b, v17.8b, v2.8b", "shl v3.2d, v3.2d, #32", "shl v2.2d, v2.2d, #32", - "mvn v2.16b, v2.16b", - "and v2.8b, v16.8b, v2.8b", + "bic v2.8b, v16.8b, v2.8b", "orr v16.8b, v2.8b, v3.8b" ] }, "insertq xmm0, xmm1": { - "ExpectedInstructionCount": 20, + "ExpectedInstructionCount": 19, "Comment": [ "SSE4a", "0xf2 0x0f 0x79" @@ -389,8 +386,7 @@ "and v4.16b, v17.16b, v3.16b", "ushl v4.2d, v4.2d, v2.2d", "ushl v3.2d, v3.2d, v2.2d", - "mvn v2.16b, v3.16b", - "and v2.8b, v16.8b, v2.8b", + "bic v2.8b, v16.8b, v3.8b", "orr v16.8b, v2.8b, v4.8b", "msr nzcv, x21" ]