From 9cef8ff7ce400cdef2e2e9a038cb903e967d05b5 Mon Sep 17 00:00:00 2001 From: Ryan Houdek Date: Sat, 4 Oct 2025 02:44:56 -0700 Subject: [PATCH] OpcodeDispatcher: Implement support for SSE4a variable extrq/insertq --- .../Source/Interface/Core/OpcodeDispatcher.h | 2 + .../Core/OpcodeDispatcher/SecondaryTables.h | 2 + .../Core/OpcodeDispatcher/Vector.cpp | 57 +++++++++++++++++++ 3 files changed, 61 insertions(+) diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h index a30f54280..3a95541b6 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher.h +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher.h @@ -910,6 +910,8 @@ public: void CRC32(OpcodeArgs); void Extrq_imm(OpcodeArgs); void Insertq_imm(OpcodeArgs); + void Extrq(OpcodeArgs); + void Insertq(OpcodeArgs); void BreakOp(OpcodeArgs, FEXCore::IR::BreakDefinition BreakDefinition); void UnimplementedOp(OpcodeArgs); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/SecondaryTables.h b/FEXCore/Source/Interface/Core/OpcodeDispatcher/SecondaryTables.h index 4ff4a72da..52a46581a 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/SecondaryTables.h +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/SecondaryTables.h @@ -199,6 +199,7 @@ constexpr DispatchTableEntry OpDispatch_SecondaryRepNEModTables[] = { {0x5F, 1, &OpDispatchBuilder::VectorScalarInsertALUOp}, {0x70, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::PSHUFWOp, true>}, {0x78, 1, &OpDispatchBuilder::Insertq_imm}, + {0x79, 1, &OpDispatchBuilder::Insertq}, {0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i32Bit>}, {0x7D, 1, &OpDispatchBuilder::HSUBP}, {0xD0, 1, &OpDispatchBuilder::ADDSUBPOp}, @@ -257,6 +258,7 @@ constexpr DispatchTableEntry OpDispatch_SecondaryOpSizeModTables[] = { {0x75, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i16Bit>}, {0x76, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VCMPEQ, OpSize::i32Bit>}, {0x78, 1, nullptr}, // GROUP 17 + {0x79, 1, &OpDispatchBuilder::Extrq}, {0x7C, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::VectorALUOp, IR::OP_VFADDP, OpSize::i64Bit>}, {0x7D, 1, &OpDispatchBuilder::HSUBP}, {0x7E, 1, &OpDispatchBuilder::Bind<&OpDispatchBuilder::MOVBetweenGPR_FPR, OpDispatchBuilder::VectorOpType::SSE>}, diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp index 77ead9c40..f003f3ef9 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/Vector.cpp @@ -5177,4 +5177,61 @@ void OpDispatchBuilder::Insertq_imm(OpcodeArgs) { StoreResultFPR(Op, Result, OpSize::iInvalid); } +void OpDispatchBuilder::Extrq(OpcodeArgs) { + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + + const Ref ElementMask = _VectorImm(OpSize::i64Bit, OpSize::i64Bit, 0x3F); + + auto GenerateMask = [this](Ref VectorWidthInBits) -> Ref { + const Ref VectorWidth = _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, VectorWidthInBits, 0); + return _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _MaskGenerateFromBitWidth(VectorWidth)); + }; + + // Bits[5:0] = Mask width in bits + const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Src, ElementMask); + + // Bits[13:8] = Shift right in bits + const Ref ShiftBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, Src, 8), ElementMask); + + // First shift in to the correct position. + Ref Result = _VUShr(OpSize::i64Bit, OpSize::i64Bit, Dest, ShiftBits, false); + + Result = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Result, GenerateMask(MaskWidthBits)); + + StoreResultFPR(Op, Result, OpSize::iInvalid); +} + +void OpDispatchBuilder::Insertq(OpcodeArgs) { + Ref Dest = LoadSourceFPR(Op, Op->Dest, Op->Flags); + Ref Src = LoadSourceFPR(Op, Op->Src[0], Op->Flags); + + auto SelectorBits = _VDupElement(OpSize::i128Bit, OpSize::i64Bit, Src, 1); + + const Ref ElementMask = _VectorImm(OpSize::i64Bit, OpSize::i64Bit, 0x3F); + + auto GenerateMask = [this](Ref VectorWidthInBits) -> Ref { + const Ref VectorWidth = _VExtractToGPR(OpSize::i64Bit, OpSize::i64Bit, VectorWidthInBits, 0); + return _VCastFromGPR(OpSize::i128Bit, OpSize::i64Bit, _MaskGenerateFromBitWidth(VectorWidth)); + }; + + // Bits[5:0] = Mask width in bits + const Ref MaskWidthBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, SelectorBits, ElementMask); + + // Bits[13:8] = Shift right in bits + const Ref ShiftBits = _VAnd(OpSize::i64Bit, OpSize::i64Bit, _VUShrI(OpSize::i64Bit, OpSize::i64Bit, SelectorBits, 8), ElementMask); + + // Extract the source data and put in to the correct location + const Ref SrcMask = GenerateMask(MaskWidthBits); + Ref SrcData = _VAnd(OpSize::i128Bit, OpSize::i64Bit, Src, SrcMask); + SrcData = _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcData, ShiftBits, false); + + // Generate a destination mask + const Ref DstMask = _VNot(OpSize::i64Bit, OpSize::i64Bit, _VUShl(OpSize::i128Bit, OpSize::i64Bit, SrcMask, ShiftBits, false)); + + Ref Result = _VAnd(OpSize::i64Bit, OpSize::i64Bit, Dest, DstMask); + Result = _VOr(OpSize::i64Bit, OpSize::i64Bit, Result, SrcData); + StoreResultFPR(Op, Result, OpSize::iInvalid); +} + } // namespace FEXCore::IR