mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 11:00:19 +02:00
Merge pull request #3240 from Sonicadvance1/optimize_palignr_zero
OpcodeDispatcher: Optimize palignr with zero immediate
This commit is contained in:
7 files changed
+50
-36
No files matched your search
@@ -1006,7 +1006,8 @@ private:
|
||||
|
||||
OrderedNode* PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm);
|
||||
const X86Tables::DecodedOperand& Imm,
|
||||
bool IsAVX);
|
||||
|
||||
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask);
|
||||
|
||||
|
||||
@@ -3391,12 +3391,9 @@ void OpDispatchBuilder::DefaultAVXState() {
|
||||
|
||||
OrderedNode* OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
|
||||
const X86Tables::DecodedOperand& Src2,
|
||||
const X86Tables::DecodedOperand& Imm) {
|
||||
const X86Tables::DecodedOperand& Imm,
|
||||
bool IsAVX) {
|
||||
LOGMAN_THROW_A_FMT(Imm.IsLiteral(), "Imm needs to be a literal");
|
||||
|
||||
OrderedNode *Src1Node = LoadSource(FPRClass, Op, Src1, Op->Flags);
|
||||
OrderedNode *Src2Node = LoadSource(FPRClass, Op, Src2, Op->Flags);
|
||||
|
||||
// For the 256-bit case we handle it as pairs of 128-bit halves.
|
||||
const auto DstSize = GetDstSize(Op);
|
||||
const auto SanitizedDstSize = std::min(DstSize, uint8_t{16});
|
||||
@@ -3404,6 +3401,18 @@ OrderedNode* OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::Decod
|
||||
const auto Is256Bit = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
|
||||
const auto Index = Imm.Data.Literal.Value;
|
||||
|
||||
OrderedNode *Src2Node = LoadSource(FPRClass, Op, Src2, Op->Flags);
|
||||
if (Index == 0) {
|
||||
if (IsAVX && !Is256Bit) {
|
||||
// 128-bit AVX needs to zero the upper bits.
|
||||
return _VMov(16, Src2Node);
|
||||
}
|
||||
else {
|
||||
return Src2Node;
|
||||
}
|
||||
}
|
||||
OrderedNode *Src1Node = LoadSource(FPRClass, Op, Src1, Op->Flags);
|
||||
|
||||
if (Index >= (SanitizedDstSize * 2)) {
|
||||
// If the immediate is greater than both vectors combined then it zeroes the vector
|
||||
return LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
|
||||
@@ -3421,12 +3430,12 @@ OrderedNode* OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::Decod
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::PAlignrOp(OpcodeArgs) {
|
||||
OrderedNode *Result = PALIGNROpImpl(Op, Op->Dest, Op->Src[0], Op->Src[1]);
|
||||
OrderedNode *Result = PALIGNROpImpl(Op, Op->Dest, Op->Src[0], Op->Src[1], false);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::VPALIGNROp(OpcodeArgs) {
|
||||
OrderedNode *Result = PALIGNROpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2]);
|
||||
OrderedNode *Result = PALIGNROpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2], true);
|
||||
StoreResult(FPRClass, Op, Result, -1);
|
||||
}
|
||||
|
||||
|
||||
@@ -4,7 +4,8 @@
|
||||
"XMM0": ["0x7861626364656667", "0x4871727374757677"],
|
||||
"XMM2": ["0x7861626364656667", "0x4871727374757677"],
|
||||
"XMM3": ["0x5354555657584142", "0x0000000000005152"],
|
||||
"XMM4": ["0x0", "0x0"]
|
||||
"XMM4": ["0x0", "0x0"],
|
||||
"XMM5": ["0x6162636465666768", "0x7172737475767778"]
|
||||
},
|
||||
"MemoryRegions": {
|
||||
"0x100000000": "4096"
|
||||
@@ -45,4 +46,9 @@ movapd xmm5, [rdx + 16]
|
||||
|
||||
palignr xmm4, xmm5, 32
|
||||
|
||||
movapd xmm5, [rdx]
|
||||
movapd xmm6, [rdx + 16]
|
||||
|
||||
palignr xmm5, xmm6, 0
|
||||
|
||||
hlt
|
||||
@@ -3,7 +3,8 @@
|
||||
"RegData": {
|
||||
"MM0": "0x4851525354555657",
|
||||
"MM2": "0x0061626364656667",
|
||||
"MM3": "0x0"
|
||||
"MM3": "0x0",
|
||||
"MM4": "0x5152535455565758"
|
||||
},
|
||||
"MemoryRegions": {
|
||||
"0x100000000": "4096"
|
||||
@@ -38,4 +39,9 @@ movq mm4, [rdx + 8 * 3]
|
||||
|
||||
palignr mm3, mm4, 16
|
||||
|
||||
movq mm4, [rdx + 8]
|
||||
movq mm5, [rdx + 8 * 1]
|
||||
|
||||
palignr mm4, mm5, 0
|
||||
|
||||
hlt
|
||||
@@ -13,7 +13,9 @@
|
||||
"XMM10": ["0x0000000000000000", "0x0000000000000000", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM11": ["0x7861626364656667", "0x4871727374757677", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM12": ["0x5354555657584142", "0x0000000000005152", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM13": ["0x0000000000000000", "0x0000000000000000", "0x0000000000000000", "0x0000000000000000"]
|
||||
"XMM13": ["0x0000000000000000", "0x0000000000000000", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM14": ["0x6162636465666768", "0x7172737475767778", "0x0000000000000000", "0x0000000000000000"],
|
||||
"XMM15": ["0x6162636465666768", "0x7172737475767778", "0x8182838485868788", "0x9192939495969798"]
|
||||
}
|
||||
}
|
||||
%endif
|
||||
@@ -39,6 +41,9 @@ vpalignr xmm11, xmm0, [rdx + 32], 1
|
||||
vpalignr xmm12, xmm0, [rdx + 32], 22
|
||||
vpalignr xmm13, xmm0, [rdx + 32], 32
|
||||
|
||||
vpalignr xmm14, xmm0, [rdx + 32], 0
|
||||
vpalignr ymm15, ymm0, [rdx + 32], 0
|
||||
|
||||
hlt
|
||||
|
||||
align 32
|
||||
|
||||
@@ -14,15 +14,13 @@
|
||||
],
|
||||
"Instructions": {
|
||||
"palignr mm0, mm1, 0": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 2,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"NP 0x0f 0x3a 0x0f"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #752]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"ext v2.8b, v3.8b, v2.8b, #0",
|
||||
"ldr d2, [x28, #768]",
|
||||
"str d2, [x28, #752]"
|
||||
]
|
||||
},
|
||||
@@ -33,9 +31,9 @@
|
||||
"NP 0x0f 0x3a 0x0f"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d2, [x28, #752]",
|
||||
"ldr d3, [x28, #768]",
|
||||
"ext v2.8b, v3.8b, v2.8b, #1",
|
||||
"ldr d2, [x28, #768]",
|
||||
"ldr d3, [x28, #752]",
|
||||
"ext v2.8b, v2.8b, v3.8b, #1",
|
||||
"str d2, [x28, #752]"
|
||||
]
|
||||
},
|
||||
@@ -586,12 +584,12 @@
|
||||
},
|
||||
"palignr xmm0, xmm1, 0": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "No",
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"0x66 0x0f 0x3a 0x0f"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ext v16.16b, v17.16b, v16.16b, #0"
|
||||
"mov v16.16b, v17.16b"
|
||||
]
|
||||
},
|
||||
"palignr xmm0, xmm1, 1": {
|
||||
|
||||
@@ -2934,7 +2934,7 @@
|
||||
"Map 3 0b01 0x0f 128-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ext v16.16b, v18.16b, v17.16b, #0"
|
||||
"mov v16.16b, v18.16b"
|
||||
]
|
||||
},
|
||||
"vpalignr xmm0, xmm1, xmm2, 1": {
|
||||
@@ -2969,24 +2969,13 @@
|
||||
]
|
||||
},
|
||||
"vpalignr ymm0, ymm1, ymm2, 0": {
|
||||
"ExpectedInstructionCount": 12,
|
||||
"Optimal": "No",
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
"Comment": [
|
||||
"Map 3 0b01 0x0f 256-bit"
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ext v2.16b, v18.16b, v17.16b, #0",
|
||||
"mov z1.q, z17.q[1]",
|
||||
"mov z3.d, z17.d",
|
||||
"mov z3.b, p6/m, z1.b",
|
||||
"mov z1.q, z18.q[1]",
|
||||
"mov z4.d, z18.d",
|
||||
"mov z4.b, p6/m, z1.b",
|
||||
"ext v3.16b, v4.16b, v3.16b, #0",
|
||||
"mov z1.q, q3",
|
||||
"mov z16.d, z2.d",
|
||||
"not p0.b, p7/z, p6.b",
|
||||
"mov z16.b, p0/m, z1.b"
|
||||
"mov z16.d, p7/m, z18.d"
|
||||
]
|
||||
},
|
||||
"vpalignr ymm0, ymm1, ymm2, 1": {
|
||||
|
||||
Reference in new issue
Block a user