Merge pull request #3240 from Sonicadvance1/optimize_palignr_zero

OpcodeDispatcher: Optimize palignr with zero immediate
This commit is contained in:
Mai authored and GitHub committed 2023-11-01 05:55:52 +01:00
commit 9612b2fe4b
7 files changed
+50 -36

No files matched your search

@@ -1006,7 +1006,8 @@ private:
OrderedNode* PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
const X86Tables::DecodedOperand& Src2,
const X86Tables::DecodedOperand& Imm);
const X86Tables::DecodedOperand& Imm,
bool IsAVX);
void PCMPXSTRXOpImpl(OpcodeArgs, bool IsExplicit, bool IsMask);
@@ -3391,12 +3391,9 @@ void OpDispatchBuilder::DefaultAVXState() {
OrderedNode* OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::DecodedOperand& Src1,
const X86Tables::DecodedOperand& Src2,
const X86Tables::DecodedOperand& Imm) {
const X86Tables::DecodedOperand& Imm,
bool IsAVX) {
LOGMAN_THROW_A_FMT(Imm.IsLiteral(), "Imm needs to be a literal");
OrderedNode *Src1Node = LoadSource(FPRClass, Op, Src1, Op->Flags);
OrderedNode *Src2Node = LoadSource(FPRClass, Op, Src2, Op->Flags);
// For the 256-bit case we handle it as pairs of 128-bit halves.
const auto DstSize = GetDstSize(Op);
const auto SanitizedDstSize = std::min(DstSize, uint8_t{16});
@@ -3404,6 +3401,18 @@ OrderedNode* OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::Decod
const auto Is256Bit = DstSize == Core::CPUState::XMM_AVX_REG_SIZE;
const auto Index = Imm.Data.Literal.Value;
OrderedNode *Src2Node = LoadSource(FPRClass, Op, Src2, Op->Flags);
if (Index == 0) {
if (IsAVX && !Is256Bit) {
// 128-bit AVX needs to zero the upper bits.
return _VMov(16, Src2Node);
}
else {
return Src2Node;
}
}
OrderedNode *Src1Node = LoadSource(FPRClass, Op, Src1, Op->Flags);
if (Index >= (SanitizedDstSize * 2)) {
// If the immediate is greater than both vectors combined then it zeroes the vector
return LoadAndCacheNamedVectorConstant(DstSize, FEXCore::IR::NamedVectorConstant::NAMED_VECTOR_ZERO);
@@ -3421,12 +3430,12 @@ OrderedNode* OpDispatchBuilder::PALIGNROpImpl(OpcodeArgs, const X86Tables::Decod
}
void OpDispatchBuilder::PAlignrOp(OpcodeArgs) {
OrderedNode *Result = PALIGNROpImpl(Op, Op->Dest, Op->Src[0], Op->Src[1]);
OrderedNode *Result = PALIGNROpImpl(Op, Op->Dest, Op->Src[0], Op->Src[1], false);
StoreResult(FPRClass, Op, Result, -1);
}
void OpDispatchBuilder::VPALIGNROp(OpcodeArgs) {
OrderedNode *Result = PALIGNROpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2]);
OrderedNode *Result = PALIGNROpImpl(Op, Op->Src[0], Op->Src[1], Op->Src[2], true);
StoreResult(FPRClass, Op, Result, -1);
}
+7 -1
View File
@@ -4,7 +4,8 @@
"XMM0": ["0x7861626364656667", "0x4871727374757677"],
"XMM2": ["0x7861626364656667", "0x4871727374757677"],
"XMM3": ["0x5354555657584142", "0x0000000000005152"],
"XMM4": ["0x0", "0x0"]
"XMM4": ["0x0", "0x0"],
"XMM5": ["0x6162636465666768", "0x7172737475767778"]
},
"MemoryRegions": {
"0x100000000": "4096"
@@ -45,4 +46,9 @@ movapd xmm5, [rdx + 16]
palignr xmm4, xmm5, 32
movapd xmm5, [rdx]
movapd xmm6, [rdx + 16]
palignr xmm5, xmm6, 0
hlt
+7 -1
View File
@@ -3,7 +3,8 @@
"RegData": {
"MM0": "0x4851525354555657",
"MM2": "0x0061626364656667",
"MM3": "0x0"
"MM3": "0x0",
"MM4": "0x5152535455565758"
},
"MemoryRegions": {
"0x100000000": "4096"
@@ -38,4 +39,9 @@ movq mm4, [rdx + 8 * 3]
palignr mm3, mm4, 16
movq mm4, [rdx + 8]
movq mm5, [rdx + 8 * 1]
palignr mm4, mm5, 0
hlt
+6 -1
View File
@@ -13,7 +13,9 @@
"XMM10": ["0x0000000000000000", "0x0000000000000000", "0x0000000000000000", "0x0000000000000000"],
"XMM11": ["0x7861626364656667", "0x4871727374757677", "0x0000000000000000", "0x0000000000000000"],
"XMM12": ["0x5354555657584142", "0x0000000000005152", "0x0000000000000000", "0x0000000000000000"],
"XMM13": ["0x0000000000000000", "0x0000000000000000", "0x0000000000000000", "0x0000000000000000"]
"XMM13": ["0x0000000000000000", "0x0000000000000000", "0x0000000000000000", "0x0000000000000000"],
"XMM14": ["0x6162636465666768", "0x7172737475767778", "0x0000000000000000", "0x0000000000000000"],
"XMM15": ["0x6162636465666768", "0x7172737475767778", "0x8182838485868788", "0x9192939495969798"]
}
}
%endif
@@ -39,6 +41,9 @@ vpalignr xmm11, xmm0, [rdx + 32], 1
vpalignr xmm12, xmm0, [rdx + 32], 22
vpalignr xmm13, xmm0, [rdx + 32], 32
vpalignr xmm14, xmm0, [rdx + 32], 0
vpalignr ymm15, ymm0, [rdx + 32], 0
hlt
align 32
+8 -10
View File
@@ -14,15 +14,13 @@
],
"Instructions": {
"palignr mm0, mm1, 0": {
"ExpectedInstructionCount": 4,
"Optimal": "No",
"ExpectedInstructionCount": 2,
"Optimal": "Yes",
"Comment": [
"NP 0x0f 0x3a 0x0f"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #752]",
"ldr d3, [x28, #768]",
"ext v2.8b, v3.8b, v2.8b, #0",
"ldr d2, [x28, #768]",
"str d2, [x28, #752]"
]
},
@@ -33,9 +31,9 @@
"NP 0x0f 0x3a 0x0f"
],
"ExpectedArm64ASM": [
"ldr d2, [x28, #752]",
"ldr d3, [x28, #768]",
"ext v2.8b, v3.8b, v2.8b, #1",
"ldr d2, [x28, #768]",
"ldr d3, [x28, #752]",
"ext v2.8b, v2.8b, v3.8b, #1",
"str d2, [x28, #752]"
]
},
@@ -586,12 +584,12 @@
},
"palignr xmm0, xmm1, 0": {
"ExpectedInstructionCount": 1,
"Optimal": "No",
"Optimal": "Yes",
"Comment": [
"0x66 0x0f 0x3a 0x0f"
],
"ExpectedArm64ASM": [
"ext v16.16b, v17.16b, v16.16b, #0"
"mov v16.16b, v17.16b"
]
},
"palignr xmm0, xmm1, 1": {
+4 -15
View File
@@ -2934,7 +2934,7 @@
"Map 3 0b01 0x0f 128-bit"
],
"ExpectedArm64ASM": [
"ext v16.16b, v18.16b, v17.16b, #0"
"mov v16.16b, v18.16b"
]
},
"vpalignr xmm0, xmm1, xmm2, 1": {
@@ -2969,24 +2969,13 @@
]
},
"vpalignr ymm0, ymm1, ymm2, 0": {
"ExpectedInstructionCount": 12,
"Optimal": "No",
"ExpectedInstructionCount": 1,
"Optimal": "Yes",
"Comment": [
"Map 3 0b01 0x0f 256-bit"
],
"ExpectedArm64ASM": [
"ext v2.16b, v18.16b, v17.16b, #0",
"mov z1.q, z17.q[1]",
"mov z3.d, z17.d",
"mov z3.b, p6/m, z1.b",
"mov z1.q, z18.q[1]",
"mov z4.d, z18.d",
"mov z4.b, p6/m, z1.b",
"ext v3.16b, v4.16b, v3.16b, #0",
"mov z1.q, q3",
"mov z16.d, z2.d",
"not p0.b, p7/z, p6.b",
"mov z16.b, p0/m, z1.b"
"mov z16.d, p7/m, z18.d"
]
},
"vpalignr ymm0, ymm1, ymm2, 1": {