mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 18:00:17 +02:00
Merge pull request #3098 from Sonicadvance1/optimize_vectors_sve
Arm64: Optimize wide shifts slightly for 64-bit OpSize
This commit is contained in:
5 files changed
+113
-108
No files matched your search
@@ -2543,16 +2543,23 @@ DEF_OP(VUShrSWide) {
|
|||||||
else if (HostSupportsSVE128) {
|
else if (HostSupportsSVE128) {
|
||||||
const auto Mask = PRED_TMP_16B.Merging();
|
const auto Mask = PRED_TMP_16B.Merging();
|
||||||
|
|
||||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
auto ShiftRegister = ShiftScalar.Z();
|
||||||
|
if (OpSize > 8) {
|
||||||
|
// SVE wide shifts don't need to duplicate the low bits unless the OpSize is 16-bytes
|
||||||
|
// Slightly more optimal for 8-byte opsize.
|
||||||
|
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||||
|
ShiftRegister = VTMP1.Z();
|
||||||
|
}
|
||||||
|
|
||||||
if (Dst != Vector) {
|
if (Dst != Vector) {
|
||||||
// NOTE: SVE LSR is a destructive operation.
|
// NOTE: SVE LSR is a destructive operation.
|
||||||
movprfx(Dst.Z(), Vector.Z());
|
movprfx(Dst.Z(), Vector.Z());
|
||||||
}
|
}
|
||||||
if (ElementSize == 8) {
|
if (ElementSize == 8) {
|
||||||
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
|
||||||
}
|
}
|
||||||
else {
|
else {
|
||||||
lsr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
lsr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
// uqshl + ushr of 57-bits leaves 7-bits remaining.
|
// uqshl + ushr of 57-bits leaves 7-bits remaining.
|
||||||
@@ -2603,16 +2610,23 @@ DEF_OP(VSShrSWide) {
|
|||||||
else if (HostSupportsSVE128) {
|
else if (HostSupportsSVE128) {
|
||||||
const auto Mask = PRED_TMP_16B.Merging();
|
const auto Mask = PRED_TMP_16B.Merging();
|
||||||
|
|
||||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
auto ShiftRegister = ShiftScalar.Z();
|
||||||
|
if (OpSize > 8) {
|
||||||
|
// SVE wide shifts don't need to duplicate the low bits unless the OpSize is 16-bytes
|
||||||
|
// Slightly more optimal for 8-byte opsize.
|
||||||
|
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||||
|
ShiftRegister = VTMP1.Z();
|
||||||
|
}
|
||||||
|
|
||||||
if (Dst != Vector) {
|
if (Dst != Vector) {
|
||||||
// NOTE: SVE LSR is a destructive operation.
|
// NOTE: SVE LSR is a destructive operation.
|
||||||
movprfx(Dst.Z(), Vector.Z());
|
movprfx(Dst.Z(), Vector.Z());
|
||||||
}
|
}
|
||||||
if (ElementSize == 8) {
|
if (ElementSize == 8) {
|
||||||
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
|
||||||
}
|
}
|
||||||
else {
|
else {
|
||||||
asr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
asr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
// uqshl + ushr of 57-bits leaves 7-bits remaining.
|
// uqshl + ushr of 57-bits leaves 7-bits remaining.
|
||||||
@@ -2663,16 +2677,23 @@ DEF_OP(VUShlSWide) {
|
|||||||
else if (HostSupportsSVE128) {
|
else if (HostSupportsSVE128) {
|
||||||
const auto Mask = PRED_TMP_16B.Merging();
|
const auto Mask = PRED_TMP_16B.Merging();
|
||||||
|
|
||||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
auto ShiftRegister = ShiftScalar.Z();
|
||||||
|
if (OpSize > 8) {
|
||||||
|
// SVE wide shifts don't need to duplicate the low bits unless the OpSize is 16-bytes
|
||||||
|
// Slightly more optimal for 8-byte opsize.
|
||||||
|
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||||
|
ShiftRegister = VTMP1.Z();
|
||||||
|
}
|
||||||
|
|
||||||
if (Dst != Vector) {
|
if (Dst != Vector) {
|
||||||
// NOTE: SVE LSR is a destructive operation.
|
// NOTE: SVE LSR is a destructive operation.
|
||||||
movprfx(Dst.Z(), Vector.Z());
|
movprfx(Dst.Z(), Vector.Z());
|
||||||
}
|
}
|
||||||
if (ElementSize == 8) {
|
if (ElementSize == 8) {
|
||||||
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
|
||||||
}
|
}
|
||||||
else {
|
else {
|
||||||
lsl_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z());
|
lsl_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
|
||||||
}
|
}
|
||||||
} else {
|
} else {
|
||||||
// uqshl + ushr of 57-bits leaves 7-bits remaining.
|
// uqshl + ushr of 57-bits leaves 7-bits remaining.
|
||||||
|
|||||||
@@ -1751,8 +1751,8 @@ OrderedNode* OpDispatchBuilder::PSRLDOpImpl(OpcodeArgs, size_t ElementSize,
|
|||||||
|
|
||||||
template<size_t ElementSize>
|
template<size_t ElementSize>
|
||||||
void OpDispatchBuilder::PSRLDOp(OpcodeArgs) {
|
void OpDispatchBuilder::PSRLDOp(OpcodeArgs) {
|
||||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
|
||||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||||
|
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||||
OrderedNode *Result = PSRLDOpImpl(Op, ElementSize, Dest, Src);
|
OrderedNode *Result = PSRLDOpImpl(Op, ElementSize, Dest, Src);
|
||||||
|
|
||||||
StoreResult(FPRClass, Op, Result, -1);
|
StoreResult(FPRClass, Op, Result, -1);
|
||||||
@@ -1887,8 +1887,8 @@ OrderedNode* OpDispatchBuilder::PSLLImpl(OpcodeArgs, size_t ElementSize,
|
|||||||
|
|
||||||
template<size_t ElementSize>
|
template<size_t ElementSize>
|
||||||
void OpDispatchBuilder::PSLL(OpcodeArgs) {
|
void OpDispatchBuilder::PSLL(OpcodeArgs) {
|
||||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
|
||||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||||
|
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||||
OrderedNode *Result = PSLLImpl(Op, ElementSize, Dest, Src);
|
OrderedNode *Result = PSLLImpl(Op, ElementSize, Dest, Src);
|
||||||
|
|
||||||
StoreResult(FPRClass, Op, Result, -1);
|
StoreResult(FPRClass, Op, Result, -1);
|
||||||
@@ -1927,8 +1927,8 @@ OrderedNode* OpDispatchBuilder::PSRAOpImpl(OpcodeArgs, size_t ElementSize,
|
|||||||
|
|
||||||
template<size_t ElementSize>
|
template<size_t ElementSize>
|
||||||
void OpDispatchBuilder::PSRAOp(OpcodeArgs) {
|
void OpDispatchBuilder::PSRAOp(OpcodeArgs) {
|
||||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
|
||||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||||
|
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||||
OrderedNode *Result = PSRAOpImpl(Op, ElementSize, Dest, Src);
|
OrderedNode *Result = PSRAOpImpl(Op, ElementSize, Dest, Src);
|
||||||
|
|
||||||
StoreResult(FPRClass, Op, Result, -1);
|
StoreResult(FPRClass, Op, Result, -1);
|
||||||
|
|||||||
@@ -4182,13 +4182,13 @@
|
|||||||
"Optimal": "Yes",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xd1",
|
"Comment": "0x0f 0xd1",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"uqshl d0, d2, #57",
|
"uqshl d0, d3, #57",
|
||||||
"ushr d0, d0, #57",
|
"ushr d0, d0, #57",
|
||||||
"dup v0.8h, v0.h[0]",
|
"dup v0.8h, v0.h[0]",
|
||||||
"neg v0.8h, v0.8h",
|
"neg v0.8h, v0.8h",
|
||||||
"ushl v2.8h, v3.8h, v0.8h",
|
"ushl v2.8h, v2.8h, v0.8h",
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
@@ -4197,13 +4197,13 @@
|
|||||||
"Optimal": "Yes",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xd2",
|
"Comment": "0x0f 0xd2",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"uqshl d0, d2, #57",
|
"uqshl d0, d3, #57",
|
||||||
"ushr d0, d0, #57",
|
"ushr d0, d0, #57",
|
||||||
"dup v0.4s, v0.s[0]",
|
"dup v0.4s, v0.s[0]",
|
||||||
"neg v0.4s, v0.4s",
|
"neg v0.4s, v0.4s",
|
||||||
"ushl v2.4s, v3.4s, v0.4s",
|
"ushl v2.4s, v2.4s, v0.4s",
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
@@ -4212,13 +4212,13 @@
|
|||||||
"Optimal": "Yes",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xd3",
|
"Comment": "0x0f 0xd3",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"uqshl d0, d2, #57",
|
"uqshl d0, d3, #57",
|
||||||
"ushr d0, d0, #57",
|
"ushr d0, d0, #57",
|
||||||
"dup v0.2d, v0.d[0]",
|
"dup v0.2d, v0.d[0]",
|
||||||
"neg v0.2d, v0.2d",
|
"neg v0.2d, v0.2d",
|
||||||
"ushl v2.2d, v3.2d, v0.2d",
|
"ushl v2.2d, v2.2d, v0.2d",
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
@@ -4367,13 +4367,13 @@
|
|||||||
"Optimal": "Yes",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xe1",
|
"Comment": "0x0f 0xe1",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"uqshl d0, d2, #57",
|
"uqshl d0, d3, #57",
|
||||||
"ushr d0, d0, #57",
|
"ushr d0, d0, #57",
|
||||||
"dup v0.8h, v0.h[0]",
|
"dup v0.8h, v0.h[0]",
|
||||||
"neg v0.8h, v0.8h",
|
"neg v0.8h, v0.8h",
|
||||||
"sshl v2.8h, v3.8h, v0.8h",
|
"sshl v2.8h, v2.8h, v0.8h",
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
@@ -4382,13 +4382,13 @@
|
|||||||
"Optimal": "Yes",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xe2",
|
"Comment": "0x0f 0xe2",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"uqshl d0, d2, #57",
|
"uqshl d0, d3, #57",
|
||||||
"ushr d0, d0, #57",
|
"ushr d0, d0, #57",
|
||||||
"dup v0.4s, v0.s[0]",
|
"dup v0.4s, v0.s[0]",
|
||||||
"neg v0.4s, v0.4s",
|
"neg v0.4s, v0.4s",
|
||||||
"sshl v2.4s, v3.4s, v0.4s",
|
"sshl v2.4s, v2.4s, v0.4s",
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
@@ -4529,12 +4529,12 @@
|
|||||||
"Optimal": "Yes",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xf1",
|
"Comment": "0x0f 0xf1",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"uqshl d0, d2, #57",
|
"uqshl d0, d3, #57",
|
||||||
"ushr d0, d0, #57",
|
"ushr d0, d0, #57",
|
||||||
"dup v0.8h, v0.h[0]",
|
"dup v0.8h, v0.h[0]",
|
||||||
"ushl v2.8h, v3.8h, v0.8h",
|
"ushl v2.8h, v2.8h, v0.8h",
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
@@ -4543,12 +4543,12 @@
|
|||||||
"Optimal": "Yes",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xf2",
|
"Comment": "0x0f 0xf2",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"uqshl d0, d2, #57",
|
"uqshl d0, d3, #57",
|
||||||
"ushr d0, d0, #57",
|
"ushr d0, d0, #57",
|
||||||
"dup v0.4s, v0.s[0]",
|
"dup v0.4s, v0.s[0]",
|
||||||
"ushl v2.4s, v3.4s, v0.4s",
|
"ushl v2.4s, v2.4s, v0.4s",
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
@@ -4557,12 +4557,12 @@
|
|||||||
"Optimal": "Yes",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xf3",
|
"Comment": "0x0f 0xf3",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"uqshl d0, d2, #57",
|
"uqshl d0, d3, #57",
|
||||||
"ushr d0, d0, #57",
|
"ushr d0, d0, #57",
|
||||||
"dup v0.2d, v0.d[0]",
|
"dup v0.2d, v0.d[0]",
|
||||||
"ushl v2.2d, v3.2d, v0.2d",
|
"ushl v2.2d, v2.2d, v0.2d",
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
|
|||||||
@@ -9,7 +9,7 @@
|
|||||||
"Instructions": {
|
"Instructions": {
|
||||||
"psrlw xmm0, xmm1": {
|
"psrlw xmm0, xmm1": {
|
||||||
"ExpectedInstructionCount": 2,
|
"ExpectedInstructionCount": 2,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x66 0x0f 0xd1",
|
"Comment": "0x66 0x0f 0xd1",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"mov z0.d, d17",
|
"mov z0.d, d17",
|
||||||
@@ -18,7 +18,7 @@
|
|||||||
},
|
},
|
||||||
"psrld xmm0, xmm1": {
|
"psrld xmm0, xmm1": {
|
||||||
"ExpectedInstructionCount": 2,
|
"ExpectedInstructionCount": 2,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x66 0x0f 0xd2",
|
"Comment": "0x66 0x0f 0xd2",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"mov z0.d, d17",
|
"mov z0.d, d17",
|
||||||
@@ -27,7 +27,7 @@
|
|||||||
},
|
},
|
||||||
"psrlq xmm0, xmm1": {
|
"psrlq xmm0, xmm1": {
|
||||||
"ExpectedInstructionCount": 2,
|
"ExpectedInstructionCount": 2,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x66 0x0f 0xd3",
|
"Comment": "0x66 0x0f 0xd3",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"mov z0.d, d17",
|
"mov z0.d, d17",
|
||||||
@@ -36,7 +36,7 @@
|
|||||||
},
|
},
|
||||||
"psraw xmm0, xmm1": {
|
"psraw xmm0, xmm1": {
|
||||||
"ExpectedInstructionCount": 2,
|
"ExpectedInstructionCount": 2,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x66 0x0f 0xe1",
|
"Comment": "0x66 0x0f 0xe1",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"mov z0.d, d17",
|
"mov z0.d, d17",
|
||||||
@@ -45,7 +45,7 @@
|
|||||||
},
|
},
|
||||||
"psrad xmm0, xmm1": {
|
"psrad xmm0, xmm1": {
|
||||||
"ExpectedInstructionCount": 2,
|
"ExpectedInstructionCount": 2,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x66 0x0f 0xe2",
|
"Comment": "0x66 0x0f 0xe2",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"mov z0.d, d17",
|
"mov z0.d, d17",
|
||||||
@@ -70,7 +70,7 @@
|
|||||||
},
|
},
|
||||||
"psllw xmm0, xmm1": {
|
"psllw xmm0, xmm1": {
|
||||||
"ExpectedInstructionCount": 2,
|
"ExpectedInstructionCount": 2,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x66 0x0f 0xf1",
|
"Comment": "0x66 0x0f 0xf1",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"mov z0.d, d17",
|
"mov z0.d, d17",
|
||||||
@@ -79,7 +79,7 @@
|
|||||||
},
|
},
|
||||||
"pslld xmm0, xmm1": {
|
"pslld xmm0, xmm1": {
|
||||||
"ExpectedInstructionCount": 2,
|
"ExpectedInstructionCount": 2,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x66 0x0f 0xf2",
|
"Comment": "0x66 0x0f 0xf2",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"mov z0.d, d17",
|
"mov z0.d, d17",
|
||||||
@@ -88,7 +88,7 @@
|
|||||||
},
|
},
|
||||||
"psllq xmm0, xmm1": {
|
"psllq xmm0, xmm1": {
|
||||||
"ExpectedInstructionCount": 2,
|
"ExpectedInstructionCount": 2,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x66 0x0f 0xf3",
|
"Comment": "0x66 0x0f 0xf3",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"mov z0.d, d17",
|
"mov z0.d, d17",
|
||||||
|
|||||||
@@ -34,106 +34,90 @@
|
|||||||
]
|
]
|
||||||
},
|
},
|
||||||
"psrlw mm0, mm1": {
|
"psrlw mm0, mm1": {
|
||||||
"ExpectedInstructionCount": 6,
|
"ExpectedInstructionCount": 4,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xd1",
|
"Comment": "0x0f 0xd1",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"mov z0.d, d2",
|
"lsr z2.h, p6/m, z2.h, z3.d",
|
||||||
"movprfx z2, z3",
|
|
||||||
"lsr z2.h, p6/m, z2.h, z0.d",
|
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"psrld mm0, mm1": {
|
"psrld mm0, mm1": {
|
||||||
"ExpectedInstructionCount": 6,
|
"ExpectedInstructionCount": 4,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xd2",
|
"Comment": "0x0f 0xd2",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"mov z0.d, d2",
|
"lsr z2.s, p6/m, z2.s, z3.d",
|
||||||
"movprfx z2, z3",
|
|
||||||
"lsr z2.s, p6/m, z2.s, z0.d",
|
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"psrlq mm0, mm1": {
|
"psrlq mm0, mm1": {
|
||||||
"ExpectedInstructionCount": 6,
|
"ExpectedInstructionCount": 4,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xd3",
|
"Comment": "0x0f 0xd3",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"mov z0.d, d2",
|
"lsr z2.d, p6/m, z2.d, z3.d",
|
||||||
"movprfx z2, z3",
|
|
||||||
"lsr z2.d, p6/m, z2.d, z0.d",
|
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"psraw mm0, mm1": {
|
"psraw mm0, mm1": {
|
||||||
"ExpectedInstructionCount": 6,
|
"ExpectedInstructionCount": 4,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xe1",
|
"Comment": "0x0f 0xe1",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"mov z0.d, d2",
|
"asr z2.h, p6/m, z2.h, z3.d",
|
||||||
"movprfx z2, z3",
|
|
||||||
"asr z2.h, p6/m, z2.h, z0.d",
|
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"psrad mm0, mm1": {
|
"psrad mm0, mm1": {
|
||||||
"ExpectedInstructionCount": 6,
|
"ExpectedInstructionCount": 4,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xe2",
|
"Comment": "0x0f 0xe2",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"mov z0.d, d2",
|
"asr z2.s, p6/m, z2.s, z3.d",
|
||||||
"movprfx z2, z3",
|
|
||||||
"asr z2.s, p6/m, z2.s, z0.d",
|
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"psllw mm0, mm1": {
|
"psllw mm0, mm1": {
|
||||||
"ExpectedInstructionCount": 6,
|
"ExpectedInstructionCount": 4,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xf1",
|
"Comment": "0x0f 0xf1",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"mov z0.d, d2",
|
"lsl z2.h, p6/m, z2.h, z3.d",
|
||||||
"movprfx z2, z3",
|
|
||||||
"lsl z2.h, p6/m, z2.h, z0.d",
|
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"pslld mm0, mm1": {
|
"pslld mm0, mm1": {
|
||||||
"ExpectedInstructionCount": 6,
|
"ExpectedInstructionCount": 4,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xf2",
|
"Comment": "0x0f 0xf2",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"mov z0.d, d2",
|
"lsl z2.s, p6/m, z2.s, z3.d",
|
||||||
"movprfx z2, z3",
|
|
||||||
"lsl z2.s, p6/m, z2.s, z0.d",
|
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
},
|
},
|
||||||
"psllq mm0, mm1": {
|
"psllq mm0, mm1": {
|
||||||
"ExpectedInstructionCount": 6,
|
"ExpectedInstructionCount": 4,
|
||||||
"Optimal": "No",
|
"Optimal": "Yes",
|
||||||
"Comment": "0x0f 0xf3",
|
"Comment": "0x0f 0xf3",
|
||||||
"ExpectedArm64ASM": [
|
"ExpectedArm64ASM": [
|
||||||
"ldr d2, [x28, #768]",
|
"ldr d2, [x28, #752]",
|
||||||
"ldr d3, [x28, #752]",
|
"ldr d3, [x28, #768]",
|
||||||
"mov z0.d, d2",
|
"lsl z2.d, p6/m, z2.d, z3.d",
|
||||||
"movprfx z2, z3",
|
|
||||||
"lsl z2.d, p6/m, z2.d, z0.d",
|
|
||||||
"str d2, [x28, #752]"
|
"str d2, [x28, #752]"
|
||||||
]
|
]
|
||||||
}
|
}
|
||||||
|
|||||||
Reference in new issue
Block a user