Merge pull request #3098 from Sonicadvance1/optimize_vectors_sve

Arm64: Optimize wide shifts slightly for 64-bit OpSize
This commit is contained in:
Mai authored and GitHub committed 2023-09-15 05:21:20 -04:00
commit f5c4e28696
5 files changed
+113 -108

No files matched your search

@@ -2543,16 +2543,23 @@ DEF_OP(VUShrSWide) {
else if (HostSupportsSVE128) { else if (HostSupportsSVE128) {
const auto Mask = PRED_TMP_16B.Merging(); const auto Mask = PRED_TMP_16B.Merging();
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0); auto ShiftRegister = ShiftScalar.Z();
if (OpSize > 8) {
// SVE wide shifts don't need to duplicate the low bits unless the OpSize is 16-bytes
// Slightly more optimal for 8-byte opsize.
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
ShiftRegister = VTMP1.Z();
}
if (Dst != Vector) { if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation. // NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z()); movprfx(Dst.Z(), Vector.Z());
} }
if (ElementSize == 8) { if (ElementSize == 8) {
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z()); lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
} }
else { else {
lsr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z()); lsr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
} }
} else { } else {
// uqshl + ushr of 57-bits leaves 7-bits remaining. // uqshl + ushr of 57-bits leaves 7-bits remaining.
@@ -2603,16 +2610,23 @@ DEF_OP(VSShrSWide) {
else if (HostSupportsSVE128) { else if (HostSupportsSVE128) {
const auto Mask = PRED_TMP_16B.Merging(); const auto Mask = PRED_TMP_16B.Merging();
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0); auto ShiftRegister = ShiftScalar.Z();
if (OpSize > 8) {
// SVE wide shifts don't need to duplicate the low bits unless the OpSize is 16-bytes
// Slightly more optimal for 8-byte opsize.
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
ShiftRegister = VTMP1.Z();
}
if (Dst != Vector) { if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation. // NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z()); movprfx(Dst.Z(), Vector.Z());
} }
if (ElementSize == 8) { if (ElementSize == 8) {
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z()); asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
} }
else { else {
asr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z()); asr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
} }
} else { } else {
// uqshl + ushr of 57-bits leaves 7-bits remaining. // uqshl + ushr of 57-bits leaves 7-bits remaining.
@@ -2663,16 +2677,23 @@ DEF_OP(VUShlSWide) {
else if (HostSupportsSVE128) { else if (HostSupportsSVE128) {
const auto Mask = PRED_TMP_16B.Merging(); const auto Mask = PRED_TMP_16B.Merging();
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0); auto ShiftRegister = ShiftScalar.Z();
if (OpSize > 8) {
// SVE wide shifts don't need to duplicate the low bits unless the OpSize is 16-bytes
// Slightly more optimal for 8-byte opsize.
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
ShiftRegister = VTMP1.Z();
}
if (Dst != Vector) { if (Dst != Vector) {
// NOTE: SVE LSR is a destructive operation. // NOTE: SVE LSR is a destructive operation.
movprfx(Dst.Z(), Vector.Z()); movprfx(Dst.Z(), Vector.Z());
} }
if (ElementSize == 8) { if (ElementSize == 8) {
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z()); lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
} }
else { else {
lsl_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), VTMP1.Z()); lsl_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
} }
} else { } else {
// uqshl + ushr of 57-bits leaves 7-bits remaining. // uqshl + ushr of 57-bits leaves 7-bits remaining.
@@ -1751,8 +1751,8 @@ OrderedNode* OpDispatchBuilder::PSRLDOpImpl(OpcodeArgs, size_t ElementSize,
template<size_t ElementSize> template<size_t ElementSize>
void OpDispatchBuilder::PSRLDOp(OpcodeArgs) { void OpDispatchBuilder::PSRLDOp(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *Result = PSRLDOpImpl(Op, ElementSize, Dest, Src); OrderedNode *Result = PSRLDOpImpl(Op, ElementSize, Dest, Src);
StoreResult(FPRClass, Op, Result, -1); StoreResult(FPRClass, Op, Result, -1);
@@ -1887,8 +1887,8 @@ OrderedNode* OpDispatchBuilder::PSLLImpl(OpcodeArgs, size_t ElementSize,
template<size_t ElementSize> template<size_t ElementSize>
void OpDispatchBuilder::PSLL(OpcodeArgs) { void OpDispatchBuilder::PSLL(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *Result = PSLLImpl(Op, ElementSize, Dest, Src); OrderedNode *Result = PSLLImpl(Op, ElementSize, Dest, Src);
StoreResult(FPRClass, Op, Result, -1); StoreResult(FPRClass, Op, Result, -1);
@@ -1927,8 +1927,8 @@ OrderedNode* OpDispatchBuilder::PSRAOpImpl(OpcodeArgs, size_t ElementSize,
template<size_t ElementSize> template<size_t ElementSize>
void OpDispatchBuilder::PSRAOp(OpcodeArgs) { void OpDispatchBuilder::PSRAOp(OpcodeArgs) {
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1); OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
OrderedNode *Result = PSRAOpImpl(Op, ElementSize, Dest, Src); OrderedNode *Result = PSRAOpImpl(Op, ElementSize, Dest, Src);
StoreResult(FPRClass, Op, Result, -1); StoreResult(FPRClass, Op, Result, -1);
+32 -32
View File
@@ -4182,13 +4182,13 @@
"Optimal": "Yes", "Optimal": "Yes",
"Comment": "0x0f 0xd1", "Comment": "0x0f 0xd1",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"uqshl d0, d2, #57", "uqshl d0, d3, #57",
"ushr d0, d0, #57", "ushr d0, d0, #57",
"dup v0.8h, v0.h[0]", "dup v0.8h, v0.h[0]",
"neg v0.8h, v0.8h", "neg v0.8h, v0.8h",
"ushl v2.8h, v3.8h, v0.8h", "ushl v2.8h, v2.8h, v0.8h",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
@@ -4197,13 +4197,13 @@
"Optimal": "Yes", "Optimal": "Yes",
"Comment": "0x0f 0xd2", "Comment": "0x0f 0xd2",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"uqshl d0, d2, #57", "uqshl d0, d3, #57",
"ushr d0, d0, #57", "ushr d0, d0, #57",
"dup v0.4s, v0.s[0]", "dup v0.4s, v0.s[0]",
"neg v0.4s, v0.4s", "neg v0.4s, v0.4s",
"ushl v2.4s, v3.4s, v0.4s", "ushl v2.4s, v2.4s, v0.4s",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
@@ -4212,13 +4212,13 @@
"Optimal": "Yes", "Optimal": "Yes",
"Comment": "0x0f 0xd3", "Comment": "0x0f 0xd3",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"uqshl d0, d2, #57", "uqshl d0, d3, #57",
"ushr d0, d0, #57", "ushr d0, d0, #57",
"dup v0.2d, v0.d[0]", "dup v0.2d, v0.d[0]",
"neg v0.2d, v0.2d", "neg v0.2d, v0.2d",
"ushl v2.2d, v3.2d, v0.2d", "ushl v2.2d, v2.2d, v0.2d",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
@@ -4367,13 +4367,13 @@
"Optimal": "Yes", "Optimal": "Yes",
"Comment": "0x0f 0xe1", "Comment": "0x0f 0xe1",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"uqshl d0, d2, #57", "uqshl d0, d3, #57",
"ushr d0, d0, #57", "ushr d0, d0, #57",
"dup v0.8h, v0.h[0]", "dup v0.8h, v0.h[0]",
"neg v0.8h, v0.8h", "neg v0.8h, v0.8h",
"sshl v2.8h, v3.8h, v0.8h", "sshl v2.8h, v2.8h, v0.8h",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
@@ -4382,13 +4382,13 @@
"Optimal": "Yes", "Optimal": "Yes",
"Comment": "0x0f 0xe2", "Comment": "0x0f 0xe2",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"uqshl d0, d2, #57", "uqshl d0, d3, #57",
"ushr d0, d0, #57", "ushr d0, d0, #57",
"dup v0.4s, v0.s[0]", "dup v0.4s, v0.s[0]",
"neg v0.4s, v0.4s", "neg v0.4s, v0.4s",
"sshl v2.4s, v3.4s, v0.4s", "sshl v2.4s, v2.4s, v0.4s",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
@@ -4529,12 +4529,12 @@
"Optimal": "Yes", "Optimal": "Yes",
"Comment": "0x0f 0xf1", "Comment": "0x0f 0xf1",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"uqshl d0, d2, #57", "uqshl d0, d3, #57",
"ushr d0, d0, #57", "ushr d0, d0, #57",
"dup v0.8h, v0.h[0]", "dup v0.8h, v0.h[0]",
"ushl v2.8h, v3.8h, v0.8h", "ushl v2.8h, v2.8h, v0.8h",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
@@ -4543,12 +4543,12 @@
"Optimal": "Yes", "Optimal": "Yes",
"Comment": "0x0f 0xf2", "Comment": "0x0f 0xf2",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"uqshl d0, d2, #57", "uqshl d0, d3, #57",
"ushr d0, d0, #57", "ushr d0, d0, #57",
"dup v0.4s, v0.s[0]", "dup v0.4s, v0.s[0]",
"ushl v2.4s, v3.4s, v0.4s", "ushl v2.4s, v2.4s, v0.4s",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
@@ -4557,12 +4557,12 @@
"Optimal": "Yes", "Optimal": "Yes",
"Comment": "0x0f 0xf3", "Comment": "0x0f 0xf3",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"uqshl d0, d2, #57", "uqshl d0, d3, #57",
"ushr d0, d0, #57", "ushr d0, d0, #57",
"dup v0.2d, v0.d[0]", "dup v0.2d, v0.d[0]",
"ushl v2.2d, v3.2d, v0.2d", "ushl v2.2d, v2.2d, v0.2d",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
@@ -9,7 +9,7 @@
"Instructions": { "Instructions": {
"psrlw xmm0, xmm1": { "psrlw xmm0, xmm1": {
"ExpectedInstructionCount": 2, "ExpectedInstructionCount": 2,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x66 0x0f 0xd1", "Comment": "0x66 0x0f 0xd1",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"mov z0.d, d17", "mov z0.d, d17",
@@ -18,7 +18,7 @@
}, },
"psrld xmm0, xmm1": { "psrld xmm0, xmm1": {
"ExpectedInstructionCount": 2, "ExpectedInstructionCount": 2,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x66 0x0f 0xd2", "Comment": "0x66 0x0f 0xd2",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"mov z0.d, d17", "mov z0.d, d17",
@@ -27,7 +27,7 @@
}, },
"psrlq xmm0, xmm1": { "psrlq xmm0, xmm1": {
"ExpectedInstructionCount": 2, "ExpectedInstructionCount": 2,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x66 0x0f 0xd3", "Comment": "0x66 0x0f 0xd3",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"mov z0.d, d17", "mov z0.d, d17",
@@ -36,7 +36,7 @@
}, },
"psraw xmm0, xmm1": { "psraw xmm0, xmm1": {
"ExpectedInstructionCount": 2, "ExpectedInstructionCount": 2,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x66 0x0f 0xe1", "Comment": "0x66 0x0f 0xe1",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"mov z0.d, d17", "mov z0.d, d17",
@@ -45,7 +45,7 @@
}, },
"psrad xmm0, xmm1": { "psrad xmm0, xmm1": {
"ExpectedInstructionCount": 2, "ExpectedInstructionCount": 2,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x66 0x0f 0xe2", "Comment": "0x66 0x0f 0xe2",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"mov z0.d, d17", "mov z0.d, d17",
@@ -70,7 +70,7 @@
}, },
"psllw xmm0, xmm1": { "psllw xmm0, xmm1": {
"ExpectedInstructionCount": 2, "ExpectedInstructionCount": 2,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x66 0x0f 0xf1", "Comment": "0x66 0x0f 0xf1",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"mov z0.d, d17", "mov z0.d, d17",
@@ -79,7 +79,7 @@
}, },
"pslld xmm0, xmm1": { "pslld xmm0, xmm1": {
"ExpectedInstructionCount": 2, "ExpectedInstructionCount": 2,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x66 0x0f 0xf2", "Comment": "0x66 0x0f 0xf2",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"mov z0.d, d17", "mov z0.d, d17",
@@ -88,7 +88,7 @@
}, },
"psllq xmm0, xmm1": { "psllq xmm0, xmm1": {
"ExpectedInstructionCount": 2, "ExpectedInstructionCount": 2,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x66 0x0f 0xf3", "Comment": "0x66 0x0f 0xf3",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"mov z0.d, d17", "mov z0.d, d17",
@@ -34,106 +34,90 @@
] ]
}, },
"psrlw mm0, mm1": { "psrlw mm0, mm1": {
"ExpectedInstructionCount": 6, "ExpectedInstructionCount": 4,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x0f 0xd1", "Comment": "0x0f 0xd1",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"mov z0.d, d2", "lsr z2.h, p6/m, z2.h, z3.d",
"movprfx z2, z3",
"lsr z2.h, p6/m, z2.h, z0.d",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
"psrld mm0, mm1": { "psrld mm0, mm1": {
"ExpectedInstructionCount": 6, "ExpectedInstructionCount": 4,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x0f 0xd2", "Comment": "0x0f 0xd2",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"mov z0.d, d2", "lsr z2.s, p6/m, z2.s, z3.d",
"movprfx z2, z3",
"lsr z2.s, p6/m, z2.s, z0.d",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
"psrlq mm0, mm1": { "psrlq mm0, mm1": {
"ExpectedInstructionCount": 6, "ExpectedInstructionCount": 4,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x0f 0xd3", "Comment": "0x0f 0xd3",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"mov z0.d, d2", "lsr z2.d, p6/m, z2.d, z3.d",
"movprfx z2, z3",
"lsr z2.d, p6/m, z2.d, z0.d",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
"psraw mm0, mm1": { "psraw mm0, mm1": {
"ExpectedInstructionCount": 6, "ExpectedInstructionCount": 4,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x0f 0xe1", "Comment": "0x0f 0xe1",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"mov z0.d, d2", "asr z2.h, p6/m, z2.h, z3.d",
"movprfx z2, z3",
"asr z2.h, p6/m, z2.h, z0.d",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
"psrad mm0, mm1": { "psrad mm0, mm1": {
"ExpectedInstructionCount": 6, "ExpectedInstructionCount": 4,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x0f 0xe2", "Comment": "0x0f 0xe2",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"mov z0.d, d2", "asr z2.s, p6/m, z2.s, z3.d",
"movprfx z2, z3",
"asr z2.s, p6/m, z2.s, z0.d",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
"psllw mm0, mm1": { "psllw mm0, mm1": {
"ExpectedInstructionCount": 6, "ExpectedInstructionCount": 4,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x0f 0xf1", "Comment": "0x0f 0xf1",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"mov z0.d, d2", "lsl z2.h, p6/m, z2.h, z3.d",
"movprfx z2, z3",
"lsl z2.h, p6/m, z2.h, z0.d",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
"pslld mm0, mm1": { "pslld mm0, mm1": {
"ExpectedInstructionCount": 6, "ExpectedInstructionCount": 4,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x0f 0xf2", "Comment": "0x0f 0xf2",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"mov z0.d, d2", "lsl z2.s, p6/m, z2.s, z3.d",
"movprfx z2, z3",
"lsl z2.s, p6/m, z2.s, z0.d",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
}, },
"psllq mm0, mm1": { "psllq mm0, mm1": {
"ExpectedInstructionCount": 6, "ExpectedInstructionCount": 4,
"Optimal": "No", "Optimal": "Yes",
"Comment": "0x0f 0xf3", "Comment": "0x0f 0xf3",
"ExpectedArm64ASM": [ "ExpectedArm64ASM": [
"ldr d2, [x28, #768]", "ldr d2, [x28, #752]",
"ldr d3, [x28, #752]", "ldr d3, [x28, #768]",
"mov z0.d, d2", "lsl z2.d, p6/m, z2.d, z3.d",
"movprfx z2, z3",
"lsl z2.d, p6/m, z2.d, z0.d",
"str d2, [x28, #752]" "str d2, [x28, #752]"
] ]
} }