From 3d23cd5765199de9711e18912e8232dc0690f15c Mon Sep 17 00:00:00 2001 From: Lioncache Date: Thu, 19 Oct 2023 11:40:07 +0200 Subject: [PATCH] VectorOps: Handle SVE VFDiv a little better In the event no source vectors alias the destination, we can just move the first source vector into it and then perform the divide without needing to move afterword. --- .../Interface/Core/JIT/Arm64/VectorOps.cpp | 11 +++- unittests/ASM/VEX/vdivpd.asm | 11 +++- unittests/ASM/VEX/vdivps.asm | 11 +++- unittests/InstructionCountCI/VEX_map1.json | 60 ++++++++++++++++--- 4 files changed, 81 insertions(+), 12 deletions(-) diff --git a/FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp b/FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp index c684aa293..bb1fe4251 100644 --- a/FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/Arm64/VectorOps.cpp @@ -1672,12 +1672,17 @@ DEF_OP(VFDiv) { // Trivial case where we already have source data to be divided in the // destination register. We can just divide by Vector2 and be done with it. fdiv(SubRegSize, Dst.Z(), Mask, Dst.Z(), Vector2.Z()); - } else { - // SVE FDIV is a destructive operation, so we need a temporary - // in the event that Dst and Vector1 don't alias. + } else if (Dst == Vector2) { + // If the destination aliases the second vector, then we need + // to use a temp. movprfx(VTMP1.Z(), Vector1.Z()); fdiv(SubRegSize, VTMP1.Z(), Mask, VTMP1.Z(), Vector2.Z()); mov(Dst.Z(), VTMP1.Z()); + } else { + // If no registers alias the destination, then we can move directly + // into the destination and then divide. + movprfx(Dst.Z(), Vector1.Z()); + fdiv(SubRegSize, Dst.Z(), Mask, Dst.Z(), Vector2.Z()); } } else { if (IsScalar) { diff --git a/unittests/ASM/VEX/vdivpd.asm b/unittests/ASM/VEX/vdivpd.asm index 81c1f7759..b70156047 100644 --- a/unittests/ASM/VEX/vdivpd.asm +++ b/unittests/ASM/VEX/vdivpd.asm @@ -7,7 +7,9 @@ "XMM2": ["0x3FE0000000000000", "0x3FE0000000000000", "0x0000000000000000", "0x0000000000000000"], "XMM3": ["0x3FE0000000000000", "0x3FE0000000000000", "0x0000000000000000", "0x0000000000000000"], "XMM4": ["0x3FE0000000000000", "0x3FE0000000000000", "0x3FE0000000000000", "0x3FE0000000000000"], - "XMM5": ["0x4000000000000000", "0x4000000000000000", "0x4000000000000000", "0x4000000000000000"] + "XMM5": ["0x4000000000000000", "0x4000000000000000", "0x4000000000000000", "0x4000000000000000"], + "XMM6": ["0x3FE0000000000000", "0x3FE0000000000000", "0x3FE0000000000000", "0x3FE0000000000000"], + "XMM7": ["0x4000000000000000", "0x4000000000000000", "0x4000000000000000", "0x4000000000000000"] }, "MemoryRegions": { "0x100000000": "4096" @@ -28,6 +30,13 @@ vdivpd ymm4, ymm0, [rdx + 32] vdivpd xmm3, xmm0, xmm1 vdivpd ymm5, ymm1, ymm0 +; Some tests for aliasing destination and source vectors +vmovapd ymm6, ymm0 +vdivpd ymm6, ymm6, ymm1 + +vmovapd ymm7, ymm0 +vdivpd ymm7, ymm1, ymm7 + hlt align 32 diff --git a/unittests/ASM/VEX/vdivps.asm b/unittests/ASM/VEX/vdivps.asm index 37897a77f..3fa78eeda 100644 --- a/unittests/ASM/VEX/vdivps.asm +++ b/unittests/ASM/VEX/vdivps.asm @@ -7,7 +7,9 @@ "XMM2": ["0x3EAAAAAB3E4CCCCD", "0x3F0000003EDB6DB7", "0x0000000000000000", "0x0000000000000000"], "XMM3": ["0x3EAAAAAB3E4CCCCD", "0x3F0000003EDB6DB7", "0x0000000000000000", "0x0000000000000000"], "XMM4": ["0x3EAAAAAB3E4CCCCD", "0x3F0000003EDB6DB7", "0x3EAAAAAB3E4CCCCD", "0x3F0000003EDB6DB7"], - "XMM5": ["0x4040000040A00000", "0x4000000040155555", "0x4040000040A00000", "0x4000000040155555"] + "XMM5": ["0x4040000040A00000", "0x4000000040155555", "0x4040000040A00000", "0x4000000040155555"], + "XMM6": ["0x3EAAAAAB3E4CCCCD", "0x3F0000003EDB6DB7", "0x3EAAAAAB3E4CCCCD", "0x3F0000003EDB6DB7"], + "XMM7": ["0x4040000040A00000", "0x4000000040155555", "0x4040000040A00000", "0x4000000040155555"] }, "MemoryRegions": { "0x100000000": "4096" @@ -28,6 +30,13 @@ vdivps ymm4, ymm0, [rdx + 32] vdivps xmm3, xmm0, xmm1 vdivps ymm5, ymm1, ymm0 +; Some tests for aliasing destination and source vectors +vmovapd ymm6, ymm0 +vdivps ymm6, ymm6, ymm1 + +vmovapd ymm7, ymm0 +vdivps ymm7, ymm1, ymm7 + hlt align 32 diff --git a/unittests/InstructionCountCI/VEX_map1.json b/unittests/InstructionCountCI/VEX_map1.json index 88385101b..efd80f37b 100644 --- a/unittests/InstructionCountCI/VEX_map1.json +++ b/unittests/InstructionCountCI/VEX_map1.json @@ -3764,18 +3764,41 @@ "fdiv v16.4s, v17.4s, v18.4s" ] }, - "vdivps ymm0, ymm1, ymm2": { - "ExpectedInstructionCount": 3, - "Optimal": "No", + "vdivps ymm0, ymm0, ymm2": { + "ExpectedInstructionCount": 1, + "Optimal": "Yes", "Comment": [ + "Aliasing source and destination", + "Map 1 0b00 0x5e 256-bit" + ], + "ExpectedArm64ASM": [ + "fdiv z16.s, p7/m, z16.s, z18.s" + ] + }, + "vdivps ymm0, ymm1, ymm0": { + "ExpectedInstructionCount": 3, + "Optimal": "Yes", + "Comment": [ + "Aliasing source and destination", "Map 1 0b00 0x5e 256-bit" ], "ExpectedArm64ASM": [ "movprfx z0, z17", - "fdiv z0.s, p7/m, z0.s, z18.s", + "fdiv z0.s, p7/m, z0.s, z16.s", "mov z16.d, z0.d" ] }, + "vdivps ymm0, ymm1, ymm2": { + "ExpectedInstructionCount": 2, + "Optimal": "Yes", + "Comment": [ + "Map 1 0b00 0x5e 256-bit" + ], + "ExpectedArm64ASM": [ + "movprfx z16, z17", + "fdiv z16.s, p7/m, z16.s, z18.s" + ] + }, "vdivpd xmm0, xmm1, xmm2": { "ExpectedInstructionCount": 1, "Optimal": "Yes", @@ -3786,18 +3809,41 @@ "fdiv v16.2d, v17.2d, v18.2d" ] }, - "vdivpd ymm0, ymm1, ymm2": { + "vdivpd ymm0, ymm1, ymm0": { "ExpectedInstructionCount": 3, - "Optimal": "No", + "Optimal": "Yes", "Comment": [ + "Aliasing source and destination", "Map 1 0b01 0x5e 256-bit" ], "ExpectedArm64ASM": [ "movprfx z0, z17", - "fdiv z0.d, p7/m, z0.d, z18.d", + "fdiv z0.d, p7/m, z0.d, z16.d", "mov z16.d, z0.d" ] }, + "vdivpd ymm0, ymm0, ymm2": { + "ExpectedInstructionCount": 1, + "Optimal": "Yes", + "Comment": [ + "Aliasing source and destination", + "Map 1 0b01 0x5e 256-bit" + ], + "ExpectedArm64ASM": [ + "fdiv z16.d, p7/m, z16.d, z18.d" + ] + }, + "vdivpd ymm0, ymm1, ymm2": { + "ExpectedInstructionCount": 2, + "Optimal": "Yes", + "Comment": [ + "Map 1 0b01 0x5e 256-bit" + ], + "ExpectedArm64ASM": [ + "movprfx z16, z17", + "fdiv z16.d, p7/m, z16.d, z18.d" + ] + }, "vdivss xmm0, xmm1, xmm2": { "ExpectedInstructionCount": 3, "Optimal": "Yes",