mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-10 23:00:30 +02:00
In order to support `vmaskmov{ps,pd}` without SVE128 this is required.
It's pretty gnarly but they aren't often used so that's fine from a
compatibility perspective.
Example SVE128 implementation:
```json
"vmaskmovps ymm0, ymm1, [rax]": {
"ExpectedInstructionCount": 9,
"Comment": [
"Map 2 0b01 0x2c 256-bit"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mrs x20, nzcv",
"cmplt p0.s, p6/z, z17.s, #0",
"ld1w {z16.s}, p0/z, [x4]",
"add x21, x4, #0x10 (16)",
"cmplt p0.s, p6/z, z2.s, #0",
"ld1w {z2.s}, p0/z, [x21]",
"str q2, [x28, #16]",
"msr nzcv, x20"
]
},
```
Example ASIMD implementation
```json
"vmaskmovps ymm0, ymm1, [rax]": {
"ExpectedInstructionCount": 37,
"Comment": [
"Map 2 0b01 0x2c 256-bit"
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mrs x20, nzcv",
"movi v0.2d, #0x0",
"mov x1, x4",
"mov x0, v17.d[0]",
"tbz x0, #63, #+0x8",
"ld1 {v0.s}[0], [x1]",
"add x1, x1, #0x4 (4)",
"tbz w0, #31, #+0x8",
"ld1 {v0.s}[1], [x1]",
"add x1, x1, #0x4 (4)",
"mov x0, v17.d[1]",
"tbz x0, #63, #+0x8",
"ld1 {v0.s}[2], [x1]",
"add x1, x1, #0x4 (4)",
"tbz w0, #31, #+0x8",
"ld1 {v0.s}[3], [x1]",
"mov v16.16b, v0.16b",
"add x21, x4, #0x10 (16)",
"movi v0.2d, #0x0",
"mov x1, x21",
"mov x0, v2.d[0]",
"tbz x0, #63, #+0x8",
"ld1 {v0.s}[0], [x1]",
"add x1, x1, #0x4 (4)",
"tbz w0, #31, #+0x8",
"ld1 {v0.s}[1], [x1]",
"add x1, x1, #0x4 (4)",
"mov x0, v2.d[1]",
"tbz x0, #63, #+0x8",
"ld1 {v0.s}[2], [x1]",
"add x1, x1, #0x4 (4)",
"tbz w0, #31, #+0x8",
"ld1 {v0.s}[3], [x1]",
"mov v2.16b, v0.16b",
"str q2, [x28, #16]",
"msr nzcv, x20"
]
},
```
There's a little bit of an improvement where nzcv isn't needed to get
touched on the ASIMD implementation, but I'll leave that for a future
improvement.