diff --git a/FEXCore/Source/Interface/Core/JIT/ALUOps.cpp b/FEXCore/Source/Interface/Core/JIT/ALUOps.cpp index d91675915..9e29e7fd9 100644 --- a/FEXCore/Source/Interface/Core/JIT/ALUOps.cpp +++ b/FEXCore/Source/Interface/Core/JIT/ALUOps.cpp @@ -515,6 +515,12 @@ DEF_OP(AndWithFlags) { } } +DEF_OP(AndShift) { + auto Op = IROp->C(); + + and_(ConvertSize48(IROp), GetReg(Node), GetReg(Op->Src1), GetReg(Op->Src2), ConvertIRShiftType(Op->Shift), Op->ShiftAmount); +} + DEF_OP(XorShift) { auto Op = IROp->C(); diff --git a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp index 3fdf16481..0402410a9 100644 --- a/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp +++ b/FEXCore/Source/Interface/Core/OpcodeDispatcher/X87.cpp @@ -32,21 +32,25 @@ Ref OpDispatchBuilder::GetX87Top() { } void OpDispatchBuilder::SetX87FTW(Ref FTW) { - Ref X87Empty = Constant(static_cast(FPState::X87Tag::Empty)); - Ref NewAbridgedFTW {}; + // For the output, we want a 1-bit for each pair not equal to 11 (Empty). + static_assert(static_cast(FPState::X87Tag::Empty) == 0b11); - for (int i = 0; i < 8; i++) { - Ref RegTag = _Bfe(OpSize::i32Bit, 2, i * 2, FTW); - Ref RegValid = _Select(FEXCore::IR::COND_NEQ, RegTag, X87Empty, Constant(1), Constant(0)); + // Make even bits 1 if the pair is equal to 11, and 0 otherwise. + FTW = _AndShift(OpSize::i32Bit, FTW, FTW, ShiftType::LSR, 1); - if (i) { - NewAbridgedFTW = _Orlshl(OpSize::i32Bit, NewAbridgedFTW, RegValid, i); - } else { - NewAbridgedFTW = RegValid; - } - } + // Invert FTW and clear the odd bits. Even bits are 1 if the pair + // is not equal to 11, and odd bits are 0. + FTW = _Andn(OpSize::i32Bit, Constant(0x55555555), FTW); - StoreContext(AbridgedFTWIndex, NewAbridgedFTW); + // All that's left is to compact away the odd bits. That is a Morton + // deinterleave operation, which has a standard solution. See + // https://stackoverflow.com/questions/3137266/how-to-de-interleave-bits-unmortonizing + FTW = _And(OpSize::i32Bit, _Orlshr(OpSize::i32Bit, FTW, FTW, 1), Constant(0x33333333)); + FTW = _And(OpSize::i32Bit, _Orlshr(OpSize::i32Bit, FTW, FTW, 2), Constant(0x0f0f0f0f)); + FTW = _Orlshr(OpSize::i32Bit, FTW, FTW, 4); + + // ...and that's it. StoreContext implicitly does the final masking. + StoreContext(AbridgedFTWIndex, FTW); } void OpDispatchBuilder::SetX87Top(Ref Value) { diff --git a/FEXCore/Source/Interface/IR/IR.json b/FEXCore/Source/Interface/IR/IR.json index 651411257..63f7d1f89 100644 --- a/FEXCore/Source/Interface/IR/IR.json +++ b/FEXCore/Source/Interface/IR/IR.json @@ -1309,6 +1309,13 @@ "Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit" ] }, + "GPR = AndShift OpSize:#Size, GPR:$Src1, GPR:$Src2, ShiftType:$Shift{ShiftType::LSL}, u8:$ShiftAmount{0}": { + "Desc": [ "Integer binary and with shifted register"], + "DestSize": "Size", + "EmitValidation": [ + "Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit" + ] + }, "GPR = AndWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": { "Desc": ["Integer binary and" ], diff --git a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp index befd4fef1..295986fb9 100644 --- a/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp +++ b/FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp @@ -380,22 +380,20 @@ private: } // Try to coalesce reserved pairs. Just a heuristic to remove some moves. - if (IROp->Op == OP_ALLOCATEGPR) { - if (IROp->C()->ForPair) { - uint32_t Available = Classes[GPRClass].Available; + if (IROp->Op == OP_ALLOCATEGPR && IROp->C()->ForPair) { + uint32_t Available = Classes[GPRClass].Available; - // Only choose base register R if R and R + 1 are both free - Available &= (Available >> 1); + // Only choose base register R if R and R + 1 are both free + Available &= (Available >> 1); - // Only consider aligned registers in the pair region - constexpr uint32_t EVEN_BITS = 0x55555555; - Available &= (EVEN_BITS & ((1u << PairRegs) - 1)); + // Only consider aligned registers in the pair region + constexpr uint32_t EVEN_BITS = 0x55555555; + Available &= (EVEN_BITS & ((1u << PairRegs) - 1)); - if (Available) { - unsigned Reg = std::countr_zero(Available); - SetReg(CodeNode, PhysicalRegister(GPRClass, Reg)); - return; - } + if (Available) { + unsigned Reg = std::countr_zero(Available); + SetReg(CodeNode, PhysicalRegister(GPRClass, Reg)); + return; } } else if (IROp->Op == OP_ALLOCATEGPRAFTER) { uint32_t Available = Classes[GPRClass].Available; diff --git a/unittests/InstructionCountCI/FlagM/x87.json b/unittests/InstructionCountCI/FlagM/x87.json index b07636eac..325e73d0e 100644 --- a/unittests/InstructionCountCI/FlagM/x87.json +++ b/unittests/InstructionCountCI/FlagM/x87.json @@ -2101,7 +2101,7 @@ ] }, "fldenv [rax]": { - "ExpectedInstructionCount": 50, + "ExpectedInstructionCount": 25, "Comment": [ "0xd9 !11b /4" ], @@ -2122,39 +2122,14 @@ "strb w20, [x28, #1022]", "add x20, x4, #0x8 (8)", "ldr w20, [x20]", - "ubfx w21, w20, #0, #2", - "mrs x22, nzcv", - "cmp x21, #0x3 (3)", - "cset x21, ne", - "ubfx w23, w20, #2, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #1", - "ubfx w23, w20, #4, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #2", - "ubfx w23, w20, #6, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #3", - "ubfx w23, w20, #8, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #4", - "ubfx w23, w20, #10, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #5", - "ubfx w23, w20, #12, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #6", - "ubfx w20, w20, #14, #2", - "cmp x20, #0x3 (3)", - "cset x20, ne", - "orr w20, w21, w20, lsl #7", - "msr nzcv, x22", + "and w20, w20, w20, lsr #1", + "mov w21, #0x55555555", + "bic w20, w21, w20", + "orr w20, w20, w20, lsr #1", + "and w20, w20, #0x33333333", + "orr w20, w20, w20, lsr #2", + "and w20, w20, #0xf0f0f0f", + "orr w20, w20, w20, lsr #4", "strb w20, [x28, #1426]" ] }, @@ -7049,7 +7024,7 @@ ] }, "frstor [rax]": { - "ExpectedInstructionCount": 99, + "ExpectedInstructionCount": 74, "Comment": [ "0xdd !11b /4" ], @@ -7068,42 +7043,18 @@ "strb w24, [x28, #1018]", "strb w20, [x28, #1022]", "ldr w20, [x4, #8]", - "ubfx w22, w20, #0, #2", - "mrs x23, nzcv", - "cmp x22, #0x3 (3)", - "cset x22, ne", - "ubfx w24, w20, #2, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #1", - "ubfx w24, w20, #4, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #2", - "ubfx w24, w20, #6, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #3", - "ubfx w24, w20, #8, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #4", - "ubfx w24, w20, #10, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #5", - "ubfx w24, w20, #12, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #6", - "ubfx w20, w20, #14, #2", - "cmp x20, #0x3 (3)", - "cset x20, ne", - "orr w20, w22, w20, lsl #7", + "and w20, w20, w20, lsr #1", + "mov w22, #0x55555555", + "bic w20, w22, w20", + "orr w20, w20, w20, lsr #1", + "and w20, w20, #0x33333333", + "orr w20, w20, w20, lsr #2", + "and w20, w20, #0xf0f0f0f", + "orr w20, w20, w20, lsr #4", "mov x22, #0xffffffffffffffff", - "mov w24, #0xffff", + "mov w23, #0xffff", "fmov d2, x22", - "fmov v2.D[1], x24", + "fmov v2.D[1], x23", "ldur q3, [x4, #28]", "and v3.16b, v3.16b, v2.16b", "add x0, x28, x21, lsl #4", @@ -7151,7 +7102,6 @@ "mov v2.h[4], v3.h[0]", "add x0, x28, x21, lsl #4", "str q2, [x0, #1040]", - "msr nzcv, x23", "strb w20, [x28, #1426]" ] }, diff --git a/unittests/InstructionCountCI/FlagM/x87_f64.json b/unittests/InstructionCountCI/FlagM/x87_f64.json index 48dbd6430..2a0dc8fe0 100644 --- a/unittests/InstructionCountCI/FlagM/x87_f64.json +++ b/unittests/InstructionCountCI/FlagM/x87_f64.json @@ -1551,7 +1551,7 @@ ] }, "fldenv [rax]": { - "ExpectedInstructionCount": 56, + "ExpectedInstructionCount": 31, "Comment": [ "0xd9 !11b /4" ], @@ -1578,39 +1578,14 @@ "strb w23, [x28, #1018]", "strb w20, [x28, #1022]", "ldr w20, [x4, #8]", - "ubfx w21, w20, #0, #2", - "mrs x22, nzcv", - "cmp x21, #0x3 (3)", - "cset x21, ne", - "ubfx w23, w20, #2, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #1", - "ubfx w23, w20, #4, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #2", - "ubfx w23, w20, #6, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #3", - "ubfx w23, w20, #8, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #4", - "ubfx w23, w20, #10, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #5", - "ubfx w23, w20, #12, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #6", - "ubfx w20, w20, #14, #2", - "cmp x20, #0x3 (3)", - "cset x20, ne", - "orr w20, w21, w20, lsl #7", - "msr nzcv, x22", + "and w20, w20, w20, lsr #1", + "mov w21, #0x55555555", + "bic w20, w21, w20", + "orr w20, w20, w20, lsr #1", + "and w20, w20, #0x33333333", + "orr w20, w20, w20, lsr #2", + "and w20, w20, #0xf0f0f0f", + "orr w20, w20, w20, lsr #4", "strb w20, [x28, #1426]" ] }, @@ -5689,7 +5664,7 @@ ] }, "frstor [rax]": { - "ExpectedInstructionCount": 164, + "ExpectedInstructionCount": 139, "Comment": [ "0xdd !11b /4" ], @@ -5717,42 +5692,18 @@ "strb w24, [x28, #1018]", "strb w20, [x28, #1022]", "ldr w20, [x4, #8]", - "ubfx w22, w20, #0, #2", - "mrs x23, nzcv", - "cmp x22, #0x3 (3)", - "cset x22, ne", - "ubfx w24, w20, #2, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #1", - "ubfx w24, w20, #4, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #2", - "ubfx w24, w20, #6, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #3", - "ubfx w24, w20, #8, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #4", - "ubfx w24, w20, #10, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #5", - "ubfx w24, w20, #12, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #6", - "ubfx w20, w20, #14, #2", - "cmp x20, #0x3 (3)", - "cset x20, ne", - "orr w20, w22, w20, lsl #7", + "and w20, w20, w20, lsr #1", + "mov w22, #0x55555555", + "bic w20, w22, w20", + "orr w20, w20, w20, lsr #1", + "and w20, w20, #0x33333333", + "orr w20, w20, w20, lsr #2", + "and w20, w20, #0xf0f0f0f", + "orr w20, w20, w20, lsr #4", "mov x22, #0xffffffffffffffff", - "mov w24, #0xffff", + "mov w23, #0xffff", "fmov d2, x22", - "fmov v2.D[1], x24", + "fmov v2.D[1], x23", "ldur q3, [x4, #28]", "and v3.16b, v3.16b, v2.16b", "str x30, [sp, #-16]!", @@ -5856,7 +5807,6 @@ "fmov d2, d0", "add x0, x28, x21, lsl #4", "str d2, [x0, #1040]", - "msr nzcv, x23", "strb w20, [x28, #1426]" ] }, diff --git a/unittests/InstructionCountCI/x87.json b/unittests/InstructionCountCI/x87.json index 4a967115e..8fa364601 100644 --- a/unittests/InstructionCountCI/x87.json +++ b/unittests/InstructionCountCI/x87.json @@ -2100,7 +2100,7 @@ ] }, "fldenv [rax]": { - "ExpectedInstructionCount": 50, + "ExpectedInstructionCount": 25, "Comment": [ "0xd9 !11b /4" ], @@ -2121,39 +2121,14 @@ "strb w20, [x28, #1022]", "add x20, x4, #0x8 (8)", "ldr w20, [x20]", - "ubfx w21, w20, #0, #2", - "mrs x22, nzcv", - "cmp x21, #0x3 (3)", - "cset x21, ne", - "ubfx w23, w20, #2, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #1", - "ubfx w23, w20, #4, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #2", - "ubfx w23, w20, #6, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #3", - "ubfx w23, w20, #8, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #4", - "ubfx w23, w20, #10, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #5", - "ubfx w23, w20, #12, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #6", - "ubfx w20, w20, #14, #2", - "cmp x20, #0x3 (3)", - "cset x20, ne", - "orr w20, w21, w20, lsl #7", - "msr nzcv, x22", + "and w20, w20, w20, lsr #1", + "mov w21, #0x55555555", + "bic w20, w21, w20", + "orr w20, w20, w20, lsr #1", + "and w20, w20, #0x33333333", + "orr w20, w20, w20, lsr #2", + "and w20, w20, #0xf0f0f0f", + "orr w20, w20, w20, lsr #4", "strb w20, [x28, #1426]" ] }, @@ -7080,7 +7055,7 @@ ] }, "frstor [rax]": { - "ExpectedInstructionCount": 99, + "ExpectedInstructionCount": 74, "Comment": [ "0xdd !11b /4" ], @@ -7099,42 +7074,18 @@ "strb w24, [x28, #1018]", "strb w20, [x28, #1022]", "ldr w20, [x4, #8]", - "ubfx w22, w20, #0, #2", - "mrs x23, nzcv", - "cmp x22, #0x3 (3)", - "cset x22, ne", - "ubfx w24, w20, #2, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #1", - "ubfx w24, w20, #4, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #2", - "ubfx w24, w20, #6, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #3", - "ubfx w24, w20, #8, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #4", - "ubfx w24, w20, #10, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #5", - "ubfx w24, w20, #12, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #6", - "ubfx w20, w20, #14, #2", - "cmp x20, #0x3 (3)", - "cset x20, ne", - "orr w20, w22, w20, lsl #7", + "and w20, w20, w20, lsr #1", + "mov w22, #0x55555555", + "bic w20, w22, w20", + "orr w20, w20, w20, lsr #1", + "and w20, w20, #0x33333333", + "orr w20, w20, w20, lsr #2", + "and w20, w20, #0xf0f0f0f", + "orr w20, w20, w20, lsr #4", "mov x22, #0xffffffffffffffff", - "mov w24, #0xffff", + "mov w23, #0xffff", "fmov d2, x22", - "fmov v2.D[1], x24", + "fmov v2.D[1], x23", "ldur q3, [x4, #28]", "and v3.16b, v3.16b, v2.16b", "add x0, x28, x21, lsl #4", @@ -7182,7 +7133,6 @@ "mov v2.h[4], v3.h[0]", "add x0, x28, x21, lsl #4", "str q2, [x0, #1040]", - "msr nzcv, x23", "strb w20, [x28, #1426]" ] }, diff --git a/unittests/InstructionCountCI/x87_f64.json b/unittests/InstructionCountCI/x87_f64.json index 8944a725a..54398a65d 100644 --- a/unittests/InstructionCountCI/x87_f64.json +++ b/unittests/InstructionCountCI/x87_f64.json @@ -1568,7 +1568,7 @@ ] }, "fldenv [rax]": { - "ExpectedInstructionCount": 56, + "ExpectedInstructionCount": 31, "Comment": [ "0xd9 !11b /4" ], @@ -1595,39 +1595,14 @@ "strb w23, [x28, #1018]", "strb w20, [x28, #1022]", "ldr w20, [x4, #8]", - "ubfx w21, w20, #0, #2", - "mrs x22, nzcv", - "cmp x21, #0x3 (3)", - "cset x21, ne", - "ubfx w23, w20, #2, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #1", - "ubfx w23, w20, #4, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #2", - "ubfx w23, w20, #6, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #3", - "ubfx w23, w20, #8, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #4", - "ubfx w23, w20, #10, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #5", - "ubfx w23, w20, #12, #2", - "cmp x23, #0x3 (3)", - "cset x23, ne", - "orr w21, w21, w23, lsl #6", - "ubfx w20, w20, #14, #2", - "cmp x20, #0x3 (3)", - "cset x20, ne", - "orr w20, w21, w20, lsl #7", - "msr nzcv, x22", + "and w20, w20, w20, lsr #1", + "mov w21, #0x55555555", + "bic w20, w21, w20", + "orr w20, w20, w20, lsr #1", + "and w20, w20, #0x33333333", + "orr w20, w20, w20, lsr #2", + "and w20, w20, #0xf0f0f0f", + "orr w20, w20, w20, lsr #4", "strb w20, [x28, #1426]" ] }, @@ -5728,7 +5703,7 @@ ] }, "frstor [rax]": { - "ExpectedInstructionCount": 164, + "ExpectedInstructionCount": 139, "Comment": [ "0xdd !11b /4" ], @@ -5756,42 +5731,18 @@ "strb w24, [x28, #1018]", "strb w20, [x28, #1022]", "ldr w20, [x4, #8]", - "ubfx w22, w20, #0, #2", - "mrs x23, nzcv", - "cmp x22, #0x3 (3)", - "cset x22, ne", - "ubfx w24, w20, #2, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #1", - "ubfx w24, w20, #4, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #2", - "ubfx w24, w20, #6, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #3", - "ubfx w24, w20, #8, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #4", - "ubfx w24, w20, #10, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #5", - "ubfx w24, w20, #12, #2", - "cmp x24, #0x3 (3)", - "cset x24, ne", - "orr w22, w22, w24, lsl #6", - "ubfx w20, w20, #14, #2", - "cmp x20, #0x3 (3)", - "cset x20, ne", - "orr w20, w22, w20, lsl #7", + "and w20, w20, w20, lsr #1", + "mov w22, #0x55555555", + "bic w20, w22, w20", + "orr w20, w20, w20, lsr #1", + "and w20, w20, #0x33333333", + "orr w20, w20, w20, lsr #2", + "and w20, w20, #0xf0f0f0f", + "orr w20, w20, w20, lsr #4", "mov x22, #0xffffffffffffffff", - "mov w24, #0xffff", + "mov w23, #0xffff", "fmov d2, x22", - "fmov v2.D[1], x24", + "fmov v2.D[1], x23", "ldur q3, [x4, #28]", "and v3.16b, v3.16b, v2.16b", "str x30, [sp, #-16]!", @@ -5895,7 +5846,6 @@ "fmov d2, d0", "add x0, x28, x21, lsl #4", "str d2, [x0, #1040]", - "msr nzcv, x23", "strb w20, [x28, #1426]" ] },