OpcodeDispatcher: optimize xor-with-self flags

Signed-off-by: Alyssa Rosenzweig <alyssa@rosenzweig.io>
This commit is contained in:
Alyssa Rosenzweig committed 2025-05-28 13:53:38 -04:00
1 parent ece817c691
commit 0fbe69ebcf
1 file changed
+13 -9
@@ -4416,18 +4416,22 @@ void OpDispatchBuilder::MOVGPRNTOp(OpcodeArgs) {
}
void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::IR::IROps AtomicFetchOp, unsigned SrcIdx) {
/* On x86, the canonical way to zero a register is XOR with itself... because
* modern x86 detects this pattern in hardware. arm64 does not detect this
* pattern, we should do it like the x86 hardware would. On arm64, "mov x0,
* #0" is faster than "eor x0, x0, x0". Additionally this lets more constant
* folding kick in for flags.
*/
// On x86, the canonical way to zero a register is XOR with itself. Detect and
// emit optimal arm64 assembly.
if (!DestIsLockedMem(Op) && ALUIROp == FEXCore::IR::IROps::OP_XOR && Op->Dest.IsGPR() && Op->Src[SrcIdx].IsGPR() &&
Op->Dest.Data.GPR == Op->Src[SrcIdx].Data.GPR) {
auto Result = _Constant(0);
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
CalculateFlags_Logical(OpSizeFromSrc(Op), Result);
// Move 0 into the register
StoreResult(GPRClass, Op, _Constant(0), OpSize::iInvalid);
// Set flags for zero result with inverted carry. We subtract an arbitrary
// register from itself to get the zero, since `subs wzr, #0` is not
// encodable. This is optimal and works regardless of the opsize.
auto Zero = LoadGPR(Op->Dest.Data.GPR.GPR);
HandleNZ00Write();
InvalidateAF();
CalculatePF(_SubWithFlags(OpSize::i32Bit, Zero, Zero));
CFInverted = true;
return;
}