Merge pull request #4027 from alyssarosenzweig/opt/global-flag

Global flag optimizations
This commit is contained in:
Ryan Houdek authored and GitHub committed 2024-09-08 09:42:18 -07:00
commit 1c59bfeb9f
24 files changed
+1077 -750

No files matched your search

-1
View File
@@ -140,7 +140,6 @@ set (SRCS
Interface/IR/Passes/IRValidation.cpp
Interface/IR/Passes/RAValidation.cpp
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
Interface/IR/Passes/DeadStoreElimination.cpp
Interface/IR/Passes/RegisterAllocationPass.cpp
Interface/IR/Passes/x87StackOptimizationPass.cpp
Utils/Telemetry.cpp
@@ -5,6 +5,7 @@ tags: backend|arm64
$end_info$
*/
#include "CodeEmitter/Emitter.h"
#include "FEXCore/IR/IR.h"
#include "Interface/Context/Context.h"
#include "Interface/Core/JIT/Arm64/JITClass.h"
@@ -190,6 +191,30 @@ DEF_OP(TestNZ) {
}
}
DEF_OP(TestZ) {
auto Op = IROp->C<IR::IROp_TestZ>();
LOGMAN_THROW_AA_FMT(IROp->Size < 4, "TestNZ used at higher sizes");
const auto EmitSize = ARMEmitter::Size::i32Bit;
uint64_t Const;
uint64_t Mask = IROp->Size == 8 ? ~0ULL : ((1ull << (IROp->Size * 8)) - 1);
auto Src1 = GetReg(Op->Src1.ID());
if (IsInlineConstant(Op->Src2, &Const)) {
// We can promote 8/16-bit tests to 32-bit since the constant is masked.
LOGMAN_THROW_AA_FMT(!(Const & ~Mask), "constant is already masked");
tst(EmitSize, Src1, Const);
} else {
const auto Src2 = GetReg(Op->Src2.ID());
if (Src1 == Src2) {
tst(EmitSize, Src1 /* Src2 */, Mask);
} else {
and_(EmitSize, TMP1, Src1, Src2);
tst(EmitSize, TMP1, Mask);
}
}
}
DEF_OP(SubShift) {
auto Op = IROp->C<IR::IROp_SubShift>();
@@ -272,8 +297,43 @@ DEF_OP(SetSmallNZV) {
}
DEF_OP(AXFlag) {
LOGMAN_THROW_A_FMT(CTX->HostFeatures.SupportsFlagM2, "Unsupported flagm2 op");
axflag();
if (CTX->HostFeatures.SupportsFlagM2) {
axflag();
} else {
// AXFLAG is defined in the Arm spec as
//
// gt: nzCv -> nzCv
// lt: Nzcv -> nzcv <==> 1 + 0
// eq: nZCv -> nZCv <==> 1 + (~0)
// un: nzCV -> nZcv <==> 0 + 0
//
// For the latter 3 cases, we therefore get the right NZCV by adding V_inv
// to (eq ? ~0 : 0). The remaining case is forced with ccmn.
auto V_inv = GetReg(IROp->Args[0].ID());
csetm(ARMEmitter::Size::i64Bit, TMP1, ARMEmitter::Condition::CC_EQ);
ccmn(ARMEmitter::Size::i64Bit, V_inv, TMP1, ARMEmitter::StatusFlags {0x2} /* nzCv */, ARMEmitter::Condition::CC_LE);
}
}
DEF_OP(Parity) {
auto Op = IROp->C<IR::IROp_Parity>();
auto Raw = GetReg(Op->Raw.ID());
auto Dest = GetReg(Node);
// Cascade to calculate parity of bottom 8-bits to bottom bit.
eor(ARMEmitter::Size::i32Bit, TMP1, Raw, Raw, ARMEmitter::ShiftType::LSR, 4);
eor(ARMEmitter::Size::i32Bit, TMP1, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 2);
if (Op->Invert) {
eon(ARMEmitter::Size::i32Bit, Dest, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 1);
} else {
eor(ARMEmitter::Size::i32Bit, Dest, TMP1, TMP1, ARMEmitter::ShiftType::LSR, 1);
}
// The above sequence leaves garbage in the upper bits.
if (Op->Mask) {
and_(ARMEmitter::Size::i32Bit, Dest, Dest, 1);
}
}
DEF_OP(CondAddNZCV) {
@@ -136,12 +136,8 @@ DEF_OP(LoadRegister) {
const auto OpSize = IROp->Size;
if (Op->Class == IR::GPRClass) {
unsigned Reg = Op->Reg == Core::CPUState::PF_AS_GREG ? (StaticRegisters.size() - 2) :
Op->Reg == Core::CPUState::AF_AS_GREG ? (StaticRegisters.size() - 1) :
Op->Reg;
LOGMAN_THROW_A_FMT(Reg < StaticRegisters.size(), "out of range reg");
const auto reg = StaticRegisters[Reg];
LOGMAN_THROW_A_FMT(Op->Reg < StaticRegisters.size(), "out of range reg");
const auto reg = StaticRegisters[Op->Reg];
if (GetReg(Node).Idx() != reg.Idx()) {
if (OpSize == 4) {
@@ -170,6 +166,30 @@ DEF_OP(LoadRegister) {
}
}
DEF_OP(LoadPF) {
const auto reg = StaticRegisters[StaticRegisters.size() - 2];
if (GetReg(Node).Idx() != reg.Idx()) {
if (IROp->Size == 4) {
mov(GetReg(Node).W(), reg.W());
} else {
mov(GetReg(Node).X(), reg.X());
}
}
}
DEF_OP(LoadAF) {
const auto reg = StaticRegisters[StaticRegisters.size() - 1];
if (GetReg(Node).Idx() != reg.Idx()) {
if (IROp->Size == 4) {
mov(GetReg(Node).W(), reg.W());
} else {
mov(GetReg(Node).X(), reg.X());
}
}
}
DEF_OP(StoreRegister) {
const auto Op = IROp->C<IR::IROp_StoreRegister>();
@@ -206,6 +226,28 @@ DEF_OP(StoreRegister) {
}
}
DEF_OP(StorePF) {
const auto Op = IROp->C<IR::IROp_StorePF>();
const auto reg = StaticRegisters[StaticRegisters.size() - 2];
const auto Src = GetReg(Op->Value.ID());
if (Src.Idx() != reg.Idx()) {
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
mov(ARMEmitter::Size::i64Bit, reg, Src);
}
}
DEF_OP(StoreAF) {
const auto Op = IROp->C<IR::IROp_StoreAF>();
const auto reg = StaticRegisters[StaticRegisters.size() - 1];
const auto Src = GetReg(Op->Value.ID());
if (Src.Idx() != reg.Idx()) {
// Always use 64-bit, it's faster. Upper bits ignored for 32-bit mode.
mov(ARMEmitter::Size::i64Bit, reg, Src);
}
}
DEF_OP(LoadContextIndexed) {
const auto Op = IROp->C<IR::IROp_LoadContextIndexed>();
const auto OpSize = IROp->Size;
@@ -598,18 +598,19 @@ void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) {
ExitFunction(JMPPCOffset); // If we get here then leave the function now
}
Ref OpDispatchBuilder::SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue) {
Ref OpDispatchBuilder::SelectPF(bool Invert, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue) {
uint64_t TrueConst, FalseConst;
if (IsValueConstant(WrapNode(TrueValue), &TrueConst) && IsValueConstant(WrapNode(FalseValue), &FalseConst) && FalseConst == 0) {
if (TrueConst == 1) {
return _And(ResultSize, Cmp, _Constant(1));
return LoadPFRaw(true, Invert);
} else if (TrueConst == 0xffffffff) {
return _Sbfe(OpSize::i32Bit, 1, 0, Cmp);
return _Sbfe(OpSize::i32Bit, 1, 0, LoadPFRaw(false, Invert));
} else if (TrueConst == 0xffffffffffffffffull) {
return _Sbfe(OpSize::i64Bit, 1, 0, Cmp);
return _Sbfe(OpSize::i64Bit, 1, 0, LoadPFRaw(false, Invert));
}
}
Ref Cmp = LoadPFRaw(false, Invert);
SaveNZCV();
// Because we're only clobbering NZCV internally, we ignore all carry flag
@@ -681,12 +682,10 @@ Ref OpDispatchBuilder::SelectCC(uint8_t OP, IR::OpSize ResultSize, Ref TrueValue
}
switch (OP) {
case 0xA: { // JP - Jump if PF == 1
// Raw value contains inverted PF in bottom bit
return SelectBit(LoadPFRaw(true), ResultSize, TrueValue, FalseValue);
}
case 0xA: // JP - Jump if PF == 1
case 0xB: { // JNP - Jump if PF == 0
return SelectBit(LoadPFRaw(false), ResultSize, TrueValue, FalseValue);
// Raw value contains inverted PF in bottom bit
return SelectPF(OP == 0xA, ResultSize, TrueValue, FalseValue);
}
default: LOGMAN_MSG_A_FMT("Unknown CC Op: 0x{:x}\n", OP); return nullptr;
}
@@ -759,7 +758,7 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
auto [Complex, SimpleCond] = DecodeNZCVCondition(OP);
if (Complex) {
LOGMAN_THROW_AA_FMT(OP == 0xA || OP == 0xB, "only PF left");
CondJump_ = CondJumpBit(LoadPFRaw(false), 0, OP == 0xB);
CondJump_ = CondJumpBit(LoadPFRaw(false, false), 0, OP == 0xB);
} else {
CondJump_ = CondJumpNZCV(SimpleCond);
}
@@ -151,11 +151,7 @@ public:
}
IRPair<IROp_CondJump> CondJumpNZCV(CondClassType Cond) {
FlushRegisterCache();
// The jump will ignore the sources, so it doesn't matter what we put here.
// Put an inline constant so RA+codegen will ignore altogether.
auto Placeholder = _InlineConstant(0);
return _CondJump(Placeholder, Placeholder, InvalidNode, InvalidNode, Cond, 0, true);
return _CondJump(InvalidNode, InvalidNode, InvalidNode, InvalidNode, Cond, 0, true);
}
IRPair<IROp_CondJump> CondJumpBit(Ref Src, unsigned Bit, bool Set) {
FlushRegisterCache();
@@ -1236,8 +1232,12 @@ public:
uint32_t Index = 63 - std::countl_zero(Bits);
Ref Value = RegCache.Value[Index];
if (Index >= GPR0Index && Index <= AFIndex) {
if (Index >= GPR0Index && Index <= GPR15Index) {
_StoreRegister(Value, Index - GPR0Index, GPRClass, GPRSize);
} else if (Index == PFIndex) {
_StorePF(Value, GPRSize);
} else if (Index == AFIndex) {
_StoreAF(Value, GPRSize);
} else if (Index >= FPR0Index && Index <= FPR15Index) {
_StoreRegister(Value, Index - FPR0Index, FPRClass, VectorSize);
} else if (Index == DFIndex) {
@@ -1896,6 +1896,10 @@ private:
if (Size == 8) {
RegCache.Partial |= Bit;
}
} else if (Index == PFIndex) {
RegCache.Value[Index] = _LoadPF(Size);
} else if (Index == AFIndex) {
RegCache.Value[Index] = _LoadAF(Size);
} else {
RegCache.Value[Index] = _LoadRegister(Offset, RegClass, Size);
}
@@ -2106,22 +2110,9 @@ private:
// Convert NZCV from the Arm representation to an eXternal representation
// that's totally not a euphemism for x86, nuh-uh. But maps to exactly we
// need, what a coincidence!
if (CTX->HostFeatures.SupportsFlagM2) {
_AXFlag();
} else {
// AXFLAG is defined in the Arm spec as
//
// gt: nzCv -> nzCv
// lt: Nzcv -> nzcv <==> 1 + 0
// eq: nZCv -> nZCv <==> 1 + (~0)
// un: nzCV -> nZcv <==> 0 + 0
//
// For the latter 3 cases, we therefore get the right NZCV by adding V_inv
// to (eq ? ~0 : 0). The remaining case is forced with ccmn.
Ref Eq = NZCVSelect(OpSize::i64Bit, {COND_EQ}, _Constant(~0ull), _Constant(0));
_CondAddNZCV(OpSize::i64Bit, V_inv, Eq, {COND_FLEU}, 0x2 /* nzCv */);
}
//
// Our AXFlag emulation on FlagM2-less systems needs V_inv passed.
_AXFlag(CTX->HostFeatures.SupportsFlagM2 ? Invalid() : V_inv);
PossiblySetNZCVBits = ~0;
CFInverted = true;
}
@@ -2135,7 +2126,7 @@ private:
LOGMAN_THROW_A_FMT(!NZCVDirty, "only expected after fcmp");
// Convert to x86 flags, saves us from or'ing after.
_AXFlag();
_AXFlag(Invalid());
PossiblySetNZCVBits = ~0;
CFInverted = true;
@@ -2321,7 +2312,8 @@ private:
/**
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
* @{ */
Ref LoadPFRaw(bool Invert);
Ref LoadPFRaw(bool Mask, bool Invert);
Ref SelectPF(bool Invert, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
Ref LoadAF();
void FixupAF();
void SetAFAndFixup(Ref AF);
@@ -117,7 +117,7 @@ Ref OpDispatchBuilder::GetPackedRFLAG(uint32_t FlagsMask) {
// instead.
if (FlagsMask & (1 << FEXCore::X86State::RFLAG_PF_RAW_LOC)) {
// Set every bit except the bottommost.
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(false), _Constant(~1ull));
auto OnesInvPF = _Or(OpSize::i64Bit, LoadPFRaw(false, false), _Constant(~1ull));
// Rotate the bottom bit to the appropriate location for PF, so we get
// something like 111P1111. Then invert that to get 000p0000. Then OR that
@@ -178,22 +178,12 @@ void OpDispatchBuilder::CalculateOF(uint8_t SrcSize, Ref Res, Ref Src1, Ref Src2
SetRFLAG<FEXCore::X86State::RFLAG_OF_RAW_LOC>(Anded, SrcSize * 8 - 1, true);
}
Ref OpDispatchBuilder::LoadPFRaw(bool Invert) {
// Read the stored byte. This is the original result (up to 64-bits), it needs
// parity calculated.
auto Result = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
Ref OpDispatchBuilder::LoadPFRaw(bool Mask, bool Invert) {
// Most blocks do not read parity, so PF optimization is gated on this flag.
CurrentHeader->ReadsParity = true;
// Cascade to calculate parity of bottom 8-bits to bottom bit.
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 4);
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 2);
if (Invert) {
Result = _XornShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1);
} else {
Result = _XorShift(OpSize::i32Bit, Result, Result, ShiftType::LSR, 1);
}
return Result;
// Evaluate parity on the deferred raw value.
return _Parity(GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC), Mask, Invert);
}
Ref OpDispatchBuilder::LoadAF() {
+36 -4
View File
@@ -168,7 +168,7 @@
"SwitchGen": false,
"JITDispatchOverride": "NoOp"
},
"IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions, i1:$HasX87{false}": {
"IRHeader SSA:$Blocks, u64:$OriginalRIP, u32:$BlockCount, u32:$NumHostInstructions, i1:$HasX87{false}, i1:$ReadsParity{false}": {
"SwitchGen": false,
"JITDispatchOverride": "NoOp"
},
@@ -366,14 +366,36 @@
"DestSize": "Size"
},
"GPR = LoadPF u8:#Size": {
"Desc": ["Loads raw PF"],
"DestSize": "Size"
},
"GPR = LoadAF u8:#Size": {
"Desc": ["Loads raw PF"],
"DestSize": "Size"
},
"StoreRegister SSA:$Value, u32:$Reg, RegisterClass:$Class, u8:#Size": {
"HasSideEffects": true,
"HasSideEffects": true,
"Desc": ["Stores a value to a given register.",
"Size must match the execution mode."],
"DestSize": "Size",
"EmitValidation": [
"WalkFindRegClass($Value) == $Class"
]
},
"StorePF GPR:$Value, u8:#Size": {
"HasSideEffects": true,
"Desc": ["Stores raw PF"],
"DestSize": "Size"
},
"StoreAF GPR:$Value, u8:#Size": {
"HasSideEffects": true,
"Desc": ["Stores raw AF"],
"DestSize": "Size"
}
},
"Memory": {
@@ -1139,10 +1161,15 @@
"Desc": ["Invert carry flag in NZCV"],
"HasSideEffects": true
},
"AXFlag": {
"Desc": ["After an FCmp, converts NZCV flags from the Arm format to a mysterious eXternal format"],
"AXFlag GPR:$V_inv": {
"Desc": ["After an FCmp, converts NZCV flags from the Arm format to a mysterious eXternal format",
"On FlagM2-less platforms, takes the inverted 1/0 overflow flag"],
"HasSideEffects": true
},
"GPR = Parity GPR:$Raw, i1:$Mask, i1:$Invert": {
"Desc": ["Calculates PF"],
"DestSize": "4"
},
"RmifNZCV GPR:$Src, u8:$Rotate, u8:$Mask": {
"Desc": ["Rotate, mask, and insert into NZCV on FlagM platforms"],
"HasSideEffects": true
@@ -1331,6 +1358,11 @@
"DestSize": "Size",
"HasSideEffects": true
},
"TestZ OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Set NZCV for the binary AND of two GPRs, setting Z accordingly and zeroing C and V. N is undefined."],
"DestSize": "Size",
"HasSideEffects": true
},
"GPR = Lshl OpSize:#Size, GPR:$Src1, GPR:$Src2": {
"Desc": ["Integer logical shift left"
],
@@ -71,7 +71,6 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
if (!DisablePasses()) {
InsertPass(CreateX87StackOptimizationPass());
InsertPass(CreateDeadStoreElimination());
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID));
InsertPass(CreateDeadFlagCalculationEliminination());
}
-1
View File
@@ -18,7 +18,6 @@ class RegisterAllocationData;
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID);
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination();
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass();
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass();
@@ -1,154 +0,0 @@
// SPDX-License-Identifier: MIT
/*
$info$
tags: ir|opts
desc: Cross block store-after-store elimination
$end_info$
*/
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <memory>
#include <stddef.h>
#include <stdint.h>
namespace FEXCore::IR {
constexpr int PropagationRounds = 5;
// Return a bit representing a single GPR or FPR.
static inline uint64_t RegBit(RegisterClassType Class, uint32_t Reg) {
uint32_t AdjustedReg = (Class == FPRClass) ? (32 + Reg) : Reg;
return 1ULL << AdjustedReg;
}
class DeadStoreElimination final : public FEXCore::IR::Pass {
public:
void Run(IREmitter* IREmit) override;
};
struct ReadWriteKill {
uint64_t reads {0};
uint64_t writes {0};
uint64_t kill {0};
};
struct Info {
ReadWriteKill reg;
};
/**
* @brief This is a temporary pass to detect simple multiblock dead reg stores
*
* First pass computes which regs are read and written per block
*
* Second pass computes which regs are stored, but overwritten by the next block(s).
* It also propagates this information a few times to catch dead regs across multiple blocks.
*
* Third pass removes the dead stores.
*
*/
void DeadStoreElimination::Run(IREmitter* IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::DSE");
auto CurrentIR = IREmit->ViewIR();
fextl::vector<Info> InfoMap(CurrentIR.GetSSACount());
// Pass 1
// Compute regs read/writes per block
// This is conservative and doesn't try to be smart about loads after writes
{
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
if (IROp->Op == OP_STOREREGISTER) {
auto Op = IROp->C<IR::IROp_StoreRegister>();
BlockInfo.reg.writes |= RegBit(Op->Class, Op->Reg);
} else if (IROp->Op == OP_LOADREGISTER) {
auto Op = IROp->C<IR::IROp_LoadRegister>();
BlockInfo.reg.reads |= RegBit(Op->Class, Op->Reg);
} else if (IROp->Op == OP_INVALIDATEFLAGS) {
auto Op = IROp->C<IR::IROp_InvalidateFlags>();
if (Op->Flags & (1u << X86State::RFLAG_PF_RAW_LOC)) {
BlockInfo.reg.writes |= RegBit(GPRClass, Core::CPUState::PF_AS_GREG);
}
if (Op->Flags & (1u << X86State::RFLAG_AF_RAW_LOC)) {
BlockInfo.reg.writes |= RegBit(GPRClass, Core::CPUState::AF_AS_GREG);
}
}
}
}
}
// Pass 2
// Compute flags/registers that are stored, but always ovewritten in the next blocks
// Propagate the information a few times to eliminate more
for (int i = 0; i < PropagationRounds; i++) {
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
auto CodeBlock = BlockIROp->C<IROp_CodeBlock>();
auto IROp = CurrentIR.GetNode(CurrentIR.GetNode(CodeBlock->Last)->Header.Previous)->Op(CurrentIR.GetData());
if (IROp->Op == OP_JUMP) {
auto Op = IROp->C<IR::IROp_Jump>();
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
auto& TargetInfo = InfoMap[Op->Header.Args[0].ID().Value];
// stores to remove are written by the next block but not read
BlockInfo.reg.kill = TargetInfo.reg.writes & ~(TargetInfo.reg.reads) & ~BlockInfo.reg.reads;
// If written by the next block can be considered as written by this block, if not read
BlockInfo.reg.writes |= BlockInfo.reg.kill & ~BlockInfo.reg.reads;
} else if (IROp->Op == OP_CONDJUMP) {
auto Op = IROp->C<IR::IROp_CondJump>();
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
auto& TrueTargetInfo = InfoMap[Op->TrueBlock.ID().Value];
auto& FalseTargetInfo = InfoMap[Op->FalseBlock.ID().Value];
// stores to remove are written by the next blocks but not read
BlockInfo.reg.kill = TrueTargetInfo.reg.writes & ~(TrueTargetInfo.reg.reads) & ~BlockInfo.reg.reads;
BlockInfo.reg.kill &= FalseTargetInfo.reg.writes & ~(FalseTargetInfo.reg.reads) & ~BlockInfo.reg.reads;
// if written by the next blocks can be considered as written by this block, if not read
BlockInfo.reg.writes |= BlockInfo.reg.kill & ~BlockInfo.reg.reads;
}
}
}
// Pass 3
// Remove the dead stores
{
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
auto& BlockInfo = InfoMap[CurrentIR.GetID(BlockNode).Value];
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
if (IROp->Op == OP_STOREREGISTER) {
auto Op = IROp->C<IR::IROp_StoreRegister>();
// If this OP_STOREREGISTER is never read, remove it
if (BlockInfo.reg.kill & RegBit(Op->Class, Op->Reg)) {
IREmit->Remove(CodeNode);
}
}
}
}
}
}
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination() {
return fextl::make_unique<DeadStoreElimination>();
}
} // namespace FEXCore::IR
@@ -7,6 +7,9 @@ $end_info$
*/
#include "FEXCore/Core/X86Enums.h"
#include "FEXCore/Utils/CompilerDefs.h"
#include "FEXCore/Utils/MathUtils.h"
#include "FEXCore/fextl/deque.h"
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
@@ -15,16 +18,13 @@ $end_info$
#include "Interface/IR/PassManager.h"
#include <array>
#include <memory>
// Flag bit flags
#define FLAG_V (1U << 0)
#define FLAG_C (1U << 1)
#define FLAG_Z (1U << 2)
#define FLAG_N (1U << 3)
#define FLAG_A (1U << 4)
#define FLAG_P (1U << 5)
#define FLAG_P (1U << 4)
#define FLAG_A (1U << 5)
#define FLAG_ZCV (FLAG_Z | FLAG_C | FLAG_V)
#define FLAG_NZCV (FLAG_N | FLAG_ZCV)
@@ -32,10 +32,7 @@ $end_info$
namespace FEXCore::IR {
struct FlagInfo {
// If set, all following fields are zero, used for a quick exit.
bool Trivial;
struct FlagInfoUnpacked {
// Set of flags read by the instruction.
unsigned Read;
@@ -46,15 +43,106 @@ struct FlagInfo {
// eliminated.
bool CanEliminate;
// If true, the opcode can be replaced with Replacement if its flag writes can
// all be eliminated.
bool CanReplace;
// If set, the opcode can be replaced with Replacement if its flag writes can
// all be eliminated, or ReplacementNoWrite if its register write can be
// eliminated.
IROps Replacement;
// If true, the opcode can be replaced with ReplacementNoWrite if its register
// write is unused but its flags are still needed.
bool CanReplaceWrite;
IROps ReplacementNoWrite;
// Needs speical handling
bool Special;
};
struct FlagInfo {
uint64_t Raw;
static constexpr struct FlagInfo Pack(struct FlagInfoUnpacked F) {
uint64_t R = F.Read | (F.Write << 8) | (F.CanEliminate << 16) | (((uint64_t)F.Replacement) << 32) |
((uint64_t)F.ReplacementNoWrite << 48) | (F.Special ? (1ull << 63) : 0);
return {.Raw = R};
}
bool Trivial() {
return Raw == 0;
}
unsigned Read() {
return Bits(0, 8);
}
unsigned Write() {
return Bits(8, 8);
}
bool CanEliminate() {
return Bits(16, 1);
}
bool Special() {
return Bits(63, 1);
}
IROps Replacement() {
return (IROps)Bits(32, 16);
}
IROps ReplacementNoWrite() {
return (IROps)Bits(48, 16);
}
private:
unsigned Bits(unsigned Start, unsigned Count) {
return (Raw >> Start) & ((1u << Count) - 1);
}
};
struct BlockInfo {
fextl::vector<Ref> Predecessors;
uint8_t Flags;
bool InWorklist;
};
struct ControlFlowGraph {
fextl::unordered_map<uint32_t, BlockInfo> BlockMap;
IRListView& IR;
void AddBlock(fextl::deque<Ref>& Worklist, Ref Block) {
uint32_t ID = IR.GetID(Block).Value;
// Add the block with conservative flags and already in the worklist.
auto Info = &BlockMap.emplace(ID, BlockInfo {{}, FLAG_ALL, true}).first->second;
// Add some initial capacity
Info->Predecessors.reserve(2);
// Add to worklist
Worklist.push_back(Block);
}
BlockInfo* Get(uint32_t Block) {
return &BlockMap.try_emplace(Block).first->second;
}
BlockInfo* Get(Ref Block) {
return Get(IR.GetID(Block).Value);
}
BlockInfo* Get(OrderedNodeWrapper Block) {
return Get(Block.ID().Value);
}
void RecordEdge(Ref From, Ref To) {
auto Info = Get(To);
Info->Predecessors.push_back(From);
}
void AddWorklist(fextl::deque<Ref>& Worklist, Ref Block) {
auto Info = Get(Block);
if (!Info->InWorklist) {
Info->InWorklist = true;
Worklist.push_front(Block);
}
}
};
class DeadFlagCalculationEliminination final : public FEXCore::IR::Pass {
@@ -66,10 +154,10 @@ private:
unsigned FlagForReg(unsigned Reg);
unsigned FlagsForCondClassType(CondClassType Cond);
bool EliminateDeadCode(IREmitter* IREmit, Ref CodeNode, IROp_Header* IROp);
};
unsigned DeadFlagCalculationEliminination::FlagForReg(unsigned Reg) {
return Reg == Core::CPUState::PF_AS_GREG ? FLAG_P : Reg == Core::CPUState::AF_AS_GREG ? FLAG_A : 0;
void FoldBranch(IREmitter* IREmit, IRListView& CurrentIR, IROp_CondJump* Op, Ref CodeNode);
CondClassType X86ToArmFloatCond(CondClassType X86);
bool ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG);
void OptimizeParity(IREmitter* IREmit, IRListView& CurrentIR, ControlFlowGraph& CFG);
};
unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType Cond) {
@@ -107,160 +195,186 @@ unsigned DeadFlagCalculationEliminination::FlagsForCondClassType(CondClassType C
}
}
FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
switch (IROp->Op) {
constexpr FlagInfo ClassifyConst(IROps Op) {
switch (Op) {
case OP_ANDWITHFLAGS:
return {
return FlagInfo::Pack({
.Write = FLAG_NZCV,
.CanReplace = true,
.Replacement = OP_AND,
};
.ReplacementNoWrite = OP_TESTNZ,
});
case OP_ADDWITHFLAGS:
return {
return FlagInfo::Pack({
.Write = FLAG_NZCV,
.CanReplace = true,
.Replacement = OP_ADD,
.CanReplaceWrite = true,
.ReplacementNoWrite = OP_ADDNZCV,
};
});
case OP_SUBWITHFLAGS:
return {
return FlagInfo::Pack({
.Write = FLAG_NZCV,
.CanReplace = true,
.Replacement = OP_SUB,
.CanReplaceWrite = true,
.ReplacementNoWrite = OP_SUBNZCV,
};
});
case OP_ADCWITHFLAGS:
return {
return FlagInfo::Pack({
.Read = FLAG_C,
.Write = FLAG_NZCV,
.CanReplace = true,
.Replacement = OP_ADC,
.CanReplaceWrite = true,
.ReplacementNoWrite = OP_ADCNZCV,
};
});
case OP_ADCZEROWITHFLAGS:
return {
return FlagInfo::Pack({
.Read = FLAG_C,
.Write = FLAG_NZCV,
.CanReplace = true,
.Replacement = OP_ADCZERO,
};
});
case OP_SBBWITHFLAGS:
return {
return FlagInfo::Pack({
.Read = FLAG_C,
.Write = FLAG_NZCV,
.CanReplace = true,
.Replacement = OP_SBB,
.CanReplaceWrite = true,
.ReplacementNoWrite = OP_SBBNZCV,
};
});
case OP_SHIFTFLAGS:
// _ShiftFlags conditionally sets NZCV+PF, which we model here as a
// read-modify-write. Logically, it also conditionally makes AF undefined,
// which we model by omitting AF from both Read and Write sets (since
// "cond ? AF : undef" may be optimized to "AF").
return {
return FlagInfo::Pack({
.Read = FLAG_NZCV | FLAG_P,
.Write = FLAG_NZCV | FLAG_P,
.CanEliminate = true,
};
});
case OP_ROTATEFLAGS:
// _RotateFlags conditionally sets CV, again modeled as RMW.
return {
return FlagInfo::Pack({
.Read = FLAG_C | FLAG_V,
.Write = FLAG_C | FLAG_V,
.CanEliminate = true,
};
});
case OP_RDRAND: return {.Write = FLAG_NZCV};
case OP_RDRAND: return FlagInfo::Pack({.Write = FLAG_NZCV});
case OP_ADDNZCV:
case OP_SUBNZCV:
case OP_TESTNZ:
case OP_FCMP:
case OP_STORENZCV:
return {
return FlagInfo::Pack({
.Write = FLAG_NZCV,
.CanEliminate = true,
};
});
case OP_AXFLAG:
// Per the Arm spec, axflag reads Z/V/C but not N. It writes all flags.
return {
return FlagInfo::Pack({
.Read = FLAG_ZCV,
.Write = FLAG_NZCV,
.CanEliminate = true,
};
});
case OP_CMPPAIRZ:
return {
return FlagInfo::Pack({
.Write = FLAG_Z,
.CanEliminate = true,
};
});
case OP_CARRYINVERT:
return {
return FlagInfo::Pack({
.Read = FLAG_C,
.Write = FLAG_C,
.CanEliminate = true,
};
});
case OP_SETSMALLNZV:
return {
return FlagInfo::Pack({
.Write = FLAG_N | FLAG_Z | FLAG_V,
.CanEliminate = true,
};
});
case OP_LOADNZCV: return {.Read = FLAG_NZCV};
case OP_LOADNZCV: return FlagInfo::Pack({.Read = FLAG_NZCV});
case OP_ADC:
case OP_SBB: return {.Read = FLAG_C};
case OP_ADCZERO:
case OP_SBB: return FlagInfo::Pack({.Read = FLAG_C});
case OP_ADCNZCV:
case OP_SBBNZCV:
return {
return FlagInfo::Pack({
.Read = FLAG_C,
.Write = FLAG_NZCV,
.CanEliminate = true,
};
});
case OP_LOADPF: return FlagInfo::Pack({.Read = FLAG_P});
case OP_LOADAF: return FlagInfo::Pack({.Read = FLAG_A});
case OP_STOREPF: return FlagInfo::Pack({.Write = FLAG_P, .CanEliminate = true});
case OP_STOREAF: return FlagInfo::Pack({.Write = FLAG_A, .CanEliminate = true});
case OP_NZCVSELECT:
case OP_NZCVSELECTINCREMENT:
case OP_NEG:
case OP_CONDJUMP:
case OP_CONDSUBNZCV:
case OP_CONDADDNZCV:
case OP_RMIFNZCV:
case OP_INVALIDATEFLAGS: return FlagInfo::Pack({.Special = true});
default: return FlagInfo::Pack({});
}
}
constexpr auto FlagInfos = std::invoke([] {
std::array<FlagInfo, OP_LAST> ret = {};
for (unsigned i = 0; i < OP_LAST; ++i) {
ret[i] = ClassifyConst((IROps)i);
}
return ret;
});
FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
FlagInfo Info = FlagInfos[IROp->Op];
if (!Info.Special()) {
return Info;
}
switch (IROp->Op) {
case OP_NZCVSELECT:
case OP_NZCVSELECTINCREMENT: {
auto Op = IROp->CW<IR::IROp_NZCVSelect>();
return {.Read = FlagsForCondClassType(Op->Cond)};
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
}
case OP_NEG: {
auto Op = IROp->CW<IR::IROp_Neg>();
return {.Read = FlagsForCondClassType(Op->Cond)};
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
}
case OP_CONDJUMP: {
auto Op = IROp->CW<IR::IROp_CondJump>();
if (!Op->FromNZCV) {
break;
return FlagInfo::Pack({});
}
return {.Read = FlagsForCondClassType(Op->Cond)};
return FlagInfo::Pack({.Read = FlagsForCondClassType(Op->Cond)});
}
case OP_CONDSUBNZCV:
case OP_CONDADDNZCV: {
auto Op = IROp->CW<IR::IROp_CondAddNZCV>();
return {
return FlagInfo::Pack({
.Read = FlagsForCondClassType(Op->Cond),
.Write = FLAG_NZCV,
.CanEliminate = true,
};
});
}
case OP_RMIFNZCV: {
@@ -271,10 +385,10 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
static_assert(FLAG_C == (1 << 1), "rmif mask lines up with our bits");
static_assert(FLAG_V == (1 << 0), "rmif mask lines up with our bits");
return {
return FlagInfo::Pack({
.Write = Op->Mask,
.CanEliminate = true,
};
});
}
case OP_INVALIDATEFLAGS: {
@@ -309,39 +423,16 @@ FlagInfo DeadFlagCalculationEliminination::Classify(IROp_Header* IROp) {
// The mental model of InvalidateFlags is writing undefined values to all
// of the selected flags, allowing the write-after-write optimizations to
// optimize invalidate-after-write for free.
return {
return FlagInfo::Pack({
.Write = Flags,
.CanEliminate = true,
};
});
}
case OP_LOADREGISTER: {
auto Op = IROp->CW<IR::IROp_LoadRegister>();
if (Op->Class != GPRClass) {
break;
}
return {.Read = FlagForReg(Op->Reg)};
default: LOGMAN_THROW_AA_FMT(false, "invalid special op"); FEX_UNREACHABLE;
}
case OP_STOREREGISTER: {
auto Op = IROp->CW<IR::IROp_StoreRegister>();
if (Op->Class != GPRClass) {
break;
}
unsigned Flag = FlagForReg(Op->Reg);
return {
.Write = Flag,
.CanEliminate = Flag != 0,
};
}
default: break;
}
return {.Trivial = true};
FEX_UNREACHABLE;
}
// General purpose dead code elimination. Returns whether flag handling should
@@ -385,83 +476,295 @@ bool DeadFlagCalculationEliminination::EliminateDeadCode(IREmitter* IREmit, Ref
return true;
}
CondClassType DeadFlagCalculationEliminination::X86ToArmFloatCond(CondClassType X86) {
// Table of x86 condition codes that map to arm64 condition codes, in the
// sense that fcmp+axflag+branch(x86) is equivalent to fcmp+branch(arm).
//
// E would be "equal or unordered", no condition code.
// G would be "greater than or less than", no condition code.
//
// SF/OF conditions are trivial and therefore shouldn't actually be generated
switch (X86) {
case COND_UGE /* A */: return {COND_FGE} /* GE */;
case COND_UGT /* AE */: return {COND_FGT} /* GT */;
case COND_ULT /* B */: return {COND_SLT} /* LT */;
case COND_ULE /* BE */: return {COND_SLE} /* LE */;
case COND_SLE /* LE */: return {COND_SLE} /* LE */;
default: return {COND_AL};
}
}
void DeadFlagCalculationEliminination::FoldBranch(IREmitter* IREmit, IRListView& CurrentIR, IROp_CondJump* Op, Ref CodeNode) {
// Skip past StoreRegisters at the end -- they don't touch flags.
auto PrevWrap = CodeNode->Header.Previous;
while (CurrentIR.GetOp<IR::IROp_Header>(PrevWrap)->Op == OP_STOREREGISTER ||
CurrentIR.GetOp<IR::IROp_Header>(PrevWrap)->Op == OP_STOREPF || CurrentIR.GetOp<IR::IROp_Header>(PrevWrap)->Op == OP_STOREAF) {
PrevWrap = CurrentIR.GetNode(PrevWrap)->Header.Previous;
}
auto Prev = CurrentIR.GetOp<IR::IROp_Header>(PrevWrap);
if (Prev->Op == OP_AXFLAG) {
// Pattern match a branch fed by AXFLAG.
CondClassType ArmCond = X86ToArmFloatCond(Op->Cond);
if (ArmCond == COND_AL) {
return;
}
Op->Cond = ArmCond;
} else if (Prev->Op == OP_SUBNZCV) {
// Pattern match a branch fed by a compare. We could also handle bit tests
// here, but tbz/tbnz has a limited offset range which we don't have a way to
// deal with yet. Let's hope that's not a big deal.
if (!(Op->Cond == COND_NEQ || Op->Cond == COND_EQ) || (Prev->Size < 4)) {
return;
}
auto SecondArg = CurrentIR.GetOp<IR::IROp_Header>(Prev->Args[1]);
if (SecondArg->Op != OP_INLINECONSTANT || SecondArg->C<IR::IROp_InlineConstant>()->Constant != 0) {
return;
}
// We've matched. Fold the compare into branch.
IREmit->ReplaceNodeArgument(CodeNode, 0, CurrentIR.GetNode(Prev->Args[0]));
IREmit->ReplaceNodeArgument(CodeNode, 1, CurrentIR.GetNode(Prev->Args[1]));
Op->FromNZCV = false;
Op->CompareSize = Prev->Size;
} else {
return;
}
// The compare/test/axflag sets flags but does not write registers. Flags are
// dead after the jump. The jump does not read flags anymore. There is no
// intervening instruction. Therefore the compare is dead.
IREmit->Remove(CurrentIR.GetNode(PrevWrap));
}
/**
* @brief This pass removes dead code locally.
*/
bool DeadFlagCalculationEliminination::ProcessBlock(IREmitter* IREmit, IRListView& CurrentIR, Ref Block, ControlFlowGraph& CFG) {
uint32_t FlagsRead = FLAG_ALL;
// Reverse iteration is not yet working with the iterators
auto BlockIROp = CurrentIR.GetOp<IR::IROp_CodeBlock>(Block);
// We grab these nodes this way so we can iterate easily
auto CodeBegin = CurrentIR.at(BlockIROp->Begin);
auto CodeLast = CurrentIR.at(BlockIROp->Last);
// Advance past EndBlock to get at the exit.
--CodeLast;
// Initialize the FlagsRead mask according to the exit instruction.
auto [ExitNode, ExitOp] = CodeLast();
if (ExitOp->Op == IR::OP_CONDJUMP) {
auto Op = ExitOp->CW<IR::IROp_CondJump>();
FlagsRead = CFG.Get(Op->TrueBlock)->Flags | CFG.Get(Op->FalseBlock)->Flags;
} else if (ExitOp->Op == IR::OP_JUMP) {
FlagsRead = CFG.Get(ExitOp->Args[0])->Flags;
}
// Iterate the block in reverse
while (true) {
auto [CodeNode, IROp] = CodeLast();
// Optimizing flags can cause earlier flag reads to become dead but dead
// flag reads should not impede optimiation of earlier dead flag writes.
// We must DCE as we go to ensure we converge in a single iteration.
if (!EliminateDeadCode(IREmit, CodeNode, IROp)) {
// Optimiation algorithm: For each flag written...
//
// If the flag has a later read (per FlagsRead), remove the flag from
// FlagsRead, since the reader is covered by this write.
//
// Else, there is no later read, so remove the flag write (if we can).
// This is the active part of the optimization.
//
// Then, add each flag read to FlagsRead.
//
// This order is important: instructions that read-modify-write flags
// (like adcs) first read flags, then write flags. Since we're iterating
// the block backwards, that means we handle the write first.
struct FlagInfo Info = Classify(IROp);
if (!Info.Trivial()) {
bool Eliminated = false;
if ((FlagsRead & Info.Write()) == 0) {
if ((Info.CanEliminate() || Info.Replacement()) && CodeNode->GetUses() == 0) {
IREmit->Remove(CodeNode);
Eliminated = true;
} else if (Info.Replacement()) {
IROp->Op = Info.Replacement();
}
} else if (Info.ReplacementNoWrite() && CodeNode->GetUses() == 0) {
IROp->Op = Info.ReplacementNoWrite();
}
// If we don't care about the sign or carry, we can optimize testnz.
// Carry is inverted between testz and testnz so we check that too. Note
// this flag is outside of the if, since the TestNZ might result from
// optimizing AndWithFlags, and we need to converge locally in a single
// iteration.
if (IROp->Op == OP_TESTNZ && IROp->Size < 4 && !(FlagsRead & (FLAG_N | FLAG_C))) {
IROp->Op = OP_TESTZ;
}
FlagsRead &= ~Info.Write();
// If we eliminated the instruction, we eliminate its read too. This
// check is required to ensure the pass converges locally in a single
// iteration.
if (!Eliminated) {
FlagsRead |= Info.Read();
}
}
}
// Iterate in reverse
if (CodeLast == CodeBegin) {
break;
}
--CodeLast;
}
// For the purposes of global propagation, the content of our progress doesn't
// matter -- only the difference in our final FlagsRead contributes to changes
// in the predecessors.
uint32_t OldFlagsRead = CFG.Get(Block)->Flags;
CFG.Get(Block)->Flags = FlagsRead;
return (OldFlagsRead != FlagsRead);
}
void DeadFlagCalculationEliminination::OptimizeParity(IREmitter* IREmit, IRListView& CurrentIR, ControlFlowGraph& CFG) {
// Mapping for flags inside this pass.
const uint8_t PARTIAL = 0;
const uint8_t FULL = 1;
// Initialize conservatively: all blocks need full parity. This initialization
// matters for proper handling of backedges.
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
CFG.Get(Block)->Flags = FULL;
}
for (auto [Block, BlockHeader] : CurrentIR.GetBlocks()) {
bool Full = false;
auto Predecessors = CFG.Get(Block)->Predecessors;
if (Predecessors.empty()) {
// Conservatively assume there was full parity before the start block
Full = true;
} else {
// If any predecessor needs full parity at the end, we need full parity.
for (auto Pred : Predecessors) {
Full |= (CFG.Get(Pred)->Flags == FULL);
}
}
for (auto [CodeNode, IROp] : CurrentIR.GetCode(Block)) {
if (IROp->Op == OP_STOREPF) {
auto Op = IROp->CW<IR::IROp_StorePF>();
auto Generator = CurrentIR.GetOp<IR::IROp_Header>(Op->Value);
// Determine if we only write 0/1 to the parity flag.
Full = true;
if (Generator->Op == OP_NZCVSELECT) {
auto C0 = CurrentIR.GetOp<IR::IROp_Header>(Generator->Args[0]);
auto C1 = CurrentIR.GetOp<IR::IROp_Header>(Generator->Args[1]);
if (C0->Op == C1->Op && C0->Op == OP_INLINECONSTANT) {
auto IC0 = CurrentIR.GetOp<IR::IROp_InlineConstant>(Generator->Args[0]);
auto IC1 = CurrentIR.GetOp<IR::IROp_InlineConstant>(Generator->Args[1]);
// We need the full 8 if the constant has upper bits set.
Full = (IC0->Constant | IC1->Constant) & ~1;
}
}
} else if (IROp->Op == OP_PARITY && !Full) {
// Eliminate parity calculations if it's only 1-bit.
auto Parity = IROp->C<IROp_Parity>();
Ref Value = CurrentIR.GetNode(Parity->Raw);
if (Parity->Invert) {
IREmit->SetWriteCursor(CodeNode);
Value = IREmit->_Xor(OpSize::i32Bit, Value, IREmit->_InlineConstant(1));
}
IREmit->ReplaceUsesWithAfter(CodeNode, Value, CurrentIR.at(CodeNode));
IREmit->Remove(CodeNode);
}
}
// Record our final state for our successors to read.
CFG.Get(Block)->Flags = Full ? FULL : PARTIAL;
}
}
void DeadFlagCalculationEliminination::Run(IREmitter* IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::DFE");
auto CurrentIR = IREmit->ViewIR();
fextl::deque<Ref> Worklist;
ControlFlowGraph CFG {.IR = CurrentIR};
// Gather blocks
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
// We model all flags as read at the end of the block, since this pass is
// presently purely local. Optimizing this requires global anslysis.
uint32_t FlagsRead = FLAG_ALL;
CFG.AddBlock(Worklist, BlockNode);
}
// Reverse iteration is not yet working with the iterators
auto BlockIROp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
// Gather CFG
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
auto CodeLast = CurrentIR.at(BlockHeader->C<IROp_CodeBlock>()->Last);
--CodeLast;
auto [ExitNode, ExitOp] = CodeLast();
if (ExitOp->Op == IR::OP_CONDJUMP) {
auto Op = ExitOp->CW<IR::IROp_CondJump>();
// We grab these nodes this way so we can iterate easily
auto CodeBegin = CurrentIR.at(BlockIROp->Begin);
auto CodeLast = CurrentIR.at(BlockIROp->Last);
// Iterate the block in reverse
while (1) {
auto [CodeNode, IROp] = CodeLast();
// Optimizing flags can cause earlier flag reads to become dead but dead
// flag reads should not impede optimiation of earlier dead flag writes.
// We must DCE as we go to ensure we converge in a single iteration.
if (!EliminateDeadCode(IREmit, CodeNode, IROp)) {
// Optimiation algorithm: For each flag written...
//
// If the flag has a later read (per FlagsRead), remove the flag from
// FlagsRead, since the reader is covered by this write.
//
// Else, there is no later read, so remove the flag write (if we can).
// This is the active part of the optimization.
//
// Then, add each flag read to FlagsRead.
//
// This order is important: instructions that read-modify-write flags
// (like adcs) first read flags, then write flags. Since we're iterating
// the block backwards, that means we handle the write first.
struct FlagInfo Info = Classify(IROp);
if (!Info.Trivial) {
bool Eliminated = false;
if ((FlagsRead & Info.Write) == 0) {
if ((Info.CanEliminate || Info.CanReplace) && CodeNode->GetUses() == 0) {
IREmit->Remove(CodeNode);
Eliminated = true;
} else if (Info.CanReplace) {
IROp->Op = Info.Replacement;
}
} else {
FlagsRead &= ~Info.Write;
if (Info.CanReplaceWrite && CodeNode->GetUses() == 0) {
IROp->Op = Info.ReplacementNoWrite;
}
}
// If we eliminated the instruction, we eliminate its read too. This
// check is required to ensure the pass converges locally in a single
// iteration.
if (!Eliminated) {
FlagsRead |= Info.Read;
}
}
}
// Iterate in reverse
if (CodeLast == CodeBegin) {
break;
}
--CodeLast;
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(Op->TrueBlock));
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(Op->FalseBlock));
} else if (ExitOp->Op == IR::OP_JUMP) {
CFG.RecordEdge(BlockNode, CurrentIR.GetNode(ExitOp->Args[0]));
}
}
// After processing a block, if we made progress, we must process its
// predecessors to propagate globally. A block will be reprocessed only if
// there is a loop backedge.
for (; !Worklist.empty(); Worklist.pop_back()) {
auto Block = Worklist.back();
auto Info = CFG.Get(Block);
Info->InWorklist = false;
if (ProcessBlock(IREmit, CurrentIR, Block, CFG)) {
for (auto Pred : Info->Predecessors) {
CFG.AddWorklist(Worklist, Pred);
}
}
}
// Fold compares into branches now that we're otherwise optimized. This needs
// to run after eliminating carries etc and it needs the global flag metadata.
// But it only needs to run once, we don't do it in the loop.
for (auto [Block, _] : CurrentIR.GetBlocks()) {
// Grab the jump
auto BlockIROp = CurrentIR.GetOp<IR::IROp_CodeBlock>(Block);
auto CodeLast = CurrentIR.at(BlockIROp->Last);
--CodeLast;
auto [ExitNode, ExitOp] = CodeLast();
if (ExitOp->Op == IR::OP_CONDJUMP) {
auto Op = ExitOp->CW<IR::IROp_CondJump>();
uint32_t FlagsOut = CFG.Get(Op->TrueBlock)->Flags | CFG.Get(Op->FalseBlock)->Flags;
if ((FlagsOut & FLAG_NZCV) == 0 && Op->FromNZCV) {
FoldBranch(IREmit, CurrentIR, Op, ExitNode);
}
}
}
if (CurrentIR.GetHeader()->ReadsParity) {
OptimizeParity(IREmit, CurrentIR, CFG);
}
}
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination() {
@@ -228,11 +228,14 @@ private:
};
Ref DecodeSRANode(const IROp_Header* IROp, Ref Node) {
if (IROp->Op == OP_LOADREGISTER) {
if (IROp->Op == OP_LOADREGISTER || IROp->Op == OP_LOADPF || IROp->Op == OP_LOADAF) {
return Node;
} else if (IROp->Op == OP_STOREREGISTER) {
const IROp_StoreRegister* Op = IROp->C<IR::IROp_StoreRegister>();
return IR->GetNode(Op->Value);
} else if (IROp->Op == OP_STOREPF || IROp->Op == OP_STOREAF) {
const IROp_StorePF* Op = IROp->C<IR::IROp_StorePF>();
return IR->GetNode(Op->Value);
}
return nullptr;
@@ -242,6 +245,8 @@ private:
RegisterClassType Class;
uint8_t Reg;
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
if (IROp->Op == OP_LOADREGISTER) {
const IROp_LoadRegister* Op = IROp->C<IR::IROp_LoadRegister>();
@@ -253,17 +258,15 @@ private:
Class = Op->Class;
Reg = Op->Reg;
} else if (IROp->Op == OP_LOADPF || IROp->Op == OP_STOREPF) {
return PhysicalRegister {GPRFixedClass, FlagOffset};
} else if (IROp->Op == OP_LOADAF || IROp->Op == OP_STOREAF) {
return PhysicalRegister {GPRFixedClass, (uint8_t)(FlagOffset + 1)};
}
LOGMAN_THROW_A_FMT(Class == GPRClass || Class == FPRClass, "SRA classes");
uint8_t FlagOffset = Classes[GPRFixedClass.Val].Count - 2;
if (Class == FPRClass) {
return PhysicalRegister {FPRFixedClass, Reg};
} else if (Reg == Core::CPUState::PF_AS_GREG) {
return PhysicalRegister {GPRFixedClass, FlagOffset};
} else if (Reg == Core::CPUState::AF_AS_GREG) {
return PhysicalRegister {GPRFixedClass, (uint8_t)(FlagOffset + 1)};
} else {
return PhysicalRegister {GPRFixedClass, Reg};
}
@@ -2891,8 +2891,8 @@
"fcmp s16, s17",
"mov w27, #0x0",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"vucomisd xmm0, xmm1": {
@@ -2904,8 +2904,8 @@
"fcmp d16, d17",
"mov w27, #0x0",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"vcomiss xmm0, xmm1": {
@@ -2917,8 +2917,8 @@
"fcmp s16, s17",
"mov w27, #0x0",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"vcomisd xmm0, xmm1": {
@@ -2930,8 +2930,8 @@
"fcmp d16, d17",
"mov w27, #0x0",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"vaddps xmm0, xmm1, xmm2": {
@@ -203,27 +203,21 @@
},
"Variable rotate-through-carry dead": {
"x86InstructionCount": 2,
"ExpectedInstructionCount": 17,
"ExpectedInstructionCount": 11,
"x86Insts": [
"rcr rax, cl",
"test rax, rdx"
],
"ExpectedArm64ASM": [
"and x20, x7, #0x3f",
"cbz x20, #+0x38",
"cbz x20, #+0x20",
"lsr x20, x4, x7",
"cset w21, lo",
"neg x22, x7",
"lsl x23, x4, x22",
"orr x20, x20, x23, lsl #1",
"sub x23, x7, #0x1 (1)",
"lsr x23, x4, x23",
"eor x23, x23, #0x1",
"rmif x23, #63, #nzCv",
"lsl x21, x21, x22",
"orr x4, x20, x21",
"eor x20, x4, x4, lsr #1",
"rmif x20, #62, #nzcV",
"ands x26, x4, x5",
"cfinv"
]
@@ -290,16 +284,85 @@
"ExpectedArm64ASM": [
"and w26, w4, w6",
"mov x4, x26",
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"and x20, x20, #0x1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"and w20, w20, #0x1",
"bfxil x7, x20, #0, #8",
"cmn wzr, w7, lsl #24",
"cfinv",
"mov x26, x7"
]
},
"UCOMISS use only PF": {
"x86InstructionCount": 3,
"ExpectedInstructionCount": 4,
"x86Insts": [
"ucomiss xmm0, xmm1",
"setnp cl",
"test rax, rax"
],
"ExpectedArm64ASM": [
"fcmp s16, s17",
"cset w26, vc",
"bfxil x7, x26, #0, #8",
"subs x26, x4, #0x0 (0)"
]
},
"Test use only zero - self 16-bit": {
"x86InstructionCount": 3,
"ExpectedInstructionCount": 6,
"x86Insts": [
"test ax, ax",
"setz al",
"test cl, cl"
],
"ExpectedArm64ASM": [
"tst w4, #0xffff",
"cset x20, eq",
"bfxil x4, x20, #0, #8",
"cmn wzr, w7, lsl #24",
"cfinv",
"mov x26, x7"
]
},
"Test use only zero - non constant 16-bit": {
"x86InstructionCount": 3,
"ExpectedInstructionCount": 7,
"x86Insts": [
"test ax, bx",
"setz al",
"test cl, cl"
],
"ExpectedArm64ASM": [
"and w0, w4, w6",
"tst w0, #0xffff",
"cset x20, eq",
"bfxil x4, x20, #0, #8",
"cmn wzr, w7, lsl #24",
"cfinv",
"mov x26, x7"
]
},
"Test use only zero - constant 8-bit": {
"x86InstructionCount": 3,
"ExpectedInstructionCount": 8,
"x86Insts": [
"test al, 137",
"setnz al",
"test cl, cl"
],
"ExpectedArm64ASM": [
"mov w20, #0x89",
"and w0, w4, w20",
"tst w0, #0xff",
"cset x20, ne",
"bfxil x4, x20, #0, #8",
"cmn wzr, w7, lsl #24",
"cfinv",
"mov x26, x7"
]
},
"Dead cmpxchg flags": {
"x86InstructionCount": 2,
"ExpectedInstructionCount": 10,
@@ -1728,9 +1728,9 @@
"orr x20, x20, x21, lsl #20",
"ldrb w21, [x28, #997]",
"orr x20, x20, x21, lsl #21",
"eor w21, w26, w26, lsr #4",
"eor w21, w21, w21, lsr #2",
"eor w21, w21, w21, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w21, w0, w0, lsr #1",
"orr x21, x21, #0xfffffffffffffffe",
"orn x20, x20, x21, ror #62",
"mrs x21, nzcv",
@@ -1773,9 +1773,9 @@
"orr x20, x20, x21, lsl #20",
"ldrb w21, [x28, #997]",
"orr x20, x20, x21, lsl #21",
"eor w21, w26, w26, lsr #4",
"eor w21, w21, w21, lsr #2",
"eor w21, w21, w21, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w21, w0, w0, lsr #1",
"orr x21, x21, #0xfffffffffffffffe",
"orn x20, x20, x21, ror #62",
"mrs x21, nzcv",
@@ -1847,9 +1847,9 @@
"eor w21, w27, w26",
"ubfx w21, w21, #4, #1",
"orr x20, x20, x21, lsl #4",
"eor w21, w26, w26, lsr #4",
"eor w21, w21, w21, lsr #2",
"eor w21, w21, w21, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w21, w0, w0, lsr #1",
"orr x21, x21, #0xfffffffffffffffe",
"orn x20, x20, x21, ror #62",
"mrs x21, nzcv",
@@ -257,9 +257,9 @@
"ExpectedInstructionCount": 8,
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"mrs x21, nzcv",
"tst w20, #0x1",
"csel w20, w6, w4, ne",
@@ -271,9 +271,9 @@
"ExpectedInstructionCount": 7,
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"mrs x21, nzcv",
"tst w20, #0x1",
"csel w4, w6, w4, ne",
@@ -284,9 +284,9 @@
"ExpectedInstructionCount": 7,
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"mrs x21, nzcv",
"tst w20, #0x1",
"csel x4, x6, x4, ne",
@@ -297,9 +297,9 @@
"ExpectedInstructionCount": 8,
"Comment": "0x0f 0x4b",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"mrs x21, nzcv",
"tst w20, #0x1",
"csel w20, w6, w4, ne",
@@ -311,9 +311,9 @@
"ExpectedInstructionCount": 7,
"Comment": "0x0f 0x4b",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"mrs x21, nzcv",
"tst w20, #0x1",
"csel w4, w6, w4, ne",
@@ -324,9 +324,9 @@
"ExpectedInstructionCount": 7,
"Comment": "0x0f 0x4b",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"mrs x21, nzcv",
"tst w20, #0x1",
"csel x4, x6, x4, ne",
@@ -505,10 +505,10 @@
"ExpectedInstructionCount": 5,
"Comment": "0x0f 0x9a",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"and x20, x20, #0x1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"and w20, w20, #0x1",
"bfxil x4, x20, #0, #8"
]
},
@@ -516,10 +516,10 @@
"ExpectedInstructionCount": 5,
"Comment": "0x0f 0x9b",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"and x20, x20, #0x1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"and w20, w20, #0x1",
"bfxil x4, x20, #0, #8"
]
},
+48 -48
View File
@@ -6439,9 +6439,9 @@
"0xda 11b 0xd8 /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6460,9 +6460,9 @@
"0xda 11b 0xd9 /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6483,9 +6483,9 @@
"0xda 11b 0xda /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6506,9 +6506,9 @@
"0xda 11b 0xdb /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6529,9 +6529,9 @@
"0xda 11b 0xdc /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6552,9 +6552,9 @@
"0xda 11b 0xdd /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6575,9 +6575,9 @@
"0xda 11b 0xde /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6598,9 +6598,9 @@
"0xda 11b 0xdf /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7377,9 +7377,9 @@
"0xdb 11b 0xd8 /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7398,9 +7398,9 @@
"0xdb 11b 0xd9 /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7421,9 +7421,9 @@
"0xdb 11b 0xda /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7444,9 +7444,9 @@
"0xdb 11b 0xdb /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7467,9 +7467,9 @@
"0xdb 11b 0xdc /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7490,9 +7490,9 @@
"0xdb 11b 0xdd /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7513,9 +7513,9 @@
"0xdb 11b 0xde /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7536,9 +7536,9 @@
"0xdb 11b 0xdf /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
+48 -48
View File
@@ -3790,9 +3790,9 @@
"0xda 11b 0xd8 /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3811,9 +3811,9 @@
"0xda 11b 0xd9 /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3834,9 +3834,9 @@
"0xda 11b 0xda /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3857,9 +3857,9 @@
"0xda 11b 0xdb /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3880,9 +3880,9 @@
"0xda 11b 0xdc /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3903,9 +3903,9 @@
"0xda 11b 0xdd /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3926,9 +3926,9 @@
"0xda 11b 0xde /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3949,9 +3949,9 @@
"0xda 11b 0xdf /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4658,9 +4658,9 @@
"0xdb 11b 0xd8 /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4679,9 +4679,9 @@
"0xdb 11b 0xd9 /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4702,9 +4702,9 @@
"0xdb 11b 0xda /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4725,9 +4725,9 @@
"0xdb 11b 0xdb /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4748,9 +4748,9 @@
"0xdb 11b 0xdc /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4771,9 +4771,9 @@
"0xdb 11b 0xdd /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4794,9 +4794,9 @@
"0xdb 11b 0xde /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4817,9 +4817,9 @@
"0xdb 11b 0xdf /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
+9 -9
View File
@@ -2625,9 +2625,9 @@
"orr x20, x20, x21, lsl #20",
"ldrb w21, [x28, #997]",
"orr x20, x20, x21, lsl #21",
"eor w21, w26, w26, lsr #4",
"eor w21, w21, w21, lsr #2",
"eor w21, w21, w21, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w21, w0, w0, lsr #1",
"orr x21, x21, #0xfffffffffffffffe",
"orn x20, x20, x21, ror #62",
"mrs x21, nzcv",
@@ -2670,9 +2670,9 @@
"orr x20, x20, x21, lsl #20",
"ldrb w21, [x28, #997]",
"orr x20, x20, x21, lsl #21",
"eor w21, w26, w26, lsr #4",
"eor w21, w21, w21, lsr #2",
"eor w21, w21, w21, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w21, w0, w0, lsr #1",
"orr x21, x21, #0xfffffffffffffffe",
"orn x20, x20, x21, ror #62",
"mrs x21, nzcv",
@@ -2754,9 +2754,9 @@
"eor w21, w27, w26",
"ubfx w21, w21, #4, #1",
"orr x20, x20, x21, lsl #4",
"eor w21, w26, w26, lsr #4",
"eor w21, w21, w21, lsr #2",
"eor w21, w21, w21, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w21, w0, w0, lsr #1",
"orr x21, x21, #0xfffffffffffffffe",
"orn x20, x20, x21, ror #62",
"mrs x21, nzcv",
+30 -30
View File
@@ -210,8 +210,8 @@
"fcmp s16, s17",
"mov w27, #0x0",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"comiss xmm0, xmm1": {
@@ -221,8 +221,8 @@
"fcmp s16, s17",
"mov w27, #0x0",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"rdtsc": {
@@ -459,9 +459,9 @@
"ExpectedInstructionCount": 8,
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"mrs x21, nzcv",
"tst w20, #0x1",
"csel w20, w6, w4, ne",
@@ -473,9 +473,9 @@
"ExpectedInstructionCount": 7,
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"mrs x21, nzcv",
"tst w20, #0x1",
"csel w4, w6, w4, ne",
@@ -486,9 +486,9 @@
"ExpectedInstructionCount": 7,
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"mrs x21, nzcv",
"tst w20, #0x1",
"csel x4, x6, x4, ne",
@@ -499,9 +499,9 @@
"ExpectedInstructionCount": 8,
"Comment": "0x0f 0x4b",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"mrs x21, nzcv",
"tst w20, #0x1",
"csel w20, w6, w4, ne",
@@ -513,9 +513,9 @@
"ExpectedInstructionCount": 7,
"Comment": "0x0f 0x4b",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"mrs x21, nzcv",
"tst w20, #0x1",
"csel w4, w6, w4, ne",
@@ -526,9 +526,9 @@
"ExpectedInstructionCount": 7,
"Comment": "0x0f 0x4b",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"mrs x21, nzcv",
"tst w20, #0x1",
"csel x4, x6, x4, ne",
@@ -1222,10 +1222,10 @@
"ExpectedInstructionCount": 5,
"Comment": "0x0f 0x9a",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"and x20, x20, #0x1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"and w20, w20, #0x1",
"bfxil x4, x20, #0, #8"
]
},
@@ -1233,10 +1233,10 @@
"ExpectedInstructionCount": 5,
"Comment": "0x0f 0x9b",
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"and x20, x20, #0x1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"and w20, w20, #0x1",
"bfxil x4, x20, #0, #8"
]
},
@@ -148,8 +148,8 @@
"fcmp d16, d17",
"mov w27, #0x0",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"comisd xmm0, xmm1": {
@@ -159,8 +159,8 @@
"fcmp d16, d17",
"mov w27, #0x0",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"movmskpd eax, xmm0": {
+8 -8
View File
@@ -3361,8 +3361,8 @@
"fcmp s16, s17",
"mov w27, #0x0",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"vucomisd xmm0, xmm1": {
@@ -3374,8 +3374,8 @@
"fcmp d16, d17",
"mov w27, #0x0",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"vcomiss xmm0, xmm1": {
@@ -3387,8 +3387,8 @@
"fcmp s16, s17",
"mov w27, #0x0",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"vcomisd xmm0, xmm1": {
@@ -3400,8 +3400,8 @@
"fcmp d16, d17",
"mov w27, #0x0",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"vaddps xmm0, xmm1, xmm2": {
+48 -48
View File
@@ -6438,9 +6438,9 @@
"0xda 11b 0xd8 /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6459,9 +6459,9 @@
"0xda 11b 0xd9 /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6482,9 +6482,9 @@
"0xda 11b 0xda /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6505,9 +6505,9 @@
"0xda 11b 0xdb /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6528,9 +6528,9 @@
"0xda 11b 0xdc /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6551,9 +6551,9 @@
"0xda 11b 0xdd /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6574,9 +6574,9 @@
"0xda 11b 0xde /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -6597,9 +6597,9 @@
"0xda 11b 0xdf /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7376,9 +7376,9 @@
"0xdb 11b 0xd8 /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7397,9 +7397,9 @@
"0xdb 11b 0xd9 /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7420,9 +7420,9 @@
"0xdb 11b 0xda /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7443,9 +7443,9 @@
"0xdb 11b 0xdb /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7466,9 +7466,9 @@
"0xdb 11b 0xdc /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7489,9 +7489,9 @@
"0xdb 11b 0xdd /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7512,9 +7512,9 @@
"0xdb 11b 0xde /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -7535,9 +7535,9 @@
"0xdb 11b 0xdf /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
+112 -112
View File
@@ -3810,9 +3810,9 @@
"0xda 11b 0xd8 /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3831,9 +3831,9 @@
"0xda 11b 0xd9 /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3854,9 +3854,9 @@
"0xda 11b 0xda /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3877,9 +3877,9 @@
"0xda 11b 0xdb /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3900,9 +3900,9 @@
"0xda 11b 0xdc /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3923,9 +3923,9 @@
"0xda 11b 0xdd /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3946,9 +3946,9 @@
"0xda 11b 0xde /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -3969,9 +3969,9 @@
"0xda 11b 0xdf /1"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eon w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4679,9 +4679,9 @@
"0xdb 11b 0xd8 /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4700,9 +4700,9 @@
"0xdb 11b 0xd9 /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4723,9 +4723,9 @@
"0xdb 11b 0xda /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4746,9 +4746,9 @@
"0xdb 11b 0xdb /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4769,9 +4769,9 @@
"0xdb 11b 0xdc /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4792,9 +4792,9 @@
"0xdb 11b 0xdd /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4815,9 +4815,9 @@
"0xdb 11b 0xde /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4838,9 +4838,9 @@
"0xdb 11b 0xdf /3"
],
"ExpectedArm64ASM": [
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eor w20, w20, w20, lsr #1",
"eor w0, w26, w26, lsr #4",
"eor w0, w0, w0, lsr #2",
"eor w20, w0, w0, lsr #1",
"sbfx x20, x20, #0, #1",
"dup v2.2d, x20",
"ldrb w20, [x28, #1019]",
@@ -4899,8 +4899,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fucomi st0, st1": {
@@ -4918,8 +4918,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fucomi st0, st2": {
@@ -4937,8 +4937,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fucomi st0, st3": {
@@ -4956,8 +4956,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fucomi st0, st4": {
@@ -4975,8 +4975,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fucomi st0, st5": {
@@ -4994,8 +4994,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fucomi st0, st6": {
@@ -5013,8 +5013,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fucomi st0, st7": {
@@ -5032,8 +5032,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fcomi st0, st0": {
@@ -5049,8 +5049,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fcomi st0, st1": {
@@ -5068,8 +5068,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fcomi st0, st2": {
@@ -5087,8 +5087,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fcomi st0, st3": {
@@ -5106,8 +5106,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fcomi st0, st4": {
@@ -5125,8 +5125,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fcomi st0, st5": {
@@ -5144,8 +5144,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fcomi st0, st6": {
@@ -5163,8 +5163,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fcomi st0, st7": {
@@ -5182,8 +5182,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x20, eq",
"ccmn x26, x20, #nzCv, le"
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le"
]
},
"fadd qword [rax]": {
@@ -9693,8 +9693,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -9720,8 +9720,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -9747,8 +9747,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -9774,8 +9774,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -9801,8 +9801,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -9828,8 +9828,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -9855,8 +9855,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -9882,8 +9882,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -9907,8 +9907,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -9934,8 +9934,8 @@
"ldr d3, [x0, #1040]",
"fcmp d3, d2",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -9961,8 +9961,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -9988,8 +9988,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -10015,8 +10015,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -10042,8 +10042,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -10069,8 +10069,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
@@ -10096,8 +10096,8 @@
"fcmp d3, d2",
"mov w21, #0x1",
"cset w26, vc",
"csetm x22, eq",
"ccmn x26, x22, #nzCv, le",
"csetm x0, eq",
"ccmn x26, x0, #nzCv, le",
"ldrb w22, [x28, #1298]",
"lsl w21, w21, w20",
"bic w21, w22, w21",