Merge pull request #3808 from alyssarosenzweig/rclse/3

Try to delete RCLSE again
This commit is contained in:
Ryan Houdek authored and GitHub committed 2024-07-12 20:38:06 -07:00
commit d79b7fcc49
46 files changed
+33563 -39646

No files matched your search

-1
View File
@@ -138,7 +138,6 @@ set (SRCS
Interface/IR/IREmitter.cpp
Interface/IR/PassManager.cpp
Interface/IR/Passes/ConstProp.cpp
Interface/IR/Passes/DeadContextStoreElimination.cpp
Interface/IR/Passes/IRDumperPass.cpp
Interface/IR/Passes/IRValidation.cpp
Interface/IR/Passes/RAValidation.cpp
+14
View File
@@ -614,6 +614,20 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
DecodedInfo = &Block.DecodedInstructions[i];
bool IsLocked = DecodedInfo->Flags & FEXCore::X86Tables::DecodeFlags::FLAG_LOCK;
// Do a partial register cache flush before every instruction. This
// prevents cross-instruction static register caching, while allowing
// context load/stores to be optimized within a block. Theoretically,
// this flush is not required for correctness, all mandatory flushes are
// included in instruction-specific handlers. Instead, this is a blunt
// heuristic to make the register cache less aggressive, as the current
// RA generates bad code in common cases with tied registers otherwise.
//
// However, it makes our exception handling behaviour more predictable.
// It is potentially correctness bearing in that sense, but that is a
// side effect here and (if that behaviour is required) we should handle
// that more explicitly later.
Thread->OpDispatcher->FlushRegisterCache(true);
if (ExtendedDebugInfo || Thread->OpDispatcher->CanHaveSideEffects(TableInfo, DecodedInfo)) {
Thread->OpDispatcher->_GuestOpcode(Block.Entry + BlockInstructionsLength - GuestRIP);
}
@@ -625,7 +625,7 @@ DEF_OP(ShiftFlags) {
// Set the output outside the branch to avoid needing an extra leg of the
// branch. We specifically do not hardcode the PF register anywhere (relying
// on a tied SRA register instead) to avoid fighting with RA/RCLSE.
// on a tied SRA register instead) to avoid fighting with RA.
if (PFTemp != PFInput) {
mov(ARMEmitter::Size::i64Bit, PFTemp, PFInput);
}
@@ -112,7 +112,7 @@ void OpDispatchBuilder::SyscallOp(OpcodeArgs, bool IsSyscallInst) {
StoreGPRRegister(X86State::REG_RCX, RIPAfterInst, 8);
}
CalculateDeferredFlags();
FlushRegisterCache();
auto SyscallOp = _Syscall(Arguments[0], Arguments[1], Arguments[2], Arguments[3], Arguments[4], Arguments[5], Arguments[6], DefaultSyscallFlags);
if (OSABI != FEXCore::HLE::SyscallOSABI::OS_HANGOVER &&
@@ -394,7 +394,7 @@ void OpDispatchBuilder::PUSHOp(OpcodeArgs) {
// Store the new stack pointer
StoreGPRRegister(X86State::REG_RSP, NewSP);
CalculateDeferredFlags();
FlushRegisterCache();
}
void OpDispatchBuilder::PUSHREGOp(OpcodeArgs) {
@@ -407,7 +407,7 @@ void OpDispatchBuilder::PUSHREGOp(OpcodeArgs) {
auto NewSP = _Push(GPRSize, Size, Src, OldSP);
// Store the new stack pointer
StoreGPRRegister(X86State::REG_RSP, NewSP);
CalculateDeferredFlags();
FlushRegisterCache();
}
void OpDispatchBuilder::PUSHAOp(OpcodeArgs) {
@@ -457,7 +457,7 @@ void OpDispatchBuilder::PUSHAOp(OpcodeArgs) {
// Store the new stack pointer
StoreGPRRegister(X86State::REG_RSP, NewSP, 4);
CalculateDeferredFlags();
FlushRegisterCache();
}
template<uint32_t SegmentReg>
@@ -842,7 +842,7 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
Target &= 0xFFFFFFFFU;
}
CalculateDeferredFlags();
FlushRegisterCache();
auto TrueBlock = JumpTargets.find(Target);
auto FalseBlock = JumpTargets.find(Op->PC + Op->InstSize);
@@ -3386,8 +3386,8 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
auto Src2 = _LoadMem(GPRClass, Size, Dest_RSI, Size);
// We'll calculate PF/AF after the loop, so use them as temporaries here.
_StoreRegister(Src1, Core::CPUState::PF_AS_GREG, GPRClass, CTX->GetGPRSize());
_StoreRegister(Src2, Core::CPUState::AF_AS_GREG, GPRClass, CTX->GetGPRSize());
StoreRegister(Core::CPUState::PF_AS_GREG, false, Src1);
StoreRegister(Core::CPUState::AF_AS_GREG, false, Src2);
Ref TailCounter = LoadGPRRegister(X86State::REG_RCX);
@@ -3426,8 +3426,8 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
// Make sure to start a new block after ending this one
{
// Grab the sources from the last iteration so we can set flags.
auto Src1 = _LoadRegister(Core::CPUState::PF_AS_GREG, GPRClass, CTX->GetGPRSize());
auto Src2 = _LoadRegister(Core::CPUState::AF_AS_GREG, GPRClass, CTX->GetGPRSize());
auto Src1 = LoadGPR(Core::CPUState::PF_AS_GREG);
auto Src2 = LoadGPR(Core::CPUState::AF_AS_GREG);
CalculateFlags_SUB(GetSrcSize(Op), Src2, Src1);
}
auto Jump_ = Jump();
@@ -4014,7 +4014,7 @@ void OpDispatchBuilder::BeginFunction(uint64_t RIP, const fextl::vector<FEXCore:
void OpDispatchBuilder::Finalize() {
// This usually doesn't emit any IR but in the case of hitting the block instruction limit it will
CalculateDeferredFlags();
FlushRegisterCache();
const uint8_t GPRSize = CTX->GetGPRSize();
// Node 0 is invalid node
@@ -4381,7 +4381,9 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
const auto highIndex = Operand.Data.GPR.HighBits ? 1 : 0;
if (gpr >= FEXCore::X86State::REG_MM_0) {
A.Base = _LoadContext(OpSize, FPRClass, offsetof(FEXCore::Core::CPUState, mm[gpr - FEXCore::X86State::REG_MM_0]));
LOGMAN_THROW_A_FMT(OpSize == 8, "full");
A.Base = LoadContext(8, MM0Index + gpr - FEXCore::X86State::REG_MM_0);
} else if (gpr >= FEXCore::X86State::REG_XMM_0) {
const auto gprIndex = gpr - X86State::REG_XMM_0;
@@ -4418,7 +4420,7 @@ Ref OpDispatchBuilder::LoadGPRRegister(uint32_t GPR, int8_t Size, uint8_t Offset
if (Size == -1) {
Size = GPRSize;
}
Ref Reg = _LoadRegister(GPR, GPRClass, GPRSize);
Ref Reg = LoadGPR(GPR);
if ((!AllowUpperGarbage && (Size != GPRSize)) || Offset != 0) {
// Extract the subregister if requested.
@@ -4432,10 +4434,6 @@ Ref OpDispatchBuilder::LoadGPRRegister(uint32_t GPR, int8_t Size, uint8_t Offset
return Reg;
}
Ref OpDispatchBuilder::LoadXMMRegister(uint32_t XMM) {
return _LoadRegister(XMM, FPRClass, GetGuestVectorLength());
}
void OpDispatchBuilder::StoreGPRRegister(uint32_t GPR, const Ref Src, int8_t Size, uint8_t Offset) {
const uint8_t GPRSize = CTX->GetGPRSize();
if (Size == -1) {
@@ -4445,15 +4443,14 @@ void OpDispatchBuilder::StoreGPRRegister(uint32_t GPR, const Ref Src, int8_t Siz
Ref Reg = Src;
if (Size != GPRSize || Offset != 0) {
// Need to do an insert if not automatic size or zero offset.
Reg = LoadGPRRegister(GPR);
Reg = _Bfi(IR::SizeToOpSize(GPRSize), Size * 8, Offset, Reg, Src);
Reg = _Bfi(IR::SizeToOpSize(GPRSize), Size * 8, Offset, LoadGPRRegister(GPR), Src);
}
_StoreRegister(Reg, GPR, GPRClass, GPRSize);
StoreRegister(GPR, false, Reg);
}
void OpDispatchBuilder::StoreXMMRegister(uint32_t XMM, const Ref Src) {
_StoreRegister(Src, XMM, FPRClass, GetGuestVectorLength());
StoreRegister(XMM, true, Src);
}
Ref OpDispatchBuilder::LoadSource(RegisterClassType Class, const X86Tables::DecodedOp& Op, const X86Tables::DecodedOperand& Operand,
@@ -4472,7 +4469,12 @@ void OpDispatchBuilder::StoreResult_WithOpSize(FEXCore::IR::RegisterClassType Cl
const auto gpr = Operand.Data.GPR.GPR;
if (gpr >= FEXCore::X86State::REG_MM_0) {
_StoreContext(OpSize, Class, Src, offsetof(FEXCore::Core::CPUState, mm[gpr - FEXCore::X86State::REG_MM_0]));
LOGMAN_THROW_A_FMT(OpSize == 8, "full");
LOGMAN_THROW_A_FMT(Class == FPRClass, "MMX is floaty");
// Partial store into bottom 64-bits, leave the upper bits unaffected.
// XXX: We actually should set the upper bits to all-1s?
StoreContextPartial(MM0Index + gpr - FEXCore::X86State::REG_MM_0, Src);
} else if (gpr >= FEXCore::X86State::REG_XMM_0) {
const auto gprIndex = gpr - X86State::REG_XMM_0;
const auto VectorSize = GetGuestVectorLength();
@@ -4570,6 +4572,8 @@ void OpDispatchBuilder::ResetWorkingList() {
DecodeFailure = false;
ShouldDump = false;
CurrentCodeBlock = nullptr;
RegCache.Written = 0;
RegCache.Cached = 0;
}
void OpDispatchBuilder::UnhandledOp(OpcodeArgs) {
@@ -4718,7 +4722,7 @@ void OpDispatchBuilder::INTOp(OpcodeArgs) {
}
// Calculate flags early.
CalculateDeferredFlags();
FlushRegisterCache();
const uint8_t GPRSize = CTX->GetGPRSize();
+213 -21
View File
@@ -16,6 +16,7 @@
#include <FEXCore/fextl/map.h>
#include <FEXCore/fextl/vector.h>
#include <bit>
#include <cstdint>
#include <fmt/format.h>
#include <stddef.h>
@@ -115,7 +116,7 @@ public:
// Changes get stored out by CalculateDeferredFlags.
CachedNZCV = nullptr;
PossiblySetNZCVBits = ~0U;
CalculateDeferredFlags();
FlushRegisterCache();
// New block needs to reset segment telemetry.
SegmentsNeedReadCheck = ~0U;
@@ -125,28 +126,28 @@ public:
}
IRPair<IROp_Jump> Jump() {
CalculateDeferredFlags();
FlushRegisterCache();
return _Jump();
}
IRPair<IROp_Jump> Jump(Ref _TargetBlock) {
CalculateDeferredFlags();
FlushRegisterCache();
return _Jump(_TargetBlock);
}
IRPair<IROp_CondJump>
CondJump(Ref _Cmp1, Ref _Cmp2, Ref _TrueBlock, Ref _FalseBlock, CondClassType _Cond = {COND_NEQ}, uint8_t _CompareSize = 0) {
CalculateDeferredFlags();
FlushRegisterCache();
return _CondJump(_Cmp1, _Cmp2, _TrueBlock, _FalseBlock, _Cond, _CompareSize);
}
IRPair<IROp_CondJump> CondJump(Ref ssa0, CondClassType cond = {COND_NEQ}) {
CalculateDeferredFlags();
FlushRegisterCache();
return _CondJump(ssa0, cond);
}
IRPair<IROp_CondJump> CondJump(Ref ssa0, Ref ssa1, Ref ssa2, CondClassType cond = {COND_NEQ}) {
CalculateDeferredFlags();
FlushRegisterCache();
return _CondJump(ssa0, ssa1, ssa2, cond);
}
IRPair<IROp_CondJump> CondJumpNZCV(CondClassType Cond) {
CalculateDeferredFlags();
FlushRegisterCache();
// The jump will ignore the sources, so it doesn't matter what we put here.
// Put an inline constant so RA+codegen will ignore altogether.
@@ -154,15 +155,15 @@ public:
return _CondJump(Placeholder, Placeholder, InvalidNode, InvalidNode, Cond, 0, true);
}
IRPair<IROp_ExitFunction> ExitFunction(Ref NewRIP) {
CalculateDeferredFlags();
FlushRegisterCache();
return _ExitFunction(NewRIP);
}
IRPair<IROp_Break> Break(BreakDefinition Reason) {
CalculateDeferredFlags();
FlushRegisterCache();
return _Break(Reason);
}
IRPair<IROp_Thunk> Thunk(Ref ArgPtr, SHA256Sum ThunkNameHash) {
CalculateDeferredFlags();
FlushRegisterCache();
return _Thunk(ArgPtr, ThunkNameHash);
}
@@ -1225,6 +1226,53 @@ public:
}
}
void FlushRegisterCache(bool SRAOnly = false) {
CalculateDeferredFlags();
const uint8_t GPRSize = CTX->GetGPRSize();
const auto VectorSize = GetGuestVectorLength();
// Write backwards. This is a heuristic to improve coalescing, since we
// often copy from (low) fixed GPRs to (high) PF/AF for celebrity
// instructions like "add rax, 1". This hack will go away with clauses.
uint64_t Bits = RegCache.Written;
// We have an SRA only mode that exists as a hack to make register caching
// less aggressive. We should get rid of this once RA can take it.
uint64_t Mask = ~0ULL;
if (SRAOnly) {
const uint64_t GPRMask = ((1ull << (AFIndex - GPR0Index + 1)) - 1) << GPR0Index;
const uint64_t FPRMask = ((1ull << (FPR15Index - FPR0Index + 1)) - 1) << FPR0Index;
Mask &= (GPRMask | FPRMask);
Bits &= Mask;
}
while (Bits != 0) {
uint32_t Index = 63 - std::countl_zero(Bits);
Ref Value = RegCache.Value[Index];
if (Index >= GPR0Index && Index <= AFIndex) {
_StoreRegister(Value, Index - GPR0Index, GPRClass, GPRSize);
} else if (Index >= FPR0Index && Index <= FPR15Index) {
_StoreRegister(Value, Index - FPR0Index, FPRClass, VectorSize);
} else if (Index == DFIndex) {
_StoreFlag(Value, X86State::RFLAG_DF_RAW_LOC);
} else {
bool Partial = RegCache.Partial & (1ull << Index);
unsigned Size = Partial ? 8 : CacheIndexToSize(Index);
_StoreContext(Size, CacheIndexClass(Index), Value, CacheIndexToContextOffset(Index));
}
Bits &= ~(1ull << Index);
}
RegCache.Written &= ~Mask;
RegCache.Cached &= ~Mask;
RegCache.Partial &= ~Mask;
}
protected:
void SaveNZCV(IROps Op = OP_DUMMY) override {
/* Some opcodes are conservatively marked as clobbering flags, but in fact
@@ -1448,7 +1496,6 @@ private:
void UpdatePrefixFromSegment(Ref Segment, uint32_t SegmentReg);
Ref LoadGPRRegister(uint32_t GPR, int8_t Size = -1, uint8_t Offset = 0, bool AllowUpperGarbage = false);
Ref LoadXMMRegister(uint32_t XMM);
void StoreGPRRegister(uint32_t GPR, const Ref Src, int8_t Size = -1, uint8_t Offset = 0);
void StoreXMMRegister(uint32_t XMM, const Ref Src);
@@ -1664,9 +1711,9 @@ private:
if (IsNZCV(BitOffset)) {
InsertNZCV(BitOffset, Value, ValueOffset, MustMask);
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
_StoreRegister(Value, Core::CPUState::PF_AS_GREG, GPRClass, CTX->GetGPRSize());
StoreRegister(Core::CPUState::PF_AS_GREG, false, Value);
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
_StoreRegister(Value, Core::CPUState::AF_AS_GREG, GPRClass, CTX->GetGPRSize());
StoreRegister(Core::CPUState::AF_AS_GREG, false, Value);
} else {
if (ValueOffset || MustMask) {
Value = _Bfe(OpSize::i32Bit, 1, ValueOffset, Value);
@@ -1674,10 +1721,10 @@ private:
// For DF, we need to transform 0/1 into 1/-1
if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
Value = _SubShift(OpSize::i64Bit, _Constant(1), Value, ShiftType::LSL, 1);
StoreDF(_SubShift(OpSize::i64Bit, _Constant(1), Value, ShiftType::LSL, 1));
} else {
_StoreFlag(Value, BitOffset);
}
_StoreFlag(Value, BitOffset);
}
}
@@ -1693,10 +1740,13 @@ private:
void InvalidateAF() {
_InvalidateFlags((1u << X86State::RFLAG_AF_RAW_LOC));
InvalidateReg(Core::CPUState::AF_AS_GREG);
}
void InvalidatePF_AF() {
_InvalidateFlags((1u << X86State::RFLAG_PF_RAW_LOC) | (1u << X86State::RFLAG_AF_RAW_LOC));
InvalidateReg(Core::CPUState::PF_AS_GREG);
InvalidateReg(Core::CPUState::AF_AS_GREG);
}
CondClassType CondForNZCVBit(unsigned BitOffset, bool Invert) {
@@ -1713,6 +1763,148 @@ private:
}
}
/* Layout of cache indices. We use a single 64-bit bitmask for the cache */
static const int GPR0Index = 0;
static const int GPR15Index = 15;
static const int PFIndex = 16;
static const int AFIndex = 17;
/* Gap 18..19 */
static const int MM0Index = 20;
static const int MM7Index = 27;
static const int AbridgedFTWIndex = 28;
/* Gap 29..30 */
static const int DFIndex = 31;
static const int FPR0Index = 32;
static const int FPR15Index = 47;
static const int AVXHigh0Index = 48;
static const int AVXHigh15Index = 63;
int CacheIndexToContextOffset(int Index) {
switch (Index) {
case MM0Index ... MM7Index: return offsetof(FEXCore::Core::CPUState, mm[Index - MM0Index]);
case AVXHigh0Index ... AVXHigh15Index: return offsetof(FEXCore::Core::CPUState, avx_high[Index - AVXHigh0Index][0]);
case AbridgedFTWIndex: return offsetof(FEXCore::Core::CPUState, AbridgedFTW);
default: return -1;
}
}
RegisterClassType CacheIndexClass(int Index) {
if ((Index >= MM0Index && Index <= MM7Index) || Index >= FPR0Index) {
return FPRClass;
} else {
return GPRClass;
}
}
unsigned CacheIndexToSize(int Index) {
// MMX registers are rounded up to 128-bit since they are shared with 80-bit
// x87 registers, even though MMX is logically only 64-bit.
if (Index >= AVXHigh0Index || ((Index >= MM0Index && Index <= MM7Index))) {
return 16;
} else {
return 1;
}
}
struct {
uint64_t Cached;
uint64_t Written;
// Indicates that Value contains only the lower 64-bit of the full 80-bit
// register. Used for MMX/x87 optimization.
uint64_t Partial;
Ref Value[64];
} RegCache {};
void InvalidateReg(uint8_t Index) {
uint64_t Bit = (1ull << (uint64_t)Index);
RegCache.Cached &= ~Bit;
RegCache.Written &= ~Bit;
}
Ref LoadRegCache(uint64_t Offset, uint8_t Index, RegisterClassType RegClass, uint8_t Size) {
LOGMAN_THROW_AA_FMT(Index < 64, "valid index");
uint64_t Bit = (1ull << (uint64_t)Index);
if (Size == 16 && (RegCache.Partial & Bit)) {
// We need to load the full register extend if we previously did a partial access.
Ref Value = RegCache.Value[Index];
Ref Full = _LoadContext(Size, RegClass, Offset);
// If we did a partial store, we're inserting into the full register
if (RegCache.Written & Bit) {
Full = _VInsElement(16, 8, 0, 0, Full, Value);
}
RegCache.Value[Index] = Full;
}
if (!(RegCache.Cached & Bit)) {
if (Index == DFIndex) {
RegCache.Value[Index] = _LoadDF();
} else if ((Index >= MM0Index && Index <= AbridgedFTWIndex) || Index >= AVXHigh0Index) {
RegCache.Value[Index] = _LoadContext(Size, RegClass, Offset);
// We may have done a partial load, this requires special handling.
if (Size == 8) {
RegCache.Partial |= Bit;
}
} else {
RegCache.Value[Index] = _LoadRegister(Offset, RegClass, Size);
}
RegCache.Cached |= Bit;
}
return RegCache.Value[Index];
}
Ref LoadGPR(uint8_t Reg) {
return LoadRegCache(Reg, GPR0Index + Reg, GPRClass, CTX->GetGPRSize());
}
Ref LoadContext(uint8_t Size, uint8_t Index) {
return LoadRegCache(CacheIndexToContextOffset(Index), Index, CacheIndexClass(Index), Size);
}
Ref LoadContext(uint8_t Index) {
return LoadContext(CacheIndexToSize(Index), Index);
}
Ref LoadXMMRegister(uint8_t Reg) {
return LoadRegCache(Reg, FPR0Index + Reg, FPRClass, GetGuestVectorLength());
}
Ref LoadDF() {
return LoadGPR(DFIndex);
}
void StoreContext(uint8_t Index, Ref Value) {
LOGMAN_THROW_AA_FMT(Index < 64, "valid index");
LOGMAN_THROW_AA_FMT(Value != InvalidNode, "storing valid");
uint64_t Bit = (1ull << (uint64_t)Index);
RegCache.Value[Index] = Value;
RegCache.Cached |= Bit;
RegCache.Written |= Bit;
}
void StoreContextPartial(uint8_t Index, Ref Value) {
StoreContext(Index, Value);
RegCache.Partial |= (1ull << (uint64_t)Index);
}
void StoreRegister(uint8_t Reg, bool FPR, Ref Value) {
StoreContext(Reg + (FPR ? FPR0Index : GPR0Index), Value);
}
void StoreDF(Ref Value) {
StoreContext(DFIndex, Value);
}
Ref GetRFLAG(unsigned BitOffset, bool Invert = false) {
if (IsNZCV(BitOffset)) {
if (!(PossiblySetNZCVBits & (1u << IndexNZCV(BitOffset)))) {
@@ -1729,12 +1921,12 @@ private:
return _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(BitOffset, Invert), _Constant(1), _Constant(0));
}
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
return _LoadRegister(Core::CPUState::PF_AS_GREG, GPRClass, CTX->GetGPRSize());
return LoadGPR(Core::CPUState::PF_AS_GREG);
} else if (BitOffset == FEXCore::X86State::RFLAG_AF_RAW_LOC) {
return _LoadRegister(Core::CPUState::AF_AS_GREG, GPRClass, CTX->GetGPRSize());
return LoadGPR(Core::CPUState::AF_AS_GREG);
} else if (BitOffset == FEXCore::X86State::RFLAG_DF_RAW_LOC) {
// Recover the sign bit, it is the logical DF value
return _Lshr(OpSize::i64Bit, _LoadDF(), _Constant(63));
return _Lshr(OpSize::i64Bit, LoadDF(), _Constant(63));
} else {
return _LoadFlag(BitOffset);
}
@@ -1742,7 +1934,7 @@ private:
// Returns (DF ? -Size : Size)
Ref LoadDir(const unsigned Size) {
auto Dir = _LoadDF();
auto Dir = LoadDF();
auto Shift = FEXCore::ilog2(Size);
if (Shift) {
@@ -1756,7 +1948,7 @@ private:
Ref OffsetByDir(Ref X, const unsigned Size) {
auto Shift = FEXCore::ilog2(Size);
return _AddShift(OpSize::i64Bit, X, _LoadDF(), ShiftType::LSL, Shift);
return _AddShift(OpSize::i64Bit, X, LoadDF(), ShiftType::LSL, Shift);
}
// Compares two floats and sets flags for a COMISS instruction
@@ -572,17 +572,17 @@ void OpDispatchBuilder::AVX128_StoreResult_WithOpSize(FEXCore::X86Tables::Decode
Ref OpDispatchBuilder::AVX128_LoadXMMRegister(uint32_t XMM, bool High) {
if (High) {
return _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, avx_high[XMM][0]));
return LoadContext(AVXHigh0Index + XMM);
} else {
return _LoadRegister(XMM, FPRClass, 16);
return LoadXMMRegister(XMM);
}
}
void OpDispatchBuilder::AVX128_StoreXMMRegister(uint32_t XMM, const Ref Src, bool High) {
if (High) {
_StoreContext(16, FPRClass, Src, offsetof(FEXCore::Core::CPUState, avx_high[XMM][0]));
StoreContext(AVXHigh0Index + XMM, Src);
} else {
_StoreRegister(Src, XMM, FPRClass, 16);
StoreXMMRegister(XMM, Src);
}
}
@@ -2517,8 +2517,7 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
{
// Abridged FTW
auto AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
_StoreMem(GPRClass, 1, AbridgedFTW, MemBase, _Constant(4), 2, MEM_OFFSET_SXTX, 1);
_StoreMem(GPRClass, 1, LoadContext(AbridgedFTWIndex), MemBase, _Constant(4), 2, MEM_OFFSET_SXTX, 1);
}
// BYTE | 0 1 | 2 3 | 4 | 5 | 6 7 | 8 9 | a b | c d | e f |
@@ -2566,7 +2565,7 @@ void OpDispatchBuilder::SaveX87State(OpcodeArgs, Ref MemBase) {
// If OSFXSR bit in CR4 is not set than FXSAVE /may/ not save the XMM registers
// This is implementation dependent
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
Ref MMReg = _LoadContext(16, FPRClass, offsetof(FEXCore::Core::CPUState, mm[i]));
Ref MMReg = LoadContext(MM0Index + i);
_StoreMem(FPRClass, 16, MMReg, MemBase, _Constant(i * 16 + 32), 16, MEM_OFFSET_SXTX, 1);
}
@@ -2693,13 +2692,12 @@ void OpDispatchBuilder::RestoreX87State(Ref MemBase) {
{
// Abridged FTW
auto NewAbridgedFTW = _LoadMem(GPRClass, 1, MemBase, _Constant(4), 2, MEM_OFFSET_SXTX, 1);
_StoreContext(1, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, _LoadMem(GPRClass, 1, MemBase, _Constant(4), 2, MEM_OFFSET_SXTX, 1));
}
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
auto MMReg = _LoadMem(FPRClass, 16, MemBase, _Constant(i * 16 + 32), 16, MEM_OFFSET_SXTX, 1);
_StoreContext(16, FPRClass, MMReg, offsetof(FEXCore::Core::CPUState, mm[i]));
StoreContext(MM0Index + i, MMReg);
}
}
@@ -2738,7 +2736,7 @@ void OpDispatchBuilder::DefaultX87State(OpcodeArgs) {
// all of the ST0-7/MM0-7 registers to zero.
Ref ZeroVector = LoadZeroVector(Core::CPUState::MM_REG_SIZE);
for (uint32_t i = 0; i < Core::CPUState::NUM_MMS; ++i) {
_StoreContext(16, FPRClass, ZeroVector, offsetof(FEXCore::Core::CPUState, mm[i]));
StoreContext(MM0Index + i, ZeroVector);
}
}
@@ -31,14 +31,14 @@ Ref OpDispatchBuilder::GetX87Top() {
void OpDispatchBuilder::SetX87ValidTag(Ref Value, bool Valid) {
// if we are popping then we must first mark this location as empty
Ref AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
Ref AbridgedFTW = LoadContext(AbridgedFTWIndex);
Ref RegMask = _Lshl(OpSize::i32Bit, _Constant(1), Value);
Ref NewAbridgedFTW = Valid ? _Or(OpSize::i32Bit, AbridgedFTW, RegMask) : _Andn(OpSize::i32Bit, AbridgedFTW, RegMask);
_StoreContext(1, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, NewAbridgedFTW);
}
Ref OpDispatchBuilder::GetX87ValidTag(Ref Value) {
Ref AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
Ref AbridgedFTW = LoadContext(AbridgedFTWIndex);
return _And(OpSize::i32Bit, _Lshr(OpSize::i32Bit, AbridgedFTW, Value), _Constant(1));
}
@@ -51,8 +51,7 @@ Ref OpDispatchBuilder::GetX87Tag(Ref Value, Ref AbridgedFTW) {
}
Ref OpDispatchBuilder::GetX87Tag(Ref Value) {
Ref AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
return GetX87Tag(Value, AbridgedFTW);
return GetX87Tag(Value, LoadContext(AbridgedFTWIndex));
}
void OpDispatchBuilder::SetX87FTW(Ref FTW) {
@@ -70,11 +69,11 @@ void OpDispatchBuilder::SetX87FTW(Ref FTW) {
}
}
_StoreContext(1, GPRClass, NewAbridgedFTW, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, NewAbridgedFTW);
}
Ref OpDispatchBuilder::GetX87FTW() {
Ref AbridgedFTW = _LoadContext(1, GPRClass, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
Ref AbridgedFTW = LoadContext(AbridgedFTWIndex);
Ref FTW = _Constant(0);
for (int i = 0; i < 8; i++) {
@@ -153,7 +152,6 @@ void OpDispatchBuilder::FLD(OpcodeArgs, size_t width) {
SetX87Top(top);
// Write to ST[TOP]
_StoreContextIndexed(converted, top, 16, MMBaseOffset(), 16, FPRClass);
//_StoreContext(converted, 16, offsetof(FEXCore::Core::CPUState, mm[7][0]));
}
void OpDispatchBuilder::FBLD(OpcodeArgs) {
@@ -565,7 +563,7 @@ void OpDispatchBuilder::FNINIT(OpcodeArgs) {
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(Zero);
// Tags all get marked as invalid
_StoreContext(1, GPRClass, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, Zero);
}
void OpDispatchBuilder::FCOMI(OpcodeArgs, size_t width, bool Integer, OpDispatchBuilder::FCOMIFlags whichflags, bool poptwice) {
@@ -1107,7 +1105,7 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
void OpDispatchBuilder::X87EMMS(OpcodeArgs) {
// Tags all get set to 0b11
_StoreContext(1, GPRClass, _Constant(0), offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, _Constant(0));
}
void OpDispatchBuilder::X87FFREE(OpcodeArgs) {
@@ -60,7 +60,7 @@ void OpDispatchBuilder::FNINITF64(OpcodeArgs) {
SetRFLAG<FEXCore::X86State::X87FLAG_C3_LOC>(Zero);
// Tags all get marked as invalid
_StoreContext(1, GPRClass, Zero, offsetof(FEXCore::Core::CPUState, AbridgedFTW));
StoreContext(AbridgedFTWIndex, Zero);
}
void OpDispatchBuilder::X87LDENVF64(OpcodeArgs) {
@@ -70,7 +70,6 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
FEX_CONFIG_OPT(DisablePasses, O0);
if (!DisablePasses()) {
InsertPass(CreateContextLoadStoreElimination(ctx->HostFeatures.SupportsAVX && ctx->HostFeatures.SupportsSVE256));
InsertPass(CreateDeadStoreElimination());
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9, &ctx->CPUID));
InsertPass(CreateDeadFlagCalculationEliminination());
-1
View File
@@ -17,7 +17,6 @@ class RegisterAllocationPass;
class RegisterAllocationData;
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9, const FEXCore::CPUIDEmu* CPUID);
fextl::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination(bool SupportsSVE256);
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadStoreElimination();
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass();
@@ -1,772 +0,0 @@
// SPDX-License-Identifier: MIT
/*
$info$
tags: ir|opts
desc: Transforms ContextLoad/Store to temporaries, similar to mem2reg
$end_info$
*/
#include "Interface/IR/IR.h"
#include "Interface/IR/IREmitter.h"
#include "Interface/IR/Passes.h"
#include "Interface/IR/PassManager.h"
#include <FEXCore/Core/CoreState.h>
#include <FEXCore/Core/X86Enums.h>
#include <FEXCore/IR/IR.h>
#include <FEXCore/Utils/EnumOperators.h>
#include <FEXCore/Utils/LogManager.h>
#include <FEXCore/Utils/Profiler.h>
#include <FEXCore/fextl/unordered_map.h>
#include <FEXCore/fextl/vector.h>
#include <array>
#include <memory>
#include <stddef.h>
#include <stdint.h>
#include <unordered_map>
#include <utility>
namespace {
struct ContextMemberClassification {
size_t Offset;
uint16_t Size;
};
enum class LastAccessType {
NONE = (0b000 << 0), ///< Was never previously accessed
WRITE = (0b001 << 0), ///< Was fully overwritten
READ = (0b010 << 0), ///< Was fully read
INVALID = (0b011 << 0), ///< Accessing this is invalid
MASK = (0b011 << 0),
PARTIAL = (0b100 << 0),
PARTIAL_WRITE = (PARTIAL | WRITE), ///< Was partially written
PARTIAL_READ = (PARTIAL | READ), ///< Was partially read
};
FEX_DEF_NUM_OPS(LastAccessType);
static bool IsWriteAccess(LastAccessType Type) {
return (Type & LastAccessType::MASK) == LastAccessType::WRITE;
}
static bool IsReadAccess(LastAccessType Type) {
return (Type & LastAccessType::MASK) == LastAccessType::READ;
}
[[maybe_unused]]
static bool IsInvalidAccess(LastAccessType Type) {
return (Type & LastAccessType::MASK) == LastAccessType::INVALID;
}
[[maybe_unused]]
static bool IsPartialAccess(LastAccessType Type) {
return (Type & LastAccessType::PARTIAL) == LastAccessType::PARTIAL;
}
[[maybe_unused]]
static bool IsFullAccess(LastAccessType Type) {
return (Type & LastAccessType::PARTIAL) == LastAccessType::NONE;
}
struct ContextMemberInfo {
ContextMemberClassification Class;
LastAccessType Accessed;
FEXCore::IR::RegisterClassType AccessRegClass;
uint32_t AccessOffset;
uint8_t AccessSize;
///< The last value that was loaded or stored.
FEXCore::IR::Ref ValueNode;
///< With a store access, the store node that is doing the operation.
FEXCore::IR::Ref StoreNode;
};
struct ContextInfo {
fextl::vector<ContextMemberInfo*> Lookup;
fextl::vector<ContextMemberInfo> ClassificationInfo;
};
static void ClassifyContextStruct(ContextInfo* ContextClassificationInfo, bool SupportsAVX256) {
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, InlineJITBlockHeader),
sizeof(FEXCore::Core::CPUState::InlineJITBlockHeader),
},
LastAccessType::INVALID,
FEXCore::IR::InvalidClass,
});
// DeferredSignalRefCount
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, DeferredSignalRefCount),
sizeof(FEXCore::Core::CPUState::DeferredSignalRefCount),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, avx_high[0][0]) + FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * i,
FEXCore::Core::CPUState::XMM_SSE_REG_SIZE,
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
}
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, rip),
sizeof(FEXCore::Core::CPUState::rip),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; ++i) {
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, gregs[0]) + sizeof(FEXCore::Core::CPUState::gregs[0]) * i,
FEXCore::Core::CPUState::GPR_REG_SIZE,
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
}
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, _pad),
sizeof(FEXCore::Core::CPUState::_pad),
},
LastAccessType::INVALID,
FEXCore::IR::InvalidClass,
});
static_assert(offsetof(FEXCore::Core::CPUState, xmm.avx.data[0][0]) == 416, "What");
static_assert(FEXCore::Core::CPUState::XMM_AVX_REG_SIZE == 32, "What");
if (SupportsAVX256) {
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, xmm.avx.data[0][0]) + FEXCore::Core::CPUState::XMM_AVX_REG_SIZE * i,
FEXCore::Core::CPUState::XMM_AVX_REG_SIZE,
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
}
} else {
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, xmm.sse.data[0][0]) + FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * i,
FEXCore::Core::CPUState::XMM_SSE_REG_SIZE,
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
}
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, xmm.sse.pad[0][0]),
static_cast<uint16_t>(FEXCore::Core::CPUState::XMM_SSE_REG_SIZE * FEXCore::Core::CPUState::NUM_XMMS),
},
LastAccessType::INVALID,
FEXCore::IR::InvalidClass,
});
}
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, es_idx),
sizeof(FEXCore::Core::CPUState::es_idx),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, cs_idx),
sizeof(FEXCore::Core::CPUState::cs_idx),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, ss_idx),
sizeof(FEXCore::Core::CPUState::ss_idx),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, ds_idx),
sizeof(FEXCore::Core::CPUState::ds_idx),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, gs_idx),
sizeof(FEXCore::Core::CPUState::gs_idx),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, fs_idx),
sizeof(FEXCore::Core::CPUState::fs_idx),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, _pad2),
sizeof(FEXCore::Core::CPUState::_pad2),
},
LastAccessType::INVALID,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, es_cached),
sizeof(FEXCore::Core::CPUState::es_cached),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, cs_cached),
sizeof(FEXCore::Core::CPUState::cs_cached),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, ss_cached),
sizeof(FEXCore::Core::CPUState::ss_cached),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, ds_cached),
sizeof(FEXCore::Core::CPUState::ds_cached),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, gs_cached),
sizeof(FEXCore::Core::CPUState::gs_cached),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, fs_cached),
sizeof(FEXCore::Core::CPUState::fs_cached),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_FLAGS; ++i) {
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, flags[0]) + sizeof(FEXCore::Core::CPUState::flags[0]) * i,
FEXCore::Core::CPUState::FLAG_SIZE,
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
}
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, pf_raw),
sizeof(FEXCore::Core::CPUState::pf_raw),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, af_raw),
sizeof(FEXCore::Core::CPUState::af_raw),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {offsetof(FEXCore::Core::CPUState, mm[0][0]) + sizeof(FEXCore::Core::CPUState::mm[0]) * i,
FEXCore::Core::CPUState::MM_REG_SIZE},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
}
// GDTs
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GDTS; ++i) {
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, gdt[0]) + sizeof(FEXCore::Core::CPUState::gdt[0]) * i,
sizeof(FEXCore::Core::CPUState::gdt[0]),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
}
// FCW
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, FCW),
sizeof(FEXCore::Core::CPUState::FCW),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
// AbridgedFTW
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, AbridgedFTW),
sizeof(FEXCore::Core::CPUState::AbridgedFTW),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
// _pad3
ContextClassification->emplace_back(ContextMemberInfo {
ContextMemberClassification {
offsetof(FEXCore::Core::CPUState, _pad3),
sizeof(FEXCore::Core::CPUState::_pad3),
},
LastAccessType::NONE,
FEXCore::IR::InvalidClass,
});
[[maybe_unused]] size_t ClassifiedStructSize {};
ContextClassificationInfo->Lookup.reserve(sizeof(FEXCore::Core::CPUState));
for (auto& it : *ContextClassification) {
LOGMAN_THROW_A_FMT(it.Class.Offset == ContextClassificationInfo->Lookup.size(), "Offset mismatch (offset={})", it.Class.Offset);
for (int i = 0; i < it.Class.Size; i++) {
ContextClassificationInfo->Lookup.push_back(&it);
}
ClassifiedStructSize += it.Class.Size;
}
LOGMAN_THROW_AA_FMT(ClassifiedStructSize == sizeof(FEXCore::Core::CPUState),
"Classified CPUStruct size doesn't match real CPUState struct size! {} (classified) != {} (real)",
ClassifiedStructSize, sizeof(FEXCore::Core::CPUState));
LOGMAN_THROW_A_FMT(ContextClassificationInfo->Lookup.size() == sizeof(FEXCore::Core::CPUState),
"Classified lookup size doesn't match real CPUState struct size! {} (classified) != {} (real)",
ContextClassificationInfo->Lookup.size(), sizeof(FEXCore::Core::CPUState));
}
static void ResetClassificationAccesses(ContextInfo* ContextClassificationInfo, bool SupportsAVX256) {
auto ContextClassification = &ContextClassificationInfo->ClassificationInfo;
auto SetAccess = [&](size_t Offset, LastAccessType Access) {
ContextClassification->at(Offset).Accessed = Access;
ContextClassification->at(Offset).AccessRegClass = FEXCore::IR::InvalidClass;
ContextClassification->at(Offset).AccessOffset = 0;
ContextClassification->at(Offset).StoreNode = nullptr;
};
size_t Offset = 0;
///< InlineJITBlockHeader
SetAccess(Offset++, LastAccessType::INVALID);
// DeferredSignalRefCount
SetAccess(Offset++, LastAccessType::INVALID);
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
///< avx_high
SetAccess(Offset++, LastAccessType::NONE);
}
// rip
SetAccess(Offset++, LastAccessType::NONE);
///< gregs
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GPRS; ++i) {
SetAccess(Offset++, LastAccessType::NONE);
}
// pad
SetAccess(Offset++, LastAccessType::NONE);
// xmm
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_XMMS; ++i) {
SetAccess(Offset++, LastAccessType::NONE);
}
// xmm_pad
if (!SupportsAVX256) {
SetAccess(Offset++, LastAccessType::NONE);
}
// Segment indexes
SetAccess(Offset++, LastAccessType::NONE);
SetAccess(Offset++, LastAccessType::NONE);
SetAccess(Offset++, LastAccessType::NONE);
SetAccess(Offset++, LastAccessType::NONE);
SetAccess(Offset++, LastAccessType::NONE);
SetAccess(Offset++, LastAccessType::NONE);
// Pad2
SetAccess(Offset++, LastAccessType::INVALID);
// Segments
SetAccess(Offset++, LastAccessType::NONE);
SetAccess(Offset++, LastAccessType::NONE);
SetAccess(Offset++, LastAccessType::NONE);
SetAccess(Offset++, LastAccessType::NONE);
SetAccess(Offset++, LastAccessType::NONE);
SetAccess(Offset++, LastAccessType::NONE);
///< flags
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_FLAGS; ++i) {
SetAccess(Offset++, LastAccessType::NONE);
}
///< pf_raw
SetAccess(Offset++, LastAccessType::NONE);
///< af_raw
SetAccess(Offset++, LastAccessType::NONE);
///< mm
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_MMS; ++i) {
SetAccess(Offset++, LastAccessType::NONE);
}
///< gdt
for (size_t i = 0; i < FEXCore::Core::CPUState::NUM_GDTS; ++i) {
SetAccess(Offset++, LastAccessType::NONE);
}
///< FCW
SetAccess(Offset++, LastAccessType::NONE);
///< AbridgedFTW
SetAccess(Offset++, LastAccessType::NONE);
// pad3
SetAccess(Offset++, LastAccessType::INVALID);
}
struct BlockInfo {
fextl::vector<FEXCore::IR::Ref> Predecessors;
fextl::vector<FEXCore::IR::Ref> Successors;
ContextInfo IncomingClassifiedStruct;
ContextInfo OutgoingClassifiedStruct;
};
class RCLSE final : public FEXCore::IR::Pass {
public:
explicit RCLSE(bool SupportsAVX256)
: SupportsAVX256 {SupportsAVX256} {
ClassifyContextStruct(&ClassifiedStruct, SupportsAVX256);
}
void Run(FEXCore::IR::IREmitter* IREmit) override;
private:
ContextInfo ClassifiedStruct;
fextl::unordered_map<FEXCore::IR::NodeID, BlockInfo> OffsetToBlockMap;
bool SupportsAVX256;
ContextMemberInfo* FindMemberInfo(ContextInfo* ClassifiedInfo, uint32_t Offset, uint8_t Size);
ContextMemberInfo* RecordAccess(ContextMemberInfo* Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size,
LastAccessType AccessType, FEXCore::IR::Ref Node, FEXCore::IR::Ref StoreNode = nullptr);
ContextMemberInfo* RecordAccess(ContextInfo* ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size,
LastAccessType AccessType, FEXCore::IR::Ref Node, FEXCore::IR::Ref StoreNode = nullptr);
void HandleLoadFlag(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::Ref CodeNode, unsigned Flag);
// Classify context loads and stores.
void ClassifyContextLoad(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset,
uint8_t Size, FEXCore::IR::Ref CodeNode, FEXCore::IR::NodeIterator BlockEnd);
void ClassifyContextStore(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset,
uint8_t Size, FEXCore::IR::Ref CodeNode, FEXCore::IR::Ref ValueNode);
// Block local Passes
void RedundantStoreLoadElimination(FEXCore::IR::IREmitter* IREmit);
unsigned OffsetForReg(FEXCore::IR::RegisterClassType Class, unsigned Reg, unsigned Size) {
if (Class == FEXCore::IR::FPRClass) {
return Size == 32 ? offsetof(FEXCore::Core::CPUState, xmm.avx.data[Reg][0]) : offsetof(FEXCore::Core::CPUState, xmm.sse.data[Reg][0]);
} else if (Reg == FEXCore::Core::CPUState::PF_AS_GREG) {
return offsetof(FEXCore::Core::CPUState, pf_raw);
} else if (Reg == FEXCore::Core::CPUState::AF_AS_GREG) {
return offsetof(FEXCore::Core::CPUState, af_raw);
} else {
return offsetof(FEXCore::Core::CPUState, gregs[Reg]);
}
}
};
ContextMemberInfo* RCLSE::FindMemberInfo(ContextInfo* ContextClassificationInfo, uint32_t Offset, uint8_t Size) {
return ContextClassificationInfo->Lookup.at(Offset);
}
ContextMemberInfo* RCLSE::RecordAccess(ContextMemberInfo* Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size,
LastAccessType AccessType, FEXCore::IR::Ref ValueNode, FEXCore::IR::Ref StoreNode) {
LOGMAN_THROW_AA_FMT((Offset + Size) <= (Info->Class.Offset + Info->Class.Size), "Access to context item went over member size");
LOGMAN_THROW_AA_FMT(Info->Accessed != LastAccessType::INVALID, "Tried to access invalid member");
// If we aren't fully overwriting the member then it is a partial write that we need to track
if (Size < Info->Class.Size) {
AccessType = AccessType == LastAccessType::WRITE ? LastAccessType::PARTIAL_WRITE : LastAccessType::PARTIAL_READ;
}
if (Size > Info->Class.Size) {
LOGMAN_MSG_A_FMT("Can't handle this");
}
Info->Accessed = AccessType;
Info->AccessRegClass = RegClass;
Info->AccessOffset = Offset;
Info->AccessSize = Size;
Info->ValueNode = ValueNode;
if (StoreNode != nullptr) {
Info->StoreNode = StoreNode;
}
return Info;
}
ContextMemberInfo* RCLSE::RecordAccess(ContextInfo* ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size,
LastAccessType AccessType, FEXCore::IR::Ref ValueNode, FEXCore::IR::Ref StoreNode) {
ContextMemberInfo* Info = FindMemberInfo(ClassifiedInfo, Offset, Size);
return RecordAccess(Info, RegClass, Offset, Size, AccessType, ValueNode, StoreNode);
}
void RCLSE::ClassifyContextLoad(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::RegisterClassType Class,
uint32_t Offset, uint8_t Size, FEXCore::IR::Ref CodeNode, FEXCore::IR::NodeIterator BlockEnd) {
auto Info = FindMemberInfo(LocalInfo, Offset, Size);
ContextMemberInfo PreviousMemberInfoCopy = *Info;
RecordAccess(Info, Class, Offset, Size, LastAccessType::READ, CodeNode);
if (PreviousMemberInfoCopy.AccessRegClass == Info->AccessRegClass && PreviousMemberInfoCopy.AccessOffset == Info->AccessOffset &&
PreviousMemberInfoCopy.AccessSize == Size) {
// This optimizes two cases:
// - Previous access was a load, and we have a redundant load of the same value.
// - Previous access was a store, and we are redundantly loading immediately after the store. Eliminating the store.
IREmit->ReplaceAllUsesWithRange(CodeNode, PreviousMemberInfoCopy.ValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
RecordAccess(Info, Class, Offset, Size, LastAccessType::READ, PreviousMemberInfoCopy.ValueNode);
}
// TODO: Optimize the case of partial loads.
}
void RCLSE::ClassifyContextStore(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::RegisterClassType Class,
uint32_t Offset, uint8_t Size, FEXCore::IR::Ref CodeNode, FEXCore::IR::Ref ValueNode) {
auto Info = FindMemberInfo(LocalInfo, Offset, Size);
ContextMemberInfo PreviousMemberInfoCopy = *Info;
RecordAccess(Info, Class, Offset, Size, LastAccessType::WRITE, ValueNode, CodeNode);
if (PreviousMemberInfoCopy.AccessRegClass == Info->AccessRegClass && PreviousMemberInfoCopy.AccessOffset == Info->AccessOffset &&
PreviousMemberInfoCopy.AccessSize == Size && PreviousMemberInfoCopy.Accessed == LastAccessType::WRITE) {
// This optimizes redundant stores with no intervening load
// TODO: this is causing RA to fall over in some titles, disabling for now.
// Revisit when the new RA lands.
#if 0
IREmit->Remove(PreviousMemberInfoCopy.StoreNode);
#endif
}
// TODO: Optimize the case of partial stores.
}
void RCLSE::HandleLoadFlag(FEXCore::IR::IREmitter* IREmit, ContextInfo* LocalInfo, FEXCore::IR::Ref CodeNode, unsigned Flag) {
const auto FlagOffset = offsetof(FEXCore::Core::CPUState, flags[Flag]);
auto Info = FindMemberInfo(LocalInfo, FlagOffset, 1);
LastAccessType LastAccess = Info->Accessed;
auto LastValueNode = Info->ValueNode;
if (IsWriteAccess(LastAccess)) { // 1 byte so always a full write
// If the last store matches this load value then we can replace the loaded value with the previous valid one
IREmit->SetWriteCursor(CodeNode);
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
RecordAccess(Info, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::READ, LastValueNode);
} else if (IsReadAccess(LastAccess)) {
IREmit->ReplaceAllUsesWith(CodeNode, LastValueNode);
RecordAccess(Info, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::READ, LastValueNode);
}
}
/**
* @brief This pass removes redundant pairs of storecontext and loadcontext ops
*
* eg.
* %26 i128 = LoadMem %25 i64, 0x10
* (%%27) StoreContext %26 i128, 0x10, 0xb0
* %28 i128 = LoadContext 0x10, 0x90
* %29 i128 = LoadContext 0x10, 0xb0
* Converts to
* %26 i128 = LoadMem %25 i64, 0x10
* (%%27) StoreContext %26 i128, 0x10, 0xb0
* %28 i128 = LoadContext 0x10, 0x90
*
* eg.
* %6 i128 = LoadContext 0x10, 0x90
* %7 i128 = LoadContext 0x10, 0x90
* %8 i128 = VXor %7 i128, %6 i128
* Converts to
* %6 i128 = LoadContext 0x10, 0x90
* %7 i128 = VXor %6 i128, %6 i128
*
* eg.
* (%%189) StoreContext %188 i128, 0x10, 0xa0
* %190 i128 = LoadContext 0x10, 0x90
* %192 i128 = VAdd %188 i128, %190 i128, 0x10, 0x4
* (%%193) StoreContext %192 i128, 0x10, 0xa0
* Converts to
* %173 i128 = LoadContext 0x10, 0x90
* %175 i128 = VAdd %172 i128, %173 i128, 0x10, 0x4
* (%%176) StoreContext %175 i128, 0x10, 0xa0
*/
void RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter* IREmit) {
using namespace FEXCore;
using namespace FEXCore::IR;
auto CurrentIR = IREmit->ViewIR();
auto OriginalWriteCursor = IREmit->GetWriteCursor();
// XXX: Walk the list and calculate the control flow
ContextInfo& LocalInfo = ClassifiedStruct;
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
auto BlockOp = BlockHeader->CW<FEXCore::IR::IROp_CodeBlock>();
auto BlockEnd = IREmit->GetIterator(BlockOp->Last);
ResetClassificationAccesses(&LocalInfo, SupportsAVX256);
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
if (IROp->Op == OP_STORECONTEXT) {
auto Op = IROp->CW<IR::IROp_StoreContext>();
ClassifyContextStore(IREmit, &LocalInfo, Op->Class, Op->Offset, IROp->Size, CodeNode, CurrentIR.GetNode(Op->Value));
} else if (IROp->Op == OP_STOREREGISTER) {
auto Op = IROp->CW<IR::IROp_StoreRegister>();
auto Offset = OffsetForReg(Op->Class, Op->Reg, IROp->Size);
ClassifyContextStore(IREmit, &LocalInfo, Op->Class, Offset, IROp->Size, CodeNode, CurrentIR.GetNode(Op->Value));
} else if (IROp->Op == OP_LOADREGISTER) {
auto Op = IROp->CW<IR::IROp_LoadRegister>();
auto Offset = OffsetForReg(Op->Class, Op->Reg, IROp->Size);
ClassifyContextLoad(IREmit, &LocalInfo, Op->Class, Offset, IROp->Size, CodeNode, BlockEnd);
} else if (IROp->Op == OP_LOADCONTEXT) {
auto Op = IROp->CW<IR::IROp_LoadContext>();
ClassifyContextLoad(IREmit, &LocalInfo, Op->Class, Op->Offset, IROp->Size, CodeNode, BlockEnd);
} else if (IROp->Op == OP_STOREFLAG) {
const auto Op = IROp->CW<IR::IROp_StoreFlag>();
const auto FlagOffset = offsetof(FEXCore::Core::CPUState, flags[0]) + Op->Flag;
auto Info = FindMemberInfo(&LocalInfo, FlagOffset, 1);
auto LastStoreNode = Info->StoreNode;
RecordAccess(&LocalInfo, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::WRITE, CurrentIR.GetNode(Op->Header.Args[0]), CodeNode);
// Flags don't alias, so we can take the simple route here. Kill any flags that have been overwritten
if (LastStoreNode != nullptr) {
IREmit->Remove(LastStoreNode);
}
} else if (IROp->Op == OP_INVALIDATEFLAGS) {
auto Op = IROp->CW<IR::IROp_InvalidateFlags>();
// Loop through non-reserved flag stores and eliminate unused ones.
for (size_t F = 0; F < Core::CPUState::NUM_EFLAG_BITS; F++) {
if (!(Op->Flags & (1ULL << F))) {
continue;
}
const auto FlagOffset = offsetof(FEXCore::Core::CPUState, flags[0]) + F;
auto Info = FindMemberInfo(&LocalInfo, FlagOffset, 1);
auto LastStoreNode = Info->StoreNode;
// Flags don't alias, so we can take the simple route here. Kill any flags that have been invalidated without a read.
if (LastStoreNode != nullptr) {
IREmit->SetWriteCursor(CodeNode);
RecordAccess(&LocalInfo, FEXCore::IR::GPRClass, FlagOffset, 1, LastAccessType::WRITE, IREmit->_Constant(0), CodeNode);
IREmit->Remove(LastStoreNode);
}
}
} else if (IROp->Op == OP_LOADFLAG) {
const auto Op = IROp->CW<IR::IROp_LoadFlag>();
HandleLoadFlag(IREmit, &LocalInfo, CodeNode, Op->Flag);
} else if (IROp->Op == OP_LOADDF) {
HandleLoadFlag(IREmit, &LocalInfo, CodeNode, X86State::RFLAG_DF_RAW_LOC);
} else if (IROp->Op == OP_SYSCALL || IROp->Op == OP_INLINESYSCALL) {
FEXCore::IR::SyscallFlags Flags {};
if (IROp->Op == OP_SYSCALL) {
auto Op = IROp->C<IR::IROp_Syscall>();
Flags = Op->Flags;
} else {
auto Op = IROp->C<IR::IROp_InlineSyscall>();
Flags = Op->Flags;
}
if ((Flags & FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH) != FEXCore::IR::SyscallFlags::OPTIMIZETHROUGH) {
// We can't track through these
ResetClassificationAccesses(&LocalInfo, SupportsAVX256);
}
} else if (IROp->Op == OP_STORECONTEXTINDEXED || IROp->Op == OP_LOADCONTEXTINDEXED || IROp->Op == OP_BREAK) {
// We can't track through these
ResetClassificationAccesses(&LocalInfo, SupportsAVX256);
}
}
}
IREmit->SetWriteCursor(OriginalWriteCursor);
}
void RCLSE::Run(FEXCore::IR::IREmitter* IREmit) {
FEXCORE_PROFILE_SCOPED("PassManager::RCLSE");
RedundantStoreLoadElimination(IREmit);
}
} // namespace
namespace FEXCore::IR {
fextl::unique_ptr<FEXCore::IR::Pass> CreateContextLoadStoreElimination(bool SupportsAVX256) {
return fextl::make_unique<RCLSE>(SupportsAVX256);
}
} // namespace FEXCore::IR
-1
View File
@@ -110,7 +110,6 @@ IR to IR Optimization
- [PassManager.cpp](../FEXCore/Source/Interface/IR/PassManager.cpp): Defines which passes are run, and runs them
- [PassManager.h](../FEXCore/Source/Interface/IR/PassManager.h)
- [ConstProp.cpp](../FEXCore/Source/Interface/IR/Passes/ConstProp.cpp): ConstProp, ZExt elim, const pooling, fcmp reduction, const inlining
- [DeadContextStoreElimination.cpp](../FEXCore/Source/Interface/IR/Passes/DeadContextStoreElimination.cpp): Transforms ContextLoad/Store to temporaries, similar to mem2reg
- [DeadStoreElimination.cpp](../FEXCore/Source/Interface/IR/Passes/DeadStoreElimination.cpp): Cross block store-after-store elimination
- [IRValidation.cpp](../FEXCore/Source/Interface/IR/Passes/IRValidation.cpp): Sanity checking pass
- [RedundantFlagCalculationElimination.cpp](../FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp): This is not used right now, possibly broken
+47
View File
@@ -0,0 +1,47 @@
%ifdef CONFIG
{
"RegData": {
"XMM0": ["0x5152535455565758", "0"]
}
}
%endif
fninit
; Load all test values
fld tword [rel .test_value]
fld tword [rel .test_value]
fld tword [rel .test_value]
fld tword [rel .test_value]
fld tword [rel .test_value]
fld tword [rel .test_value]
fld tword [rel .test_value]
fld tword [rel .test_value]
; Setup for MMX usage
emms
; Load XMM value
movups xmm0, [rel .test_xmm_value]
; Load MMX value
movq mm0, [rel .test_mmx_value]
jmp .test
.test:
; Move MMX register in to XMM
; Should set the upper 64-bits of xmm0 to zero
movq2dq xmm0, mm0
hlt
align 32
.test_value:
dq 0x4142434445464748
dw 0x7fff
.test_mmx_value:
dq 0x5152535455565758
.test_xmm_value:
dq 0x6162636465666768
dq 0x7172737475767778
@@ -1685,22 +1685,22 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"str q2, [x28, #16]",
"str q2, [x28, #32]",
"str q2, [x28, #48]",
"str q2, [x28, #64]",
"str q2, [x28, #80]",
"str q2, [x28, #96]",
"str q2, [x28, #112]",
"str q2, [x28, #128]",
"str q2, [x28, #144]",
"str q2, [x28, #160]",
"str q2, [x28, #176]",
"str q2, [x28, #192]",
"str q2, [x28, #208]",
"str q2, [x28, #224]",
"str q2, [x28, #256]",
"str q2, [x28, #240]",
"str q2, [x28, #256]"
"str q2, [x28, #224]",
"str q2, [x28, #208]",
"str q2, [x28, #192]",
"str q2, [x28, #176]",
"str q2, [x28, #160]",
"str q2, [x28, #144]",
"str q2, [x28, #128]",
"str q2, [x28, #112]",
"str q2, [x28, #96]",
"str q2, [x28, #80]",
"str q2, [x28, #64]",
"str q2, [x28, #48]",
"str q2, [x28, #32]",
"str q2, [x28, #16]"
]
},
"vzeroall": {
@@ -1726,22 +1726,22 @@
"movi v29.2d, #0x0",
"movi v30.2d, #0x0",
"movi v31.2d, #0x0",
"str q31, [x28, #16]",
"str q31, [x28, #32]",
"str q31, [x28, #48]",
"str q31, [x28, #64]",
"str q31, [x28, #80]",
"str q31, [x28, #96]",
"str q31, [x28, #112]",
"str q31, [x28, #128]",
"str q31, [x28, #144]",
"str q31, [x28, #160]",
"str q31, [x28, #176]",
"str q31, [x28, #192]",
"str q31, [x28, #208]",
"str q31, [x28, #224]",
"str q31, [x28, #256]",
"str q31, [x28, #240]",
"str q31, [x28, #256]"
"str q31, [x28, #224]",
"str q31, [x28, #208]",
"str q31, [x28, #192]",
"str q31, [x28, #176]",
"str q31, [x28, #160]",
"str q31, [x28, #144]",
"str q31, [x28, #128]",
"str q31, [x28, #112]",
"str q31, [x28, #96]",
"str q31, [x28, #80]",
"str q31, [x28, #64]",
"str q31, [x28, #48]",
"str q31, [x28, #32]",
"str q31, [x28, #16]"
]
},
"vcmpps xmm0, xmm1, xmm2, 0x00": {
@@ -2692,8 +2692,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vmovaps ymm0, ymm1": {
@@ -2703,8 +2703,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vmovapd xmm0, [rax]": {
@@ -2736,8 +2736,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vmovapd ymm0, ymm1": {
@@ -2747,8 +2747,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vmovaps [rax], xmm0": {
@@ -3473,8 +3473,8 @@
"fcsel s0, s17, s18, mi",
"mov v16.s[0], v0.s[0]",
"movi v2.2d, #0x0",
"str q2, [x28, #16]",
"msr nzcv, x20"
"msr nzcv, x20",
"str q2, [x28, #16]"
]
},
"vminsd xmm0, xmm1, xmm2": {
@@ -3489,8 +3489,8 @@
"fcsel d0, d17, d18, mi",
"mov v16.d[0], v0.d[0]",
"movi v2.2d, #0x0",
"str q2, [x28, #16]",
"msr nzcv, x20"
"msr nzcv, x20",
"str q2, [x28, #16]"
]
},
"vdivps xmm0, xmm1, xmm2": {
@@ -3693,8 +3693,8 @@
"fcsel s0, s18, s17, mi",
"mov v16.s[0], v0.s[0]",
"movi v2.2d, #0x0",
"str q2, [x28, #16]",
"msr nzcv, x20"
"msr nzcv, x20",
"str q2, [x28, #16]"
]
},
"vmaxsd xmm0, xmm1, xmm2": {
@@ -3709,8 +3709,8 @@
"fcsel d0, d18, d17, mi",
"mov v16.d[0], v0.d[0]",
"movi v2.2d, #0x0",
"str q2, [x28, #16]",
"msr nzcv, x20"
"msr nzcv, x20",
"str q2, [x28, #16]"
]
},
"vpunpckhbw xmm0, xmm1, xmm2": {
+128 -128
View File
@@ -2577,8 +2577,8 @@
"add x1, x4, w0, sxtw",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherdd xmm0, [xmm1*2 + rax], xmm2": {
@@ -2608,8 +2608,8 @@
"add x1, x4, w0, sxtw #1",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherdd xmm0, [xmm1*4 + rax], xmm2": {
@@ -2639,8 +2639,8 @@
"add x1, x4, w0, sxtw #2",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherdd xmm0, [xmm1*8 + rax], xmm2": {
@@ -2670,8 +2670,8 @@
"add x1, x4, w0, sxtw #3",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherdd ymm0, [ymm1*1 + rax], ymm2": {
@@ -2723,9 +2723,9 @@
"smov x0, v3.s[3]",
"add x1, x4, w0, sxtw",
"ld1 {v2.s}[3], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vpgatherdd ymm0, [ymm1*2 + rax], ymm2": {
@@ -2777,9 +2777,9 @@
"smov x0, v3.s[3]",
"add x1, x4, w0, sxtw #1",
"ld1 {v2.s}[3], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vpgatherdd ymm0, [ymm1*4 + rax], ymm2": {
@@ -2831,9 +2831,9 @@
"smov x0, v3.s[3]",
"add x1, x4, w0, sxtw #2",
"ld1 {v2.s}[3], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vpgatherdd ymm0, [ymm1*8 + rax], ymm2": {
@@ -2885,9 +2885,9 @@
"smov x0, v3.s[3]",
"add x1, x4, w0, sxtw #3",
"ld1 {v2.s}[3], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vpgatherdq xmm0, [xmm1*1 + rax], xmm2": {
@@ -2907,8 +2907,8 @@
"add x1, x4, w0, sxtw",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherdq xmm0, [xmm1*2 + rax], xmm2": {
@@ -2928,8 +2928,8 @@
"add x1, x4, w0, sxtw #1",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherdq xmm0, [xmm1*4 + rax], xmm2": {
@@ -2949,8 +2949,8 @@
"add x1, x4, w0, sxtw #2",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherdq xmm0, [xmm1*8 + rax], xmm2": {
@@ -2970,8 +2970,8 @@
"add x1, x4, w0, sxtw #3",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherdq ymm0, [xmm1*1 + rax], ymm2": {
@@ -3002,9 +3002,9 @@
"smov x0, v17.s[3]",
"add x1, x4, w0, sxtw",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vpgatherdq ymm0, [xmm1*2 + rax], ymm2": {
@@ -3035,9 +3035,9 @@
"smov x0, v17.s[3]",
"add x1, x4, w0, sxtw #1",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vpgatherdq ymm0, [xmm1*4 + rax], ymm2": {
@@ -3068,9 +3068,9 @@
"smov x0, v17.s[3]",
"add x1, x4, w0, sxtw #2",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vpgatherdq ymm0, [xmm1*8 + rax], ymm2": {
@@ -3101,9 +3101,9 @@
"smov x0, v17.s[3]",
"add x1, x4, w0, sxtw #3",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vpgatherqd xmm0, [xmm1*1 + rax], xmm2": {
@@ -3125,8 +3125,8 @@
"ld1 {v2.s}[1], [x1]",
"movi v18.2d, #0x0",
"zip1 v16.2d, v2.2d, v18.2d",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherqd xmm0, [xmm1*2 + rax], xmm2": {
@@ -3148,8 +3148,8 @@
"ld1 {v2.s}[1], [x1]",
"movi v18.2d, #0x0",
"zip1 v16.2d, v2.2d, v18.2d",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherqd xmm0, [xmm1*4 + rax], xmm2": {
@@ -3171,8 +3171,8 @@
"ld1 {v2.s}[1], [x1]",
"movi v18.2d, #0x0",
"zip1 v16.2d, v2.2d, v18.2d",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherqd xmm0, [xmm1*8 + rax], xmm2": {
@@ -3194,8 +3194,8 @@
"ld1 {v2.s}[1], [x1]",
"movi v18.2d, #0x0",
"zip1 v16.2d, v2.2d, v18.2d",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherqd xmm0, [ymm1*1 + rax], xmm2": {
@@ -3226,8 +3226,8 @@
"add x1, x4, x0",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherqd xmm0, [ymm1*2 + rax], xmm2": {
@@ -3258,8 +3258,8 @@
"add x1, x4, x0, lsl #1",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherqd xmm0, [ymm1*4 + rax], xmm2": {
@@ -3290,8 +3290,8 @@
"add x1, x4, x0, lsl #2",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherqd xmm0, [ymm1*8 + rax], xmm2": {
@@ -3322,8 +3322,8 @@
"add x1, x4, x0, lsl #3",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherqq xmm0, [xmm1*1 + rax], xmm2": {
@@ -3343,8 +3343,8 @@
"add x1, x4, x0",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherqq xmm0, [xmm1*2 + rax], xmm2": {
@@ -3364,8 +3364,8 @@
"add x1, x4, x0, lsl #1",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherqq xmm0, [xmm1*4 + rax], xmm2": {
@@ -3385,8 +3385,8 @@
"add x1, x4, x0, lsl #2",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherqq xmm0, [xmm1*8 + rax], xmm2": {
@@ -3406,8 +3406,8 @@
"add x1, x4, x0, lsl #3",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vpgatherqq ymm0, [ymm1*1 + rax], ymm2": {
@@ -3439,9 +3439,9 @@
"mov x0, v3.d[1]",
"add x1, x4, x0",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vpgatherqq ymm0, [ymm1*2 + rax], ymm2": {
@@ -3473,9 +3473,9 @@
"mov x0, v3.d[1]",
"add x1, x4, x0, lsl #1",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vpgatherqq ymm0, [ymm1*4 + rax], ymm2": {
@@ -3507,9 +3507,9 @@
"mov x0, v3.d[1]",
"add x1, x4, x0, lsl #2",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vpgatherqq ymm0, [ymm1*8 + rax], ymm2": {
@@ -3541,9 +3541,9 @@
"mov x0, v3.d[1]",
"add x1, x4, x0, lsl #3",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vgatherdps xmm0, [xmm1*1 + rax], xmm2": {
@@ -3573,8 +3573,8 @@
"add x1, x4, w0, sxtw",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherdps xmm0, [xmm1*2 + rax], xmm2": {
@@ -3604,8 +3604,8 @@
"add x1, x4, w0, sxtw #1",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherdps xmm0, [xmm1*4 + rax], xmm2": {
@@ -3635,8 +3635,8 @@
"add x1, x4, w0, sxtw #2",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherdps xmm0, [xmm1*8 + rax], xmm2": {
@@ -3666,8 +3666,8 @@
"add x1, x4, w0, sxtw #3",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherdps ymm0, [ymm1*1 + rax], ymm2": {
@@ -3719,9 +3719,9 @@
"smov x0, v3.s[3]",
"add x1, x4, w0, sxtw",
"ld1 {v2.s}[3], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vgatherdps ymm0, [ymm1*2 + rax], ymm2": {
@@ -3773,9 +3773,9 @@
"smov x0, v3.s[3]",
"add x1, x4, w0, sxtw #1",
"ld1 {v2.s}[3], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vgatherdps ymm0, [ymm1*4 + rax], ymm2": {
@@ -3827,9 +3827,9 @@
"smov x0, v3.s[3]",
"add x1, x4, w0, sxtw #2",
"ld1 {v2.s}[3], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vgatherdps ymm0, [ymm1*8 + rax], ymm2": {
@@ -3881,9 +3881,9 @@
"smov x0, v3.s[3]",
"add x1, x4, w0, sxtw #3",
"ld1 {v2.s}[3], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vgatherdpd xmm0, [xmm1*1 + rax], xmm2": {
@@ -3903,8 +3903,8 @@
"add x1, x4, w0, sxtw",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherdpd xmm0, [xmm1*2 + rax], xmm2": {
@@ -3924,8 +3924,8 @@
"add x1, x4, w0, sxtw #1",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherdpd xmm0, [xmm1*4 + rax], xmm2": {
@@ -3945,8 +3945,8 @@
"add x1, x4, w0, sxtw #2",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherdpd xmm0, [xmm1*8 + rax], xmm2": {
@@ -3966,8 +3966,8 @@
"add x1, x4, w0, sxtw #3",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherdpd ymm0, [xmm1*1 + rax], ymm2": {
@@ -3998,9 +3998,9 @@
"smov x0, v17.s[3]",
"add x1, x4, w0, sxtw",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vgatherdpd ymm0, [xmm1*2 + rax], ymm2": {
@@ -4031,9 +4031,9 @@
"smov x0, v17.s[3]",
"add x1, x4, w0, sxtw #1",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vgatherdpd ymm0, [xmm1*4 + rax], ymm2": {
@@ -4064,9 +4064,9 @@
"smov x0, v17.s[3]",
"add x1, x4, w0, sxtw #2",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vgatherdpd ymm0, [xmm1*8 + rax], ymm2": {
@@ -4097,9 +4097,9 @@
"smov x0, v17.s[3]",
"add x1, x4, w0, sxtw #3",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vgatherqps xmm0, [xmm1*1 + rax], xmm2": {
@@ -4121,8 +4121,8 @@
"ld1 {v2.s}[1], [x1]",
"movi v18.2d, #0x0",
"zip1 v16.2d, v2.2d, v18.2d",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherqps xmm0, [xmm1*2 + rax], xmm2": {
@@ -4144,8 +4144,8 @@
"ld1 {v2.s}[1], [x1]",
"movi v18.2d, #0x0",
"zip1 v16.2d, v2.2d, v18.2d",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherqps xmm0, [xmm1*4 + rax], xmm2": {
@@ -4167,8 +4167,8 @@
"ld1 {v2.s}[1], [x1]",
"movi v18.2d, #0x0",
"zip1 v16.2d, v2.2d, v18.2d",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherqps xmm0, [xmm1*8 + rax], xmm2": {
@@ -4190,8 +4190,8 @@
"ld1 {v2.s}[1], [x1]",
"movi v18.2d, #0x0",
"zip1 v16.2d, v2.2d, v18.2d",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherqps xmm0, [ymm1*1 + rax], xmm2": {
@@ -4222,8 +4222,8 @@
"add x1, x4, x0",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherqps xmm0, [ymm1*2 + rax], xmm2": {
@@ -4254,8 +4254,8 @@
"add x1, x4, x0, lsl #1",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherqps xmm0, [ymm1*4 + rax], xmm2": {
@@ -4286,8 +4286,8 @@
"add x1, x4, x0, lsl #2",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherqps xmm0, [ymm1*8 + rax], xmm2": {
@@ -4318,8 +4318,8 @@
"add x1, x4, x0, lsl #3",
"ld1 {v16.s}[3], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherqpd xmm0, [xmm1*1 + rax], xmm2": {
@@ -4339,8 +4339,8 @@
"add x1, x4, x0",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherqpd xmm0, [xmm1*2 + rax], xmm2": {
@@ -4360,8 +4360,8 @@
"add x1, x4, x0, lsl #1",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherqpd xmm0, [xmm1*4 + rax], xmm2": {
@@ -4381,8 +4381,8 @@
"add x1, x4, x0, lsl #2",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherqpd xmm0, [xmm1*8 + rax], xmm2": {
@@ -4402,8 +4402,8 @@
"add x1, x4, x0, lsl #3",
"ld1 {v16.d}[1], [x1]",
"movi v18.2d, #0x0",
"str q18, [x28, #16]",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q18, [x28, #16]"
]
},
"vgatherqpd ymm0, [ymm1*1 + rax], ymm2": {
@@ -4435,9 +4435,9 @@
"mov x0, v3.d[1]",
"add x1, x4, x0",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vgatherqpd ymm0, [ymm1*2 + rax], ymm2": {
@@ -4469,9 +4469,9 @@
"mov x0, v3.d[1]",
"add x1, x4, x0, lsl #1",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vgatherqpd ymm0, [ymm1*4 + rax], ymm2": {
@@ -4503,9 +4503,9 @@
"mov x0, v3.d[1]",
"add x1, x4, x0, lsl #2",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vgatherqpd ymm0, [ymm1*8 + rax], ymm2": {
@@ -4537,9 +4537,9 @@
"mov x0, v3.d[1]",
"add x1, x4, x0, lsl #3",
"ld1 {v2.d}[1], [x1]",
"str q2, [x28, #16]",
"movi v18.2d, #0x0",
"str q18, [x28, #48]"
"str q18, [x28, #48]",
"str q2, [x28, #16]"
]
},
"vfmaddsub132ps xmm0, xmm1, xmm2": {
File diff suppressed because it is too large. Load diff
@@ -621,9 +621,9 @@
"Map 2 0b00 0xf2 32-bit"
],
"ExpectedArm64ASM": [
"bic w4, w5, w7",
"mov x26, x4",
"tst w4, w4"
"bic w26, w5, w7",
"tst w26, w26",
"mov x4, x26"
]
},
"andn rax, rbx, rcx": {
@@ -632,9 +632,9 @@
"Map 2 0b00 0xf2 64-bit"
],
"ExpectedArm64ASM": [
"bic x4, x5, x7",
"mov x26, x4",
"tst x4, x4"
"bic x26, x5, x7",
"tst x26, x26",
"mov x4, x26"
]
},
"bzhi eax, ebx, ecx": {
@@ -927,8 +927,8 @@
"Map 3 0b01 0x06 256-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"str q17, [x28, #16]"
"str q17, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vperm2f128 ymm0, ymm1, ymm2, 00000001b": {
@@ -947,8 +947,8 @@
"Map 3 0b01 0x06 256-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v18.16b",
"str q17, [x28, #16]"
"str q17, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vperm2f128 ymm0, ymm1, ymm2, 00000011b": {
@@ -968,8 +968,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vperm2f128 ymm0, ymm1, ymm2, 00010001b": {
@@ -989,8 +989,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v18.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vperm2f128 ymm0, ymm1, ymm2, 00010011b": {
@@ -1010,8 +1010,8 @@
"Map 3 0b01 0x06 256-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"str q18, [x28, #16]"
"str q18, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vperm2f128 ymm0, ymm1, ymm2, 00100001b": {
@@ -1030,8 +1030,8 @@
"Map 3 0b01 0x06 256-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v18.16b",
"str q18, [x28, #16]"
"str q18, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vperm2f128 ymm0, ymm1, ymm2, 00100011b": {
@@ -1051,8 +1051,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #48]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vperm2f128 ymm0, ymm1, ymm2, 00110001b": {
@@ -1073,8 +1073,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #48]",
"mov v16.16b, v18.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vperm2f128 ymm0, ymm1, ymm2, 00110011b": {
@@ -1146,8 +1146,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vperm2f128 ymm0, ymm1, ymm2, 10000001b": {
@@ -1168,8 +1168,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v18.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vperm2f128 ymm0, ymm1, ymm2, 10000011b": {
@@ -2166,8 +2166,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v18.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vpalignr xmm0, xmm1, xmm2, 1": {
@@ -2211,8 +2211,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #48]",
"mov v16.16b, v18.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vpalignr ymm0, ymm1, ymm2, 1": {
@@ -2389,8 +2389,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v18.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vinsertf128 ymm0, ymm1, xmm2, 1": {
@@ -2399,8 +2399,8 @@
"Map 3 0b01 0x18 256-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"str q18, [x28, #16]"
"str q18, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vextractf128 xmm0, ymm1, 0": {
@@ -2410,8 +2410,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vextractf128 xmm0, ymm1, 1": {
@@ -2732,8 +2732,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v18.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vinserti128 ymm0, ymm1, xmm2, 1": {
@@ -2742,8 +2742,8 @@
"Map 3 0b01 0x38 256-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"str q18, [x28, #16]"
"str q18, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vextracti128 xmm0, ymm1, 0": {
@@ -2753,8 +2753,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vextracti128 xmm0, ymm1, 1": {
@@ -3516,8 +3516,8 @@
"Map 3 0b01 0x46 256-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"str q17, [x28, #16]"
"str q17, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vperm2i128 ymm0, ymm1, ymm2, 00000001b": {
@@ -3536,8 +3536,8 @@
"Map 3 0b01 0x46 256-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v18.16b",
"str q17, [x28, #16]"
"str q17, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vperm2i128 ymm0, ymm1, ymm2, 00000011b": {
@@ -3557,8 +3557,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vperm2i128 ymm0, ymm1, ymm2, 00010001b": {
@@ -3578,8 +3578,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v18.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vperm2i128 ymm0, ymm1, ymm2, 00010011b": {
@@ -3599,8 +3599,8 @@
"Map 3 0b01 0x46 256-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v17.16b",
"str q18, [x28, #16]"
"str q18, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vperm2i128 ymm0, ymm1, ymm2, 00100001b": {
@@ -3619,8 +3619,8 @@
"Map 3 0b01 0x46 256-bit"
],
"ExpectedArm64ASM": [
"mov v16.16b, v18.16b",
"str q18, [x28, #16]"
"str q18, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vperm2i128 ymm0, ymm1, ymm2, 00100011b": {
@@ -3640,8 +3640,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #48]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vperm2i128 ymm0, ymm1, ymm2, 00110001b": {
@@ -3662,8 +3662,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #48]",
"mov v16.16b, v18.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vperm2i128 ymm0, ymm1, ymm2, 00110011b": {
@@ -3735,8 +3735,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vperm2i128 ymm0, ymm1, ymm2, 10000001b": {
@@ -3757,8 +3757,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v18.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v18.16b"
]
},
"vperm2i128 ymm0, ymm1, ymm2, 10000011b": {
@@ -18,8 +18,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsrlw xmm0, xmm1, 15": {
@@ -51,8 +51,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsrlw ymm0, ymm1, 15": {
@@ -86,8 +86,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsraw xmm0, xmm1, 15": {
@@ -119,8 +119,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsraw ymm0, ymm1, 15": {
@@ -154,8 +154,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsllw xmm0, xmm1, 15": {
@@ -187,8 +187,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsllw ymm0, ymm1, 15": {
@@ -222,8 +222,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsrld xmm0, xmm1, 31": {
@@ -255,8 +255,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsrld ymm0, ymm1, 31": {
@@ -290,8 +290,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsrad xmm0, xmm1, 31": {
@@ -323,8 +323,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsrad ymm0, ymm1, 31": {
@@ -358,8 +358,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpslld xmm0, xmm1, 31": {
@@ -391,8 +391,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpslld ymm0, ymm1, 31": {
@@ -426,8 +426,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsrlq xmm0, xmm1, 63": {
@@ -459,8 +459,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsrlq ymm0, ymm1, 63": {
@@ -494,8 +494,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsrldq xmm0, xmm1, 15": {
@@ -526,8 +526,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsrldq ymm0, ymm1, 15": {
@@ -559,8 +559,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsllq xmm0, xmm1, 63": {
@@ -592,8 +592,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpsllq ymm0, ymm1, 63": {
@@ -627,8 +627,8 @@
],
"ExpectedArm64ASM": [
"movi v2.2d, #0x0",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpslldq xmm0, xmm1, 15": {
@@ -659,8 +659,8 @@
],
"ExpectedArm64ASM": [
"ldr q2, [x28, #32]",
"mov v16.16b, v17.16b",
"str q2, [x28, #16]"
"str q2, [x28, #16]",
"mov v16.16b, v17.16b"
]
},
"vpslldq ymm0, ymm1, 15": {
@@ -1045,10 +1045,11 @@
"mov rax, gs:0x100",
"mov rbx, gs:0x14"
],
"ExpectedInstructionCount": 3,
"ExpectedInstructionCount": 4,
"ExpectedArm64ASM": [
"ldr x20, [x28, #960]",
"ldr x4, [x20, #256]",
"ldr x20, [x28, #960]",
"ldur x7, [x20, #20]"
]
}
@@ -323,10 +323,11 @@
"mov eax, gs:0x100",
"mov ebx, gs:0x14"
],
"ExpectedInstructionCount": 3,
"ExpectedInstructionCount": 4,
"ExpectedArm64ASM": [
"ldr w20, [x28, #960]",
"ldr w4, [x20, #256]",
"ldr w20, [x28, #960]",
"ldr w7, [x20, #20]"
]
}
+103 -104
View File
@@ -69,11 +69,11 @@
],
"ExpectedArm64ASM": [
"mov w20, #0x1",
"strb w20, [x28, #986]",
"ldrb w20, [x10]",
"strb w20, [x11]",
"ldrb w21, [x10]",
"strb w21, [x11]",
"add x10, x10, #0x1 (1)",
"add x11, x11, #0x1 (1)"
"add x11, x11, #0x1 (1)",
"strb w20, [x28, #986]"
]
},
"positive movsw": {
@@ -88,11 +88,11 @@
],
"ExpectedArm64ASM": [
"mov w20, #0x1",
"strb w20, [x28, #986]",
"ldrh w20, [x10]",
"strh w20, [x11]",
"ldrh w21, [x10]",
"strh w21, [x11]",
"add x10, x10, #0x2 (2)",
"add x11, x11, #0x2 (2)"
"add x11, x11, #0x2 (2)",
"strb w20, [x28, #986]"
]
},
"positive movsd": {
@@ -107,11 +107,11 @@
],
"ExpectedArm64ASM": [
"mov w20, #0x1",
"strb w20, [x28, #986]",
"ldr w20, [x10]",
"str w20, [x11]",
"ldr w21, [x10]",
"str w21, [x11]",
"add x10, x10, #0x4 (4)",
"add x11, x11, #0x4 (4)"
"add x11, x11, #0x4 (4)",
"strb w20, [x28, #986]"
]
},
"positive movsq": {
@@ -126,11 +126,11 @@
],
"ExpectedArm64ASM": [
"mov w20, #0x1",
"strb w20, [x28, #986]",
"ldr x20, [x10]",
"str x20, [x11]",
"ldr x21, [x10]",
"str x21, [x11]",
"add x10, x10, #0x8 (8)",
"add x11, x11, #0x8 (8)"
"add x11, x11, #0x8 (8)",
"strb w20, [x28, #986]"
]
},
"negative movsb": {
@@ -145,11 +145,11 @@
],
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffffff",
"strb w20, [x28, #986]",
"ldrb w20, [x10]",
"strb w20, [x11]",
"ldrb w21, [x10]",
"strb w21, [x11]",
"sub x10, x10, #0x1 (1)",
"sub x11, x11, #0x1 (1)"
"sub x11, x11, #0x1 (1)",
"strb w20, [x28, #986]"
]
},
"negative movsw": {
@@ -164,11 +164,11 @@
],
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffffff",
"strb w20, [x28, #986]",
"ldrh w20, [x10]",
"strh w20, [x11]",
"ldrh w21, [x10]",
"strh w21, [x11]",
"sub x10, x10, #0x2 (2)",
"sub x11, x11, #0x2 (2)"
"sub x11, x11, #0x2 (2)",
"strb w20, [x28, #986]"
]
},
"negative movsd": {
@@ -183,11 +183,11 @@
],
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffffff",
"strb w20, [x28, #986]",
"ldr w20, [x10]",
"str w20, [x11]",
"ldr w21, [x10]",
"str w21, [x11]",
"sub x10, x10, #0x4 (4)",
"sub x11, x11, #0x4 (4)"
"sub x11, x11, #0x4 (4)",
"strb w20, [x28, #986]"
]
},
"negative movsq": {
@@ -202,11 +202,11 @@
],
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffffff",
"strb w20, [x28, #986]",
"ldr x20, [x10]",
"str x20, [x11]",
"ldr x21, [x10]",
"str x21, [x11]",
"sub x10, x10, #0x8 (8)",
"sub x11, x11, #0x8 (8)"
"sub x11, x11, #0x8 (8)",
"strb w20, [x28, #986]"
]
},
"positive rep movsb": {
@@ -222,7 +222,6 @@
"ExpectedArm64ASM": [
"mov w20, #0x0",
"mov w21, #0x1",
"strb w21, [x28, #986]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
@@ -263,6 +262,7 @@
"add x23, x1, x2",
"mov x10, x23",
"mov x11, x22",
"strb w21, [x28, #986]",
"mov x5, x20"
]
},
@@ -279,7 +279,6 @@
"ExpectedArm64ASM": [
"mov w20, #0x0",
"mov w21, #0x1",
"strb w21, [x28, #986]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
@@ -320,6 +319,7 @@
"add x23, x1, x2, lsl #1",
"mov x10, x23",
"mov x11, x22",
"strb w21, [x28, #986]",
"mov x5, x20"
]
},
@@ -336,7 +336,6 @@
"ExpectedArm64ASM": [
"mov w20, #0x0",
"mov w21, #0x1",
"strb w21, [x28, #986]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
@@ -377,6 +376,7 @@
"add x23, x1, x2, lsl #2",
"mov x10, x23",
"mov x11, x22",
"strb w21, [x28, #986]",
"mov x5, x20"
]
},
@@ -393,7 +393,6 @@
"ExpectedArm64ASM": [
"mov w20, #0x0",
"mov w21, #0x1",
"strb w21, [x28, #986]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
@@ -434,6 +433,7 @@
"add x23, x1, x2, lsl #3",
"mov x10, x23",
"mov x11, x22",
"strb w21, [x28, #986]",
"mov x5, x20"
]
},
@@ -449,7 +449,6 @@
],
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffffff",
"strb w20, [x28, #986]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
@@ -490,11 +489,12 @@
"mov x0, x11",
"mov x1, x10",
"mov x2, x5",
"sub x20, x0, x2",
"sub x21, x1, x2",
"mov x10, x21",
"mov x11, x20",
"mov w5, #0x0"
"sub x22, x0, x2",
"sub x23, x1, x2",
"mov x10, x23",
"mov x11, x22",
"mov w5, #0x0",
"strb w20, [x28, #986]"
]
},
"negative rep movsw": {
@@ -509,7 +509,6 @@
],
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffffff",
"strb w20, [x28, #986]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
@@ -550,11 +549,12 @@
"mov x0, x11",
"mov x1, x10",
"mov x2, x5",
"sub x20, x0, x2, lsl #1",
"sub x21, x1, x2, lsl #1",
"mov x10, x21",
"mov x11, x20",
"mov w5, #0x0"
"sub x22, x0, x2, lsl #1",
"sub x23, x1, x2, lsl #1",
"mov x10, x23",
"mov x11, x22",
"mov w5, #0x0",
"strb w20, [x28, #986]"
]
},
"negative rep movsd": {
@@ -569,7 +569,6 @@
],
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffffff",
"strb w20, [x28, #986]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
@@ -610,11 +609,12 @@
"mov x0, x11",
"mov x1, x10",
"mov x2, x5",
"sub x20, x0, x2, lsl #2",
"sub x21, x1, x2, lsl #2",
"mov x10, x21",
"mov x11, x20",
"mov w5, #0x0"
"sub x22, x0, x2, lsl #2",
"sub x23, x1, x2, lsl #2",
"mov x10, x23",
"mov x11, x22",
"mov w5, #0x0",
"strb w20, [x28, #986]"
]
},
"negative rep movsq": {
@@ -629,7 +629,6 @@
],
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffffff",
"strb w20, [x28, #986]",
"mov x0, x5",
"mov x1, x11",
"mov x2, x10",
@@ -670,11 +669,12 @@
"mov x0, x11",
"mov x1, x10",
"mov x2, x5",
"sub x20, x0, x2, lsl #3",
"sub x21, x1, x2, lsl #3",
"mov x10, x21",
"mov x11, x20",
"mov w5, #0x0"
"sub x22, x0, x2, lsl #3",
"sub x23, x1, x2, lsl #3",
"mov x10, x23",
"mov x11, x22",
"mov w5, #0x0",
"strb w20, [x28, #986]"
]
},
"positive rep stosb": {
@@ -690,14 +690,13 @@
"ExpectedArm64ASM": [
"mov w20, #0x0",
"mov w21, #0x1",
"strb w21, [x28, #986]",
"uxtb w21, w4",
"uxtb w22, w4",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x58",
"sub x0, x0, #0x20 (32)",
"tbnz x0, #63, #+0x3c",
"dup v1.16b, w21",
"dup v1.16b, w22",
"sub x0, x0, #0x20 (32)",
"tbnz x0, #63, #+0x14",
"stp q1, q1, [x1], #32",
@@ -713,10 +712,11 @@
"tbz x0, #63, #-0x8",
"add x0, x0, #0x20 (32)",
"cbz x0, #+0x10",
"strb w21, [x1], #1",
"strb w22, [x1], #1",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"add x11, x11, x5",
"strb w21, [x28, #986]",
"mov x5, x20"
]
},
@@ -733,14 +733,13 @@
"ExpectedArm64ASM": [
"mov w20, #0x0",
"mov w21, #0x1",
"strb w21, [x28, #986]",
"uxth w21, w4",
"uxth w22, w4",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x58",
"sub x0, x0, #0x10 (16)",
"tbnz x0, #63, #+0x3c",
"dup v1.8h, w21",
"dup v1.8h, w22",
"sub x0, x0, #0x10 (16)",
"tbnz x0, #63, #+0x14",
"stp q1, q1, [x1], #32",
@@ -756,10 +755,11 @@
"tbz x0, #63, #-0x8",
"add x0, x0, #0x10 (16)",
"cbz x0, #+0x10",
"strh w21, [x1], #2",
"strh w22, [x1], #2",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"add x11, x11, x5, lsl #1",
"strb w21, [x28, #986]",
"mov x5, x20"
]
},
@@ -776,14 +776,13 @@
"ExpectedArm64ASM": [
"mov w20, #0x0",
"mov w21, #0x1",
"strb w21, [x28, #986]",
"mov w21, w4",
"mov w22, w4",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x58",
"sub x0, x0, #0x8 (8)",
"tbnz x0, #63, #+0x3c",
"dup v1.4s, w21",
"dup v1.4s, w22",
"sub x0, x0, #0x8 (8)",
"tbnz x0, #63, #+0x14",
"stp q1, q1, [x1], #32",
@@ -799,10 +798,11 @@
"tbz x0, #63, #-0x8",
"add x0, x0, #0x8 (8)",
"cbz x0, #+0x10",
"str w21, [x1], #4",
"str w22, [x1], #4",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"add x11, x11, x5, lsl #2",
"strb w21, [x28, #986]",
"mov x5, x20"
]
},
@@ -819,7 +819,6 @@
"ExpectedArm64ASM": [
"mov w20, #0x0",
"mov w21, #0x1",
"strb w21, [x28, #986]",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x58",
@@ -845,6 +844,7 @@
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"add x11, x11, x5, lsl #3",
"strb w21, [x28, #986]",
"mov x5, x20"
]
},
@@ -860,15 +860,14 @@
],
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffffff",
"strb w20, [x28, #986]",
"uxtb w20, w4",
"uxtb w21, w4",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x60",
"sub x1, x1, #0x1f (31)",
"sub x0, x0, #0x20 (32)",
"tbnz x0, #63, #+0x3c",
"dup v1.16b, w20",
"dup v1.16b, w21",
"sub x0, x0, #0x20 (32)",
"tbnz x0, #63, #+0x14",
"stp q1, q1, [x1], #-32",
@@ -885,11 +884,12 @@
"add x0, x0, #0x20 (32)",
"cbz x0, #+0x14",
"add x1, x1, #0x1f (31)",
"strb w20, [x1], #-1",
"strb w21, [x1], #-1",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"sub x11, x11, x5",
"mov w5, #0x0"
"mov w5, #0x0",
"strb w20, [x28, #986]"
]
},
"negative rep stosw": {
@@ -904,15 +904,14 @@
],
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffffff",
"strb w20, [x28, #986]",
"uxth w20, w4",
"uxth w21, w4",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x60",
"sub x1, x1, #0x1e (30)",
"sub x0, x0, #0x10 (16)",
"tbnz x0, #63, #+0x3c",
"dup v1.8h, w20",
"dup v1.8h, w21",
"sub x0, x0, #0x10 (16)",
"tbnz x0, #63, #+0x14",
"stp q1, q1, [x1], #-32",
@@ -929,11 +928,12 @@
"add x0, x0, #0x10 (16)",
"cbz x0, #+0x14",
"add x1, x1, #0x1e (30)",
"strh w20, [x1], #-2",
"strh w21, [x1], #-2",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"sub x11, x11, x5, lsl #1",
"mov w5, #0x0"
"mov w5, #0x0",
"strb w20, [x28, #986]"
]
},
"negative rep stosd": {
@@ -948,15 +948,14 @@
],
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffffff",
"strb w20, [x28, #986]",
"mov w20, w4",
"mov w21, w4",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x60",
"sub x1, x1, #0x1c (28)",
"sub x0, x0, #0x8 (8)",
"tbnz x0, #63, #+0x3c",
"dup v1.4s, w20",
"dup v1.4s, w21",
"sub x0, x0, #0x8 (8)",
"tbnz x0, #63, #+0x14",
"stp q1, q1, [x1], #-32",
@@ -973,11 +972,12 @@
"add x0, x0, #0x8 (8)",
"cbz x0, #+0x14",
"add x1, x1, #0x1c (28)",
"str w20, [x1], #-4",
"str w21, [x1], #-4",
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"sub x11, x11, x5, lsl #2",
"mov w5, #0x0"
"mov w5, #0x0",
"strb w20, [x28, #986]"
]
},
"negative rep stosq": {
@@ -992,7 +992,6 @@
],
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffffff",
"strb w20, [x28, #986]",
"mov x0, x5",
"mov x1, x11",
"cbz x0, #+0x60",
@@ -1020,11 +1019,12 @@
"sub x0, x0, #0x1 (1)",
"cbnz x0, #-0x8",
"sub x11, x11, x5, lsl #3",
"mov w5, #0x0"
"mov w5, #0x0",
"strb w20, [x28, #986]"
]
},
"Sekiro spill block": {
"ExpectedInstructionCount": 148,
"ExpectedInstructionCount": 147,
"Comment": [
"This block of code came from the settings screen when it loaded",
"It was originally at RIP: 0x14232cca0 and has been deobfuscated"
@@ -1173,7 +1173,8 @@
"ldr w11, [x6, #32]",
"mov w20, #0x13",
"mul w4, w5, w20",
"str w5, [x8, #104]",
"mov w21, w5",
"str w21, [x8, #104]",
"mov w21, #0x1000000",
"add w4, w4, w21",
"lsr w4, w4, #25",
@@ -1258,33 +1259,31 @@
"sub w17, w17, w14",
"mov w20, w17",
"str w20, [x4, #20]",
"ldr w20, [x8, #112]",
"mov x21, x4",
"mov x4, x20",
"ldr w4, [x8, #112]",
"lsl w15, w15, #26",
"sub w4, w20, w15",
"sub w4, w4, w15",
"lsl w7, w7, #25",
"mov w20, w4",
"str w20, [x21, #24]",
"str w20, [x10, #24]",
"ldr w4, [x8, #120]",
"sub w4, w4, w7",
"lsl w11, w11, #26",
"mov w20, w4",
"str w20, [x21, #28]",
"str w20, [x10, #28]",
"ldr w4, [x8]",
"sub w4, w4, w11",
"mov w20, w4",
"str w20, [x21, #32]",
"mov x4, x5",
"and w4, w5, #0xfe000000",
"str w20, [x10, #32]",
"mov w4, w5",
"and w4, w4, #0xfe000000",
"sub w5, w5, w4",
"mov w20, w5",
"str w20, [x21, #36]",
"str w20, [x10, #36]",
"mvn w27, w8",
"adds x26, x8, #0x18 (24)",
"mov x8, x26",
"ldr x29, [x26]",
"add x8, x26, #0x8 (8)",
"ldr x29, [x8]",
"add x8, x8, #0x8 (8)",
"ldr x19, [x8]",
"add x8, x8, #0x8 (8)",
"ldr x17, [x8]",
+34 -35
View File
@@ -16,7 +16,7 @@
"Comment": [],
"Instructions": {
"libnss3 sha": {
"ExpectedInstructionCount": 2367,
"ExpectedInstructionCount": 2366,
"Comment": [
"This block of code comes from libnss3 which causes panic spilling in FEX's RA.",
"This code is hit in steamwebhelper calling in to this function.",
@@ -206,7 +206,7 @@
"ldr q22, [x11, #32]",
"ldr q21, [x11, #48]",
"mov v19.16b, v16.16b",
"ext v19.16b, v18.16b, v16.16b, #8",
"ext v19.16b, v18.16b, v19.16b, #8",
"mov v18.d[1], v16.d[1]",
"mov w20, #0x1000",
"movk w20, #0x1, lsl #16",
@@ -230,7 +230,7 @@
"unimplemented (Unimplemented)",
"mov w21, v19.s[1]",
"mov w22, v19.s[0]",
"mov w23, v18.s[1]",
"mov w23, v20.s[1]",
"and w24, w21, w22",
"bic w23, w23, w21",
"eor w23, w24, w23",
@@ -240,11 +240,11 @@
"add w23, w23, w24",
"mov w24, v16.s[0]",
"add w23, w23, w24",
"mov w24, v18.s[0]",
"mov w24, v20.s[0]",
"add w23, w23, w24",
"mov w24, v19.s[3]",
"mov w25, v19.s[2]",
"mov w30, v18.s[3]",
"mov w30, v20.s[3]",
"and w18, w25, w30",
"orr w30, w25, w30",
"and w30, w24, w30",
@@ -254,7 +254,7 @@
"eor w18, w18, w24, ror #13",
"eor w18, w18, w24, ror #22",
"add w30, w30, w18",
"mov w18, v18.s[2]",
"mov w18, v20.s[2]",
"add w23, w23, w18",
"and w21, w23, w21",
"bic w22, w22, w23",
@@ -265,7 +265,7 @@
"add w21, w21, w22",
"mov w22, v16.s[1]",
"add w21, w21, w22",
"mov w22, v18.s[1]",
"mov w22, v20.s[1]",
"add w21, w21, w22",
"and w22, w24, w25",
"orr w24, w24, w25",
@@ -276,9 +276,9 @@
"eor w24, w24, w30, ror #13",
"eor w24, w24, w30, ror #22",
"add w22, w22, w24",
"mov w24, v18.s[3]",
"mov w24, v20.s[3]",
"add w21, w21, w24",
"mov v4.16b, v18.16b",
"mov v4.16b, v20.16b",
"mov v4.s[3], w22",
"mov v4.s[2], w30",
"mov v4.s[1], w21",
@@ -289,7 +289,7 @@
"tbl v16.16b, {v16.16b}, v4.16b",
"mov w21, v20.s[1]",
"mov w22, v20.s[0]",
"mov w23, v19.s[1]",
"mov w23, v17.s[1]",
"and w24, w21, w22",
"bic w23, w23, w21",
"eor w23, w24, w23",
@@ -299,11 +299,11 @@
"add w23, w23, w24",
"mov w24, v16.s[0]",
"add w23, w23, w24",
"mov w24, v19.s[0]",
"mov w24, v17.s[0]",
"add w23, w23, w24",
"mov w24, v20.s[3]",
"mov w25, v20.s[2]",
"mov w30, v19.s[3]",
"mov w30, v17.s[3]",
"and w18, w25, w30",
"orr w30, w25, w30",
"and w30, w24, w30",
@@ -313,7 +313,7 @@
"eor w18, w18, w24, ror #13",
"eor w18, w18, w24, ror #22",
"add w30, w30, w18",
"mov w18, v19.s[2]",
"mov w18, v17.s[2]",
"add w23, w23, w18",
"and w21, w23, w21",
"bic w22, w22, w23",
@@ -324,7 +324,7 @@
"add w21, w21, w22",
"mov w22, v16.s[1]",
"add w21, w21, w22",
"mov w22, v19.s[1]",
"mov w22, v17.s[1]",
"add w21, w21, w22",
"and w22, w24, w25",
"orr w24, w24, w25",
@@ -335,16 +335,16 @@
"eor w24, w24, w30, ror #13",
"eor w24, w24, w30, ror #22",
"add w22, w22, w24",
"mov w24, v19.s[3]",
"mov w24, v17.s[3]",
"add w21, w21, w24",
"mov v5.16b, v19.16b",
"mov v5.16b, v17.16b",
"mov v5.s[3], w22",
"mov v5.s[2], w30",
"mov v5.s[1], w21",
"mov v17.16b, v5.16b",
"mov v17.s[0], w23",
"mov v16.16b, v21.16b",
"ext v16.16b, v22.16b, v21.16b, #4",
"ext v16.16b, v22.16b, v16.16b, #4",
"add v24.4s, v24.4s, v16.4s",
"ldr q16, [x29, x20, sxtx]",
"mov w21, v21.s[2]",
@@ -499,7 +499,7 @@
"mov v17.16b, v5.16b",
"mov v17.s[0], w23",
"mov v16.16b, v24.16b",
"ext v16.16b, v21.16b, v24.16b, #4",
"ext v16.16b, v21.16b, v16.16b, #4",
"add v23.4s, v23.4s, v16.4s",
"ldr q16, [x29, x20, sxtx]",
"mov w21, v24.s[2]",
@@ -654,7 +654,7 @@
"mov v17.16b, v5.16b",
"mov v17.s[0], w23",
"mov v16.16b, v23.16b",
"ext v16.16b, v24.16b, v23.16b, #4",
"ext v16.16b, v24.16b, v16.16b, #4",
"add v22.4s, v22.4s, v16.4s",
"ldr q16, [x29, x20, sxtx]",
"mov w21, v23.s[2]",
@@ -809,7 +809,7 @@
"mov v17.16b, v5.16b",
"mov v17.s[0], w23",
"mov v16.16b, v22.16b",
"ext v16.16b, v23.16b, v22.16b, #4",
"ext v16.16b, v23.16b, v16.16b, #4",
"add v21.4s, v21.4s, v16.4s",
"ldr q16, [x29, x20, sxtx]",
"mov w21, v22.s[2]",
@@ -964,7 +964,7 @@
"mov v17.16b, v5.16b",
"mov v17.s[0], w23",
"mov v16.16b, v21.16b",
"ext v16.16b, v22.16b, v21.16b, #4",
"ext v16.16b, v22.16b, v16.16b, #4",
"add v24.4s, v24.4s, v16.4s",
"ldr q16, [x29, x20, sxtx]",
"mov w21, v21.s[2]",
@@ -1119,7 +1119,7 @@
"mov v17.16b, v5.16b",
"mov v17.s[0], w23",
"mov v16.16b, v24.16b",
"ext v16.16b, v21.16b, v24.16b, #4",
"ext v16.16b, v21.16b, v16.16b, #4",
"add v23.4s, v23.4s, v16.4s",
"ldr q16, [x29, x20, sxtx]",
"mov w21, v24.s[2]",
@@ -1274,7 +1274,7 @@
"mov v17.16b, v5.16b",
"mov v17.s[0], w23",
"mov v16.16b, v23.16b",
"ext v16.16b, v24.16b, v23.16b, #4",
"ext v16.16b, v24.16b, v16.16b, #4",
"add v22.4s, v22.4s, v16.4s",
"ldr q16, [x29, x20, sxtx]",
"mov w21, v23.s[2]",
@@ -1429,7 +1429,7 @@
"mov v17.16b, v5.16b",
"mov v17.s[0], w23",
"mov v16.16b, v22.16b",
"ext v16.16b, v23.16b, v22.16b, #4",
"ext v16.16b, v23.16b, v16.16b, #4",
"add v21.4s, v21.4s, v16.4s",
"ldr q16, [x29, x20, sxtx]",
"mov w21, v22.s[2]",
@@ -1584,7 +1584,7 @@
"mov v17.16b, v5.16b",
"mov v17.s[0], w23",
"mov v16.16b, v21.16b",
"ext v16.16b, v22.16b, v21.16b, #4",
"ext v16.16b, v22.16b, v16.16b, #4",
"add v24.4s, v24.4s, v16.4s",
"ldr q16, [x29, x20, sxtx]",
"mov w21, v21.s[2]",
@@ -1739,7 +1739,7 @@
"mov v17.16b, v5.16b",
"mov v17.s[0], w23",
"mov v16.16b, v24.16b",
"ext v16.16b, v21.16b, v24.16b, #4",
"ext v16.16b, v21.16b, v16.16b, #4",
"add v23.4s, v23.4s, v16.4s",
"ldr q16, [x29, x20, sxtx]",
"mov w21, v24.s[2]",
@@ -1894,7 +1894,7 @@
"mov v17.16b, v5.16b",
"mov v17.s[0], w23",
"mov v16.16b, v23.16b",
"ext v16.16b, v24.16b, v23.16b, #4",
"ext v16.16b, v24.16b, v16.16b, #4",
"add v22.4s, v22.4s, v16.4s",
"ldr q16, [x29, x20, sxtx]",
"mov w21, v23.s[2]",
@@ -2049,7 +2049,7 @@
"mov v17.16b, v5.16b",
"mov v17.s[0], w23",
"mov v16.16b, v22.16b",
"ext v16.16b, v23.16b, v22.16b, #4",
"ext v16.16b, v23.16b, v16.16b, #4",
"ldr q5, [x29, x20, sxtx]",
"add v23.4s, v23.4s, v5.4s",
"add v21.4s, v21.4s, v16.4s",
@@ -2219,7 +2219,7 @@
"eor w23, w23, w20, ror #11",
"eor w23, w23, w20, ror #25",
"add w22, w22, w23",
"mov w23, v23.s[0]",
"mov w23, v16.s[0]",
"add w22, w22, w23",
"mov w23, v20.s[0]",
"add w22, w22, w23",
@@ -2244,7 +2244,7 @@
"eor w21, w21, w22, ror #11",
"eor w21, w21, w22, ror #25",
"add w20, w20, w21",
"mov w21, v23.s[1]",
"mov w21, v16.s[1]",
"add w20, w20, w21",
"mov w21, v20.s[1]",
"add w20, w20, w21",
@@ -2333,7 +2333,7 @@
"eor w23, w23, w20, ror #11",
"eor w23, w23, w20, ror #25",
"add w22, w22, w23",
"mov w23, v22.s[0]",
"mov w23, v16.s[0]",
"add w22, w22, w23",
"mov w23, v20.s[0]",
"add w22, w22, w23",
@@ -2358,7 +2358,7 @@
"eor w21, w21, w22, ror #11",
"eor w21, w21, w22, ror #25",
"add w20, w20, w21",
"mov w21, v22.s[1]",
"mov w21, v16.s[1]",
"add w20, w20, w21",
"mov w21, v20.s[1]",
"add w20, w20, w21",
@@ -2447,7 +2447,7 @@
"eor w23, w23, w20, ror #11",
"eor w23, w23, w20, ror #25",
"add w22, w22, w23",
"mov w23, v21.s[0]",
"mov w23, v16.s[0]",
"add w22, w22, w23",
"mov w23, v20.s[0]",
"add w22, w22, w23",
@@ -2472,7 +2472,7 @@
"eor w21, w21, w22, ror #11",
"eor w21, w21, w22, ror #25",
"add w20, w20, w21",
"mov w21, v21.s[1]",
"mov w21, v16.s[1]",
"add w20, w20, w21",
"mov w21, v20.s[1]",
"add w20, w20, w21",
@@ -2555,7 +2555,6 @@
"tbl v20.16b, {v20.16b}, v2.16b",
"tbl v17.16b, {v17.16b}, v3.16b",
"mov v16.16b, v17.16b",
"mov v16.16b, v17.16b",
"mov v16.d[1], v20.d[1]",
"ext v20.16b, v17.16b, v20.16b, #8",
"str q16, [x11, #256]",
@@ -95,9 +95,9 @@
"ExpectedArm64ASM": [
"adds x4, x4, x7",
"cset w20, hs",
"mov x27, x4",
"adds x26, x4, #0x1 (1)",
"rmif x20, #63, #nzCv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -124,9 +124,9 @@
"subs x4, x4, x7",
"cfinv",
"cset w20, hs",
"mov x27, x4",
"subs x26, x4, #0x1 (1)",
"rmif x20, #63, #nzCv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -263,15 +263,16 @@
]
},
"AND use only PF": {
"ExpectedInstructionCount": 8,
"ExpectedInstructionCount": 9,
"x86Insts": [
"and eax, ebx",
"setp cl",
"test cl, cl"
],
"ExpectedArm64ASM": [
"and w4, w4, w7",
"eor w20, w4, w4, lsr #4",
"and w26, w4, w7",
"mov x4, x26",
"eor w20, w26, w26, lsr #4",
"eor w20, w20, w20, lsr #2",
"eon w20, w20, w20, lsr #1",
"and x20, x20, #0x1",
@@ -27,7 +27,7 @@
"mov w4, #0x1",
"ldaddal x4, x4, [x5]",
"mov x6, x4",
"and w6, w4, #0x1f",
"and w6, w6, #0x1f",
"add x6, x6, #0x1 (1)",
"lsl x6, x6, #6",
"eor w27, w6, w5",
@@ -81,11 +81,11 @@
"mov w10, w5",
"mov x6, x9",
"mov x4, x7",
"ldr s18, [x9]",
"add x4, x7, #0x20 (32)",
"ldr s18, [x6]",
"add x4, x4, #0x20 (32)",
"fmul s0, s18, s16",
"mov v18.s[0], v0.s[0]",
"add x6, x9, #0x20 (32)",
"add x6, x6, #0x20 (32)",
"ldur s2, [x4, #-32]",
"fadd s0, s18, s2",
"mov v18.s[0], v0.s[0]",
@@ -139,9 +139,9 @@
"fadd s0, s18, s2",
"mov v18.s[0], v0.s[0]",
"stur s18, [x4, #-4]",
"mov x27, x10",
"subs w26, w10, #0x1 (1)",
"cfinv",
"mov x27, x10",
"mov x10, x26"
]
},
@@ -168,7 +168,7 @@
]
},
"bytemark data xor loop": {
"ExpectedInstructionCount": 15,
"ExpectedInstructionCount": 13,
"Comment": [
"Saw this in bytemark"
],
@@ -187,17 +187,15 @@
"mov x6, x4",
"mov x5, x4",
"mov x19, x10",
"add x20, x4, #0x1 (1)",
"mov x21, x4",
"mov x4, x20",
"lsr x6, x21, #6",
"and w5, w21, #0x3f",
"lsl x19, x10, x5",
"ldr x21, [x7, x6, sxtx #3]",
"eor x21, x21, x19",
"str x21, [x7, x6, sxtx #3]",
"eor w27, w11, w20",
"subs x26, x11, x20",
"add x4, x4, #0x1 (1)",
"lsr x6, x6, #6",
"and w5, w5, #0x3f",
"lsl x19, x19, x5",
"ldr x20, [x7, x6, sxtx #3]",
"eor x20, x20, x19",
"str x20, [x7, x6, sxtx #3]",
"eor w27, w11, w4",
"subs x26, x11, x4",
"cfinv"
]
},
@@ -215,7 +213,7 @@
"ExpectedArm64ASM": [
"ldr x17, [x10, x13, sxtx #3]",
"mov x15, x13",
"orr x15, x13, #0x1",
"orr x15, x15, #0x1",
"ldr x20, [x10, x15, sxtx #3]",
"eor w27, w17, w20",
"subs x26, x17, x20",
File diff suppressed because it is too large. Load diff
@@ -58,10 +58,10 @@
"mov w10, w5",
"mov x6, x9",
"mov x4, x7",
"ldr s18, [x9]",
"add x4, x7, #0x20 (32)",
"ldr s18, [x6]",
"add x4, x4, #0x20 (32)",
"fmul s18, s18, s16",
"add x6, x9, #0x20 (32)",
"add x6, x6, #0x20 (32)",
"ldur s2, [x4, #-32]",
"fadd s18, s18, s2",
"stur s18, [x4, #-32]",
@@ -100,9 +100,9 @@
"ldur s2, [x4, #-4]",
"fadd s18, s18, s2",
"stur s18, [x4, #-4]",
"mov x27, x10",
"subs w26, w10, #0x1 (1)",
"cfinv",
"mov x27, x10",
"mov x10, x26"
]
}
+117 -109
View File
@@ -105,35 +105,39 @@
]
},
"add al, 1": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 8,
"Comment": "0x04",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #24",
"cmn w0, w20, lsl #24",
"add w26, w4, #0x1 (1)",
"bfxil x4, x26, #0, #8"
"mov x20, x4",
"bfxil x20, x26, #0, #8",
"mov x27, x4",
"mov x4, x20"
]
},
"add ax, 1": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 8,
"Comment": "0x05",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #16",
"cmn w0, w20, lsl #16",
"add w26, w4, #0x1 (1)",
"bfxil x4, x26, #0, #16"
"mov x20, x4",
"bfxil x20, x26, #0, #16",
"mov x27, x4",
"mov x4, x20"
]
},
"add eax, 1": {
"ExpectedInstructionCount": 3,
"Comment": "0x05",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds w26, w4, #0x1 (1)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -141,8 +145,8 @@
"ExpectedInstructionCount": 3,
"Comment": "0x05",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds x26, x4, #0x1 (1)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -486,10 +490,9 @@
]
},
"adc al, 1": {
"ExpectedInstructionCount": 14,
"ExpectedInstructionCount": 16,
"Comment": "0x14",
"ExpectedArm64ASM": [
"mov x27, x4",
"mov w20, #0x1",
"adc w20, wzr, w20",
"add w21, w4, w20",
@@ -502,14 +505,16 @@
"eor w21, w26, w4",
"bic w20, w21, w20",
"rmif x20, #7, #nzcV",
"bfxil x4, x26, #0, #8"
"mov x20, x4",
"bfxil x20, x26, #0, #8",
"mov x27, x4",
"mov x4, x20"
]
},
"adc ax, 1": {
"ExpectedInstructionCount": 14,
"ExpectedInstructionCount": 16,
"Comment": "0x15",
"ExpectedArm64ASM": [
"mov x27, x4",
"mov w20, #0x1",
"adc w20, wzr, w20",
"add w21, w4, w20",
@@ -522,7 +527,10 @@
"eor w21, w26, w4",
"bic w20, w21, w20",
"rmif x20, #15, #nzcV",
"bfxil x4, x26, #0, #16"
"mov x20, x4",
"bfxil x20, x26, #0, #16",
"mov x27, x4",
"mov x4, x20"
]
},
"adc eax, 1": {
@@ -530,8 +538,8 @@
"Comment": "0x15",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"adcs w26, w4, w20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -540,8 +548,8 @@
"Comment": "0x15",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"adcs x26, x4, x20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -746,10 +754,9 @@
]
},
"sbb al, 1": {
"ExpectedInstructionCount": 15,
"ExpectedInstructionCount": 17,
"Comment": "0x1C",
"ExpectedArm64ASM": [
"mov x27, x4",
"uxtb w20, w4",
"mov w21, #0x1",
"adc w21, wzr, w21",
@@ -763,14 +770,16 @@
"eor w20, w26, w20",
"and w20, w20, w21",
"rmif x20, #7, #nzcV",
"bfxil x4, x26, #0, #8"
"mov x20, x4",
"bfxil x20, x26, #0, #8",
"mov x27, x4",
"mov x4, x20"
]
},
"sbb ax, 1": {
"ExpectedInstructionCount": 15,
"ExpectedInstructionCount": 17,
"Comment": "0x1D",
"ExpectedArm64ASM": [
"mov x27, x4",
"uxth w20, w4",
"mov w21, #0x1",
"adc w21, wzr, w21",
@@ -784,7 +793,10 @@
"eor w20, w26, w20",
"and w20, w20, w21",
"rmif x20, #15, #nzcV",
"bfxil x4, x26, #0, #16"
"mov x20, x4",
"bfxil x20, x26, #0, #16",
"mov x27, x4",
"mov x4, x20"
]
},
"sbb eax, 1": {
@@ -792,10 +804,10 @@
"Comment": "0x1D",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"cfinv",
"sbcs w26, w4, w20",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -804,10 +816,10 @@
"Comment": "0x1D",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"cfinv",
"sbcs x26, x4, x20",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -1128,38 +1140,42 @@
]
},
"sub al, 1": {
"ExpectedInstructionCount": 7,
"ExpectedInstructionCount": 9,
"Comment": "0x2C",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #24",
"cmp w0, w20, lsl #24",
"sub w26, w4, #0x1 (1)",
"cfinv",
"bfxil x4, x26, #0, #8"
"mov x20, x4",
"bfxil x20, x26, #0, #8",
"mov x27, x4",
"mov x4, x20"
]
},
"sub ax, 1": {
"ExpectedInstructionCount": 7,
"ExpectedInstructionCount": 9,
"Comment": "0x2D",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #16",
"cmp w0, w20, lsl #16",
"sub w26, w4, #0x1 (1)",
"cfinv",
"bfxil x4, x26, #0, #16"
"mov x20, x4",
"bfxil x20, x26, #0, #16",
"mov x27, x4",
"mov x4, x20"
]
},
"sub eax, 1": {
"ExpectedInstructionCount": 4,
"Comment": "0x2D",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs w26, w4, #0x1 (1)",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -1167,9 +1183,9 @@
"ExpectedInstructionCount": 4,
"Comment": "0x2D",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x1 (1)",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -1475,11 +1491,11 @@
"Comment": "0x3C",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #24",
"cmp w0, w20, lsl #24",
"sub w26, w4, #0x1 (1)",
"cfinv"
"cfinv",
"mov x27, x4"
]
},
"cmp ax, 1": {
@@ -1487,29 +1503,29 @@
"Comment": "0x3D",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #16",
"cmp w0, w20, lsl #16",
"sub w26, w4, #0x1 (1)",
"cfinv"
"cfinv",
"mov x27, x4"
]
},
"cmp eax, 1": {
"ExpectedInstructionCount": 3,
"Comment": "0x3D",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs w26, w4, #0x1 (1)",
"cfinv"
"cfinv",
"mov x27, x4"
]
},
"cmp rax, 1": {
"ExpectedInstructionCount": 3,
"Comment": "0x3D",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x1 (1)",
"cfinv"
"cfinv",
"mov x27, x4"
]
},
"cmp al, -1": {
@@ -1773,24 +1789,24 @@
"strb w20, [x28, #985]",
"ubfx w20, w27, #10, #1",
"sub x20, x21, x20, lsl #1",
"strb w20, [x28, #986]",
"rmif x27, #11, #nzcV",
"ubfx w20, w27, #12, #1",
"strb w20, [x28, #988]",
"ubfx w20, w27, #14, #1",
"strb w20, [x28, #990]",
"ubfx w20, w27, #16, #1",
"strb w20, [x28, #992]",
"ubfx w20, w27, #17, #1",
"strb w20, [x28, #993]",
"ubfx w20, w27, #18, #1",
"strb w20, [x28, #994]",
"ubfx w20, w27, #19, #1",
"strb w20, [x28, #995]",
"ubfx w20, w27, #20, #1",
"strb w20, [x28, #996]",
"ubfx w20, w27, #21, #1",
"strb w20, [x28, #997]"
"ubfx w21, w27, #12, #1",
"strb w21, [x28, #988]",
"ubfx w21, w27, #14, #1",
"strb w21, [x28, #990]",
"ubfx w21, w27, #16, #1",
"strb w21, [x28, #992]",
"ubfx w21, w27, #17, #1",
"strb w21, [x28, #993]",
"ubfx w21, w27, #18, #1",
"strb w21, [x28, #994]",
"ubfx w21, w27, #19, #1",
"strb w21, [x28, #995]",
"ubfx w21, w27, #20, #1",
"strb w21, [x28, #996]",
"ubfx w21, w27, #21, #1",
"strb w21, [x28, #997]",
"strb w20, [x28, #986]"
]
},
"sahf": {
@@ -1900,10 +1916,10 @@
]
},
"repz cmpsb": {
"ExpectedInstructionCount": 27,
"ExpectedInstructionCount": 26,
"Comment": "0xa6",
"ExpectedArm64ASM": [
"cbz x5, #+0x6c",
"cbz x5, #+0x68",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -1924,19 +1940,18 @@
"ccmp x27, x26, #nzcv, ne",
"b.eq #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"lsl w0, w21, #24",
"lsl w0, w27, #24",
"cmp w0, w26, lsl #24",
"sub w26, w21, w26",
"cfinv"
"sub w26, w27, w26",
"cfinv",
"mov x27, x20"
]
},
"repz cmpsw": {
"ExpectedInstructionCount": 27,
"ExpectedInstructionCount": 26,
"Comment": "0xa7",
"ExpectedArm64ASM": [
"cbz x5, #+0x6c",
"cbz x5, #+0x68",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -1957,19 +1972,18 @@
"ccmp x27, x26, #nzcv, ne",
"b.eq #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"lsl w0, w21, #16",
"lsl w0, w27, #16",
"cmp w0, w26, lsl #16",
"sub w26, w21, w26",
"cfinv"
"sub w26, w27, w26",
"cfinv",
"mov x27, x20"
]
},
"repz cmpsd": {
"ExpectedInstructionCount": 25,
"ExpectedInstructionCount": 24,
"Comment": "0xa7",
"ExpectedArm64ASM": [
"cbz x5, #+0x64",
"cbz x5, #+0x60",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -1990,17 +2004,16 @@
"ccmp x27, x26, #nzcv, ne",
"b.eq #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"subs w26, w21, w26",
"cfinv"
"subs w26, w27, w26",
"cfinv",
"mov x27, x20"
]
},
"repz cmpsq": {
"ExpectedInstructionCount": 25,
"ExpectedInstructionCount": 24,
"Comment": "0xa7",
"ExpectedArm64ASM": [
"cbz x5, #+0x64",
"cbz x5, #+0x60",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -2021,17 +2034,16 @@
"ccmp x27, x26, #nzcv, ne",
"b.eq #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"subs x26, x21, x26",
"cfinv"
"subs x26, x27, x26",
"cfinv",
"mov x27, x20"
]
},
"repnz cmpsb": {
"ExpectedInstructionCount": 27,
"ExpectedInstructionCount": 26,
"Comment": "0xa6",
"ExpectedArm64ASM": [
"cbz x5, #+0x6c",
"cbz x5, #+0x68",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -2052,19 +2064,18 @@
"ccmp x27, x26, #nZcv, ne",
"b.ne #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"lsl w0, w21, #24",
"lsl w0, w27, #24",
"cmp w0, w26, lsl #24",
"sub w26, w21, w26",
"cfinv"
"sub w26, w27, w26",
"cfinv",
"mov x27, x20"
]
},
"repnz cmpsw": {
"ExpectedInstructionCount": 27,
"ExpectedInstructionCount": 26,
"Comment": "0xa7",
"ExpectedArm64ASM": [
"cbz x5, #+0x6c",
"cbz x5, #+0x68",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -2085,19 +2096,18 @@
"ccmp x27, x26, #nZcv, ne",
"b.ne #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"lsl w0, w21, #16",
"lsl w0, w27, #16",
"cmp w0, w26, lsl #16",
"sub w26, w21, w26",
"cfinv"
"sub w26, w27, w26",
"cfinv",
"mov x27, x20"
]
},
"repnz cmpsd": {
"ExpectedInstructionCount": 25,
"ExpectedInstructionCount": 24,
"Comment": "0xa7",
"ExpectedArm64ASM": [
"cbz x5, #+0x64",
"cbz x5, #+0x60",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -2118,17 +2128,16 @@
"ccmp x27, x26, #nZcv, ne",
"b.ne #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"subs w26, w21, w26",
"cfinv"
"subs w26, w27, w26",
"cfinv",
"mov x27, x20"
]
},
"repnz cmpsq": {
"ExpectedInstructionCount": 25,
"ExpectedInstructionCount": 24,
"Comment": "0xa7",
"ExpectedArm64ASM": [
"cbz x5, #+0x64",
"cbz x5, #+0x60",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -2149,10 +2158,9 @@
"ccmp x27, x26, #nZcv, ne",
"b.ne #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"subs x26, x21, x26",
"cfinv"
"subs x26, x27, x26",
"cfinv",
"mov x27, x20"
]
},
"test al, 1": {
@@ -13,15 +13,17 @@
},
"Instructions": {
"add al, 1": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 8,
"Comment": "GROUP1 0x80 /0",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #24",
"cmn w0, w20, lsl #24",
"add w26, w4, #0x1 (1)",
"bfxil x4, x26, #0, #8"
"mov x20, x4",
"bfxil x20, x26, #0, #8",
"mov x27, x4",
"mov x4, x20"
]
},
"or al, 1": {
@@ -34,10 +36,9 @@
]
},
"adc al, 1": {
"ExpectedInstructionCount": 14,
"ExpectedInstructionCount": 16,
"Comment": "GROUP1 0x80 /2",
"ExpectedArm64ASM": [
"mov x27, x4",
"mov w20, #0x1",
"adc w20, wzr, w20",
"add w21, w4, w20",
@@ -50,14 +51,16 @@
"eor w21, w26, w4",
"bic w20, w21, w20",
"rmif x20, #7, #nzcV",
"bfxil x4, x26, #0, #8"
"mov x20, x4",
"bfxil x20, x26, #0, #8",
"mov x27, x4",
"mov x4, x20"
]
},
"sbb al, 1": {
"ExpectedInstructionCount": 15,
"ExpectedInstructionCount": 17,
"Comment": "GROUP1 0x80 /3",
"ExpectedArm64ASM": [
"mov x27, x4",
"uxtb w20, w4",
"mov w21, #0x1",
"adc w21, wzr, w21",
@@ -71,7 +74,10 @@
"eor w20, w26, w20",
"and w20, w20, w21",
"rmif x20, #7, #nzcV",
"bfxil x4, x26, #0, #8"
"mov x20, x4",
"bfxil x20, x26, #0, #8",
"mov x27, x4",
"mov x4, x20"
]
},
"and al, 1": {
@@ -84,16 +90,18 @@
]
},
"sub al, 1": {
"ExpectedInstructionCount": 7,
"ExpectedInstructionCount": 9,
"Comment": "GROUP1 0x80 /5",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #24",
"cmp w0, w20, lsl #24",
"sub w26, w4, #0x1 (1)",
"cfinv",
"bfxil x4, x26, #0, #8"
"mov x20, x4",
"bfxil x20, x26, #0, #8",
"mov x27, x4",
"mov x4, x20"
]
},
"xor al, 1": {
@@ -110,11 +118,11 @@
"Comment": "GROUP1 0x80 /7",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #24",
"cmp w0, w20, lsl #24",
"sub w26, w4, #0x1 (1)",
"cfinv"
"cfinv",
"mov x27, x4"
]
},
"add al, -1": {
@@ -223,23 +231,25 @@
]
},
"add ax, 256": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 8,
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov w20, #0x100",
"mov x27, x4",
"lsl w0, w4, #16",
"cmn w0, w20, lsl #16",
"add w26, w4, #0x100 (256)",
"bfxil x4, x26, #0, #16"
"mov x20, x4",
"bfxil x20, x26, #0, #16",
"mov x27, x4",
"mov x4, x20"
]
},
"add eax, 256": {
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds w26, w4, #0x100 (256)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -247,8 +257,8 @@
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds x26, x4, #0x100 (256)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -275,8 +285,8 @@
"Comment": "GROUP1 0x81 /2",
"ExpectedArm64ASM": [
"mov w20, #0x100",
"mov x27, x4",
"adcs w26, w4, w20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -285,8 +295,8 @@
"Comment": "GROUP1 0x81 /2",
"ExpectedArm64ASM": [
"mov w20, #0x100",
"mov x27, x4",
"adcs x26, x4, x20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -295,10 +305,10 @@
"Comment": "GROUP1 0x81 /3",
"ExpectedArm64ASM": [
"mov w20, #0x100",
"mov x27, x4",
"cfinv",
"sbcs w26, w4, w20",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -307,10 +317,10 @@
"Comment": "GROUP1 0x81 /3",
"ExpectedArm64ASM": [
"mov w20, #0x100",
"mov x27, x4",
"cfinv",
"sbcs x26, x4, x20",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -334,9 +344,9 @@
"ExpectedInstructionCount": 4,
"Comment": "GROUP1 0x81 /5",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs w26, w4, #0x100 (256)",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -344,9 +354,9 @@
"ExpectedInstructionCount": 4,
"Comment": "GROUP1 0x81 /5",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x100 (256)",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -372,30 +382,32 @@
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x81 /7",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs w26, w4, #0x100 (256)",
"cfinv"
"cfinv",
"mov x27, x4"
]
},
"cmp rax, 256": {
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x81 /7",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x100 (256)",
"cfinv"
"cfinv",
"mov x27, x4"
]
},
"add ax, -256": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 8,
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov w20, #0xff00",
"mov x27, x4",
"lsl w0, w4, #16",
"cmn w0, w20, lsl #16",
"add w26, w4, w20",
"bfxil x4, x26, #0, #16"
"mov x20, x4",
"bfxil x20, x26, #0, #16",
"mov x27, x4",
"mov x4, x20"
]
},
"add eax, -256": {
@@ -403,8 +415,8 @@
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov w20, #0xffffff00",
"mov x27, x4",
"adds w26, w4, w20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -412,8 +424,8 @@
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x100 (256)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -440,8 +452,8 @@
"Comment": "GROUP1 0x81 /2",
"ExpectedArm64ASM": [
"mov w20, #0xffffff00",
"mov x27, x4",
"adcs w26, w4, w20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -450,8 +462,8 @@
"Comment": "GROUP1 0x81 /2",
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffff00",
"mov x27, x4",
"adcs x26, x4, x20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -460,10 +472,10 @@
"Comment": "GROUP1 0x81 /3",
"ExpectedArm64ASM": [
"mov w20, #0xffffff00",
"mov x27, x4",
"cfinv",
"sbcs w26, w4, w20",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -472,10 +484,10 @@
"Comment": "GROUP1 0x81 /3",
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffff00",
"mov x27, x4",
"cfinv",
"sbcs x26, x4, x20",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -500,9 +512,9 @@
"Comment": "GROUP1 0x81 /5",
"ExpectedArm64ASM": [
"mov w20, #0xffffff00",
"mov x27, x4",
"subs w26, w4, w20",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -510,9 +522,9 @@
"ExpectedInstructionCount": 4,
"Comment": "GROUP1 0x81 /5",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds x26, x4, #0x100 (256)",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -539,38 +551,40 @@
"Comment": "GROUP1 0x81 /7",
"ExpectedArm64ASM": [
"mov w20, #0xffffff00",
"mov x27, x4",
"subs w26, w4, w20",
"cfinv"
"cfinv",
"mov x27, x4"
]
},
"cmp rax, -256": {
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x81 /7",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds x26, x4, #0x100 (256)",
"cfinv"
"cfinv",
"mov x27, x4"
]
},
"add ax, 1": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 8,
"Comment": "GROUP1 0x83 /0",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #16",
"cmn w0, w20, lsl #16",
"add w26, w4, #0x1 (1)",
"bfxil x4, x26, #0, #16"
"mov x20, x4",
"bfxil x20, x26, #0, #16",
"mov x27, x4",
"mov x4, x20"
]
},
"add eax, 1": {
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x83 /0",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds w26, w4, #0x1 (1)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -578,8 +592,8 @@
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x83 /0",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds x26, x4, #0x1 (1)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -606,8 +620,8 @@
"Comment": "GROUP1 0x83 /2",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"adcs w26, w4, w20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -616,8 +630,8 @@
"Comment": "GROUP1 0x83 /2",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"adcs x26, x4, x20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -626,10 +640,10 @@
"Comment": "GROUP1 0x83 /3",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"cfinv",
"sbcs w26, w4, w20",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -638,10 +652,10 @@
"Comment": "GROUP1 0x83 /3",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"cfinv",
"sbcs x26, x4, x20",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -665,9 +679,9 @@
"ExpectedInstructionCount": 4,
"Comment": "GROUP1 0x83 /5",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs w26, w4, #0x1 (1)",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -675,9 +689,9 @@
"ExpectedInstructionCount": 4,
"Comment": "GROUP1 0x83 /5",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x1 (1)",
"cfinv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -703,18 +717,18 @@
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x83 /7",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs w26, w4, #0x1 (1)",
"cfinv"
"cfinv",
"mov x27, x4"
]
},
"cmp rax, 1": {
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x83 /7",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x1 (1)",
"cfinv"
"cfinv",
"mov x27, x4"
]
},
"add ax, -1": {
@@ -2047,14 +2061,16 @@
]
},
"neg bl": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 7,
"Comment": "GROUP2 0xf6 /3",
"ExpectedArm64ASM": [
"mov x27, x7",
"cmp wzr, w7, lsl #24",
"neg w26, w7",
"cfinv",
"bfxil x7, x26, #0, #8"
"mov x20, x7",
"bfxil x20, x26, #0, #8",
"mov x27, x7",
"mov x7, x20"
]
},
"mul bl": {
@@ -2161,23 +2177,25 @@
]
},
"neg bx": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 7,
"Comment": "GROUP2 0xf7 /2",
"ExpectedArm64ASM": [
"mov x27, x7",
"cmp wzr, w7, lsl #16",
"neg w26, w7",
"cfinv",
"bfxil x7, x26, #0, #16"
"mov x20, x7",
"bfxil x20, x26, #0, #16",
"mov x27, x7",
"mov x7, x20"
]
},
"neg ebx": {
"ExpectedInstructionCount": 4,
"Comment": "GROUP2 0xf7 /2",
"ExpectedArm64ASM": [
"mov x27, x7",
"negs w26, w7",
"cfinv",
"mov x27, x7",
"mov x7, x26"
]
},
@@ -2185,9 +2203,9 @@
"ExpectedInstructionCount": 4,
"Comment": "GROUP2 0xf7 /2",
"ExpectedArm64ASM": [
"mov x27, x7",
"negs x26, x7",
"cfinv",
"mov x27, x7",
"mov x7, x26"
]
},
@@ -2224,9 +2242,9 @@
"ExpectedArm64ASM": [
"mul x20, x7, x4",
"umulh x6, x7, x4",
"mov x4, x20",
"cmp x6, #0x0 (0)",
"ccmn xzr, #0, #nzCV, eq"
"ccmn xzr, #0, #nzCV, eq",
"mov x4, x20"
]
},
"imul bx": {
@@ -2330,9 +2348,9 @@
"Comment": "GROUP4 0xfe /0",
"ExpectedArm64ASM": [
"cset w20, hs",
"mov x27, x4",
"adds w26, w4, #0x1 (1)",
"rmif x20, #63, #nzCv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -2341,9 +2359,9 @@
"Comment": "GROUP4 0xfe /0",
"ExpectedArm64ASM": [
"cset w20, hs",
"mov x27, x4",
"adds x26, x4, #0x1 (1)",
"rmif x20, #63, #nzCv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -2364,9 +2382,9 @@
"Comment": "GROUP4 0xfe /1",
"ExpectedArm64ASM": [
"cset w20, hs",
"mov x27, x4",
"subs w26, w4, #0x1 (1)",
"rmif x20, #63, #nzCv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -2375,9 +2393,9 @@
"Comment": "GROUP4 0xfe /1",
"ExpectedArm64ASM": [
"cset w20, hs",
"mov x27, x4",
"subs x26, x4, #0x1 (1)",
"rmif x20, #63, #nzCv",
"mov x27, x4",
"mov x4, x26"
]
}
@@ -197,9 +197,9 @@
"Comment": "0x40",
"ExpectedArm64ASM": [
"cset w20, hs",
"mov x27, x4",
"adds w26, w4, #0x1 (1)",
"rmif x20, #63, #nzCv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -234,9 +234,9 @@
"Comment": "0x48",
"ExpectedArm64ASM": [
"cset w20, hs",
"mov x27, x4",
"subs w26, w4, #0x1 (1)",
"rmif x20, #63, #nzCv",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -1057,13 +1057,15 @@
]
},
"cmpxchg rcx, rbx": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 7,
"ExpectedArm64ASM": [
"eor w27, w4, w5",
"subs x26, x4, x5",
"cfinv",
"mov x4, x5",
"csel x5, x7, x5, eq"
"csel x20, x7, x5, eq",
"mov x21, x5",
"mov x5, x20",
"mov x4, x21"
]
},
"cmpxchg [rax], rbx": {
@@ -1280,23 +1280,14 @@
"strb w23, [x28, #1018]",
"strb w20, [x28, #1022]",
"ldrb w20, [x4, #4]",
"strb w20, [x28, #1298]",
"ldr q2, [x4, #32]",
"str q2, [x28, #1040]",
"ldr q2, [x4, #48]",
"str q2, [x28, #1056]",
"ldr q2, [x4, #64]",
"str q2, [x28, #1072]",
"ldr q2, [x4, #80]",
"str q2, [x28, #1088]",
"ldr q2, [x4, #96]",
"str q2, [x28, #1104]",
"ldr q2, [x4, #112]",
"str q2, [x28, #1120]",
"ldr q2, [x4, #128]",
"str q2, [x28, #1136]",
"ldr q2, [x4, #144]",
"str q2, [x28, #1152]",
"ldr q3, [x4, #48]",
"ldr q4, [x4, #64]",
"ldr q5, [x4, #80]",
"ldr q6, [x4, #96]",
"ldr q7, [x4, #112]",
"ldr q8, [x4, #128]",
"ldr q9, [x4, #144]",
"ldr q16, [x4, #160]",
"ldr q17, [x4, #176]",
"ldr q18, [x4, #192]",
@@ -1313,15 +1304,24 @@
"ldr q29, [x4, #368]",
"ldr q30, [x4, #384]",
"ldr q31, [x4, #400]",
"ldr w20, [x4, #24]",
"ubfx w20, w20, #13, #3",
"rbit w1, w20",
"ldr w21, [x4, #24]",
"ubfx w21, w21, #13, #3",
"rbit w1, w21",
"lsr w1, w1, #30",
"mrs x0, fpcr",
"bfi x0, x1, #22, #2",
"lsr x1, x20, #2",
"lsr x1, x21, #2",
"bfi x0, x1, #24, #1",
"msr fpcr, x0"
"msr fpcr, x0",
"strb w20, [x28, #1298]",
"str q9, [x28, #1152]",
"str q8, [x28, #1136]",
"str q7, [x28, #1120]",
"str q6, [x28, #1104]",
"str q5, [x28, #1088]",
"str q4, [x28, #1072]",
"str q3, [x28, #1056]",
"str q2, [x28, #1040]"
]
},
"rdgsbase eax": {
@@ -1513,9 +1513,10 @@
]
},
"xrstor [rax]": {
"ExpectedInstructionCount": 159,
"ExpectedInstructionCount": 165,
"Comment": "GROUP15 0x0F 0xAE /5",
"ExpectedArm64ASM": [
"sub sp, sp, #0x40 (64)",
"ldr x20, [x4, #512]",
"ubfx x20, x20, #0, #1",
"cbnz x20, #+0x8",
@@ -1534,23 +1535,23 @@
"strb w23, [x28, #1018]",
"strb w20, [x28, #1022]",
"ldrb w20, [x4, #4]",
"strb w20, [x28, #1298]",
"ldr q2, [x4, #32]",
"ldr q3, [x4, #48]",
"ldr q4, [x4, #64]",
"ldr q5, [x4, #80]",
"ldr q6, [x4, #96]",
"ldr q7, [x4, #112]",
"ldr q8, [x4, #128]",
"ldr q9, [x4, #144]",
"strb w20, [x28, #1298]",
"str q9, [x28, #1152]",
"str q8, [x28, #1136]",
"str q7, [x28, #1120]",
"str q6, [x28, #1104]",
"str q5, [x28, #1088]",
"str q4, [x28, #1072]",
"str q3, [x28, #1056]",
"str q2, [x28, #1040]",
"ldr q2, [x4, #48]",
"str q2, [x28, #1056]",
"ldr q2, [x4, #64]",
"str q2, [x28, #1072]",
"ldr q2, [x4, #80]",
"str q2, [x28, #1088]",
"ldr q2, [x4, #96]",
"str q2, [x28, #1104]",
"ldr q2, [x4, #112]",
"str q2, [x28, #1120]",
"ldr q2, [x4, #128]",
"str q2, [x28, #1136]",
"ldr q2, [x4, #144]",
"str q2, [x28, #1152]",
"b #+0x4c",
"mov w20, #0x0",
"mov w21, #0x37f",
@@ -1560,16 +1561,16 @@
"strb w20, [x28, #1017]",
"strb w20, [x28, #1018]",
"strb w20, [x28, #1022]",
"strb w20, [x28, #1298]",
"movi v2.2d, #0x0",
"str q2, [x28, #1040]",
"str q2, [x28, #1056]",
"str q2, [x28, #1072]",
"str q2, [x28, #1088]",
"str q2, [x28, #1104]",
"str q2, [x28, #1120]",
"str q2, [x28, #1136]",
"strb w20, [x28, #1298]",
"str q2, [x28, #1152]",
"str q2, [x28, #1136]",
"str q2, [x28, #1120]",
"str q2, [x28, #1104]",
"str q2, [x28, #1088]",
"str q2, [x28, #1072]",
"str q2, [x28, #1056]",
"str q2, [x28, #1040]",
"ldr x20, [x4, #512]",
"ubfx x20, x20, #1, #1",
"cbnz x20, #+0x8",
@@ -1591,76 +1592,80 @@
"ldr q30, [x4, #384]",
"ldr q31, [x4, #400]",
"b #+0x44",
"movi v16.2d, #0x0",
"mov v17.16b, v16.16b",
"mov v18.16b, v16.16b",
"mov v19.16b, v16.16b",
"mov v20.16b, v16.16b",
"mov v21.16b, v16.16b",
"mov v22.16b, v16.16b",
"mov v23.16b, v16.16b",
"mov v24.16b, v16.16b",
"mov v25.16b, v16.16b",
"mov v26.16b, v16.16b",
"mov v27.16b, v16.16b",
"mov v28.16b, v16.16b",
"mov v29.16b, v16.16b",
"mov v30.16b, v16.16b",
"mov v31.16b, v16.16b",
"movi v31.2d, #0x0",
"mov v30.16b, v31.16b",
"mov v29.16b, v31.16b",
"mov v28.16b, v31.16b",
"mov v27.16b, v31.16b",
"mov v26.16b, v31.16b",
"mov v25.16b, v31.16b",
"mov v24.16b, v31.16b",
"mov v23.16b, v31.16b",
"mov v22.16b, v31.16b",
"mov v21.16b, v31.16b",
"mov v20.16b, v31.16b",
"mov v19.16b, v31.16b",
"mov v18.16b, v31.16b",
"mov v17.16b, v31.16b",
"mov v16.16b, v31.16b",
"ldr x20, [x4, #512]",
"ubfx x20, x20, #2, #1",
"cbnz x20, #+0x8",
"b #+0x88",
"b #+0x98",
"ldr q2, [x4, #576]",
"str q2, [x28, #16]",
"ldr q2, [x4, #592]",
"str q2, [x28, #32]",
"ldr q2, [x4, #608]",
"str q2, [x28, #48]",
"ldr q2, [x4, #624]",
"str q2, [x28, #64]",
"ldr q2, [x4, #640]",
"str q2, [x28, #80]",
"ldr q2, [x4, #656]",
"str q2, [x28, #96]",
"ldr q2, [x4, #672]",
"str q2, [x28, #112]",
"ldr q2, [x4, #688]",
"str q2, [x28, #128]",
"ldr q2, [x4, #704]",
"str q2, [x28, #144]",
"ldr q2, [x4, #720]",
"str q2, [x28, #160]",
"ldr q2, [x4, #736]",
"str q2, [x28, #176]",
"ldr q2, [x4, #752]",
"str q2, [x28, #192]",
"ldr q2, [x4, #768]",
"str q2, [x28, #208]",
"ldr q2, [x4, #784]",
"str q2, [x28, #224]",
"ldr q3, [x4, #592]",
"ldr q4, [x4, #608]",
"ldr q5, [x4, #624]",
"ldr q6, [x4, #640]",
"ldr q7, [x4, #656]",
"ldr q8, [x4, #672]",
"ldr q9, [x4, #688]",
"ldr q10, [x4, #704]",
"ldr q11, [x4, #720]",
"ldr q12, [x4, #736]",
"ldr q13, [x4, #752]",
"ldr q14, [x4, #768]",
"ldr q15, [x4, #784]",
"str q2, [sp]",
"ldr q2, [x4, #800]",
"str q3, [sp, #32]",
"ldr q3, [x4, #816]",
"str q3, [x28, #256]",
"str q2, [x28, #240]",
"ldr q2, [x4, #816]",
"str q2, [x28, #256]",
"str q15, [x28, #224]",
"str q14, [x28, #208]",
"str q13, [x28, #192]",
"str q12, [x28, #176]",
"str q11, [x28, #160]",
"str q10, [x28, #144]",
"str q9, [x28, #128]",
"str q8, [x28, #112]",
"str q7, [x28, #96]",
"str q6, [x28, #80]",
"str q5, [x28, #64]",
"str q4, [x28, #48]",
"ldr q2, [sp, #32]",
"str q2, [x28, #32]",
"ldr q2, [sp]",
"str q2, [x28, #16]",
"b #+0x48",
"movi v2.2d, #0x0",
"str q2, [x28, #16]",
"str q2, [x28, #32]",
"str q2, [x28, #48]",
"str q2, [x28, #64]",
"str q2, [x28, #80]",
"str q2, [x28, #96]",
"str q2, [x28, #112]",
"str q2, [x28, #128]",
"str q2, [x28, #144]",
"str q2, [x28, #160]",
"str q2, [x28, #176]",
"str q2, [x28, #192]",
"str q2, [x28, #208]",
"str q2, [x28, #224]",
"str q2, [x28, #240]",
"str q2, [x28, #256]",
"str q2, [x28, #240]",
"str q2, [x28, #224]",
"str q2, [x28, #208]",
"str q2, [x28, #192]",
"str q2, [x28, #176]",
"str q2, [x28, #160]",
"str q2, [x28, #144]",
"str q2, [x28, #128]",
"str q2, [x28, #112]",
"str q2, [x28, #96]",
"str q2, [x28, #80]",
"str q2, [x28, #64]",
"str q2, [x28, #48]",
"str q2, [x28, #32]",
"str q2, [x28, #16]",
"ldr x20, [x4, #512]",
"ubfx x20, x20, #1, #2",
"cbnz x20, #+0x8",
@@ -1674,7 +1679,8 @@
"lsr x1, x20, #2",
"bfi x0, x1, #24, #1",
"msr fpcr, x0",
"b #+0x4"
"b #+0x4",
"add sp, sp, #0x40 (64)"
]
},
"mfence": {
@@ -350,9 +350,9 @@
"Map 2 0b00 0xf2 32-bit"
],
"ExpectedArm64ASM": [
"bic w4, w5, w7",
"mov x26, x4",
"tst w4, w4"
"bic w26, w5, w7",
"tst w26, w26",
"mov x4, x26"
]
},
"andn rax, rbx, rcx": {
@@ -361,9 +361,9 @@
"Map 2 0b00 0xf2 64-bit"
],
"ExpectedArm64ASM": [
"bic x4, x5, x7",
"mov x26, x4",
"tst x4, x4"
"bic x26, x5, x7",
"tst x26, x26",
"mov x4, x26"
]
},
"bzhi eax, ebx, ecx": {
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff
+192 -189
View File
@@ -104,35 +104,39 @@
]
},
"add al, 1": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 8,
"Comment": "0x04",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #24",
"cmn w0, w20, lsl #24",
"add w26, w4, #0x1 (1)",
"bfxil x4, x26, #0, #8"
"mov x20, x4",
"bfxil x20, x26, #0, #8",
"mov x27, x4",
"mov x4, x20"
]
},
"add ax, 1": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 8,
"Comment": "0x05",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #16",
"cmn w0, w20, lsl #16",
"add w26, w4, #0x1 (1)",
"bfxil x4, x26, #0, #16"
"mov x20, x4",
"bfxil x20, x26, #0, #16",
"mov x27, x4",
"mov x4, x20"
]
},
"add eax, 1": {
"ExpectedInstructionCount": 3,
"Comment": "0x05",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds w26, w4, #0x1 (1)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -140,8 +144,8 @@
"ExpectedInstructionCount": 3,
"Comment": "0x05",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds x26, x4, #0x1 (1)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -497,10 +501,9 @@
]
},
"adc al, 1": {
"ExpectedInstructionCount": 17,
"ExpectedInstructionCount": 19,
"Comment": "0x14",
"ExpectedArm64ASM": [
"mov x27, x4",
"mov w20, #0x1",
"adc w20, wzr, w20",
"add w21, w4, w20",
@@ -515,15 +518,17 @@
"bic w21, w22, w21",
"ubfx x21, x21, #7, #1",
"orr w20, w20, w21, lsl #28",
"bfxil x4, x26, #0, #8",
"msr nzcv, x20"
"mov x21, x4",
"bfxil x21, x26, #0, #8",
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x21"
]
},
"adc ax, 1": {
"ExpectedInstructionCount": 17,
"ExpectedInstructionCount": 19,
"Comment": "0x15",
"ExpectedArm64ASM": [
"mov x27, x4",
"mov w20, #0x1",
"adc w20, wzr, w20",
"add w21, w4, w20",
@@ -538,8 +543,11 @@
"bic w21, w22, w21",
"ubfx x21, x21, #15, #1",
"orr w20, w20, w21, lsl #28",
"bfxil x4, x26, #0, #16",
"msr nzcv, x20"
"mov x21, x4",
"bfxil x21, x26, #0, #16",
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x21"
]
},
"adc eax, 1": {
@@ -547,8 +555,8 @@
"Comment": "0x15",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"adcs w26, w4, w20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -557,8 +565,8 @@
"Comment": "0x15",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"adcs x26, x4, x20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -687,8 +695,8 @@
"sbcs w26, w7, w5",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x7, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x7, x26"
]
},
"sbb rbx, rcx": {
@@ -702,8 +710,8 @@
"sbcs x26, x7, x5",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x7, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x7, x26"
]
},
"db 0x1A, 0xcb": {
@@ -774,8 +782,8 @@
"sbcs w26, w5, w7",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x5, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x5, x26"
]
},
"db 0x48, 0x1B, 0xcb": {
@@ -792,15 +800,14 @@
"sbcs x26, x5, x7",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x5, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x5, x26"
]
},
"sbb al, 1": {
"ExpectedInstructionCount": 18,
"ExpectedInstructionCount": 20,
"Comment": "0x1C",
"ExpectedArm64ASM": [
"mov x27, x4",
"uxtb w20, w4",
"mov w21, #0x1",
"adc w21, wzr, w21",
@@ -816,15 +823,17 @@
"and w20, w20, w22",
"ubfx x20, x20, #7, #1",
"orr w20, w21, w20, lsl #28",
"bfxil x4, x26, #0, #8",
"msr nzcv, x20"
"mov x21, x4",
"bfxil x21, x26, #0, #8",
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x21"
]
},
"sbb ax, 1": {
"ExpectedInstructionCount": 18,
"ExpectedInstructionCount": 20,
"Comment": "0x1D",
"ExpectedArm64ASM": [
"mov x27, x4",
"uxth w20, w4",
"mov w21, #0x1",
"adc w21, wzr, w21",
@@ -840,8 +849,11 @@
"and w20, w20, w22",
"ubfx x20, x20, #15, #1",
"orr w20, w21, w20, lsl #28",
"bfxil x4, x26, #0, #16",
"msr nzcv, x20"
"mov x21, x4",
"bfxil x21, x26, #0, #16",
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x21"
]
},
"sbb eax, 1": {
@@ -849,15 +861,15 @@
"Comment": "0x1D",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"sbcs w26, w4, w20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"sbb rax, 1": {
@@ -865,15 +877,15 @@
"Comment": "0x1D",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"sbcs x26, x4, x20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"sbb al, -1": {
@@ -936,8 +948,8 @@
"sbcs w26, w4, w20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x4, x26"
]
},
"sbb rax, -1": {
@@ -952,8 +964,8 @@
"sbcs x26, x4, x20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x4, x26"
]
},
"and bl, cl": {
@@ -1142,8 +1154,8 @@
"subs w26, w7, w5",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x7, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x7, x26"
]
},
"sub rbx, rcx": {
@@ -1154,8 +1166,8 @@
"subs x26, x7, x5",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x7, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x7, x26"
]
},
"db 0x2A, 0xcb": {
@@ -1203,8 +1215,8 @@
"subs w26, w5, w7",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x5, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x5, x26"
]
},
"db 0x48, 0x2B, 0xcb": {
@@ -1218,62 +1230,66 @@
"subs x26, x5, x7",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x5, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x5, x26"
]
},
"sub al, 1": {
"ExpectedInstructionCount": 9,
"ExpectedInstructionCount": 11,
"Comment": "0x2C",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #24",
"cmp w0, w20, lsl #24",
"sub w26, w4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"bfxil x4, x26, #0, #8",
"msr nzcv, x20"
"mov x21, x4",
"bfxil x21, x26, #0, #8",
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x21"
]
},
"sub ax, 1": {
"ExpectedInstructionCount": 9,
"ExpectedInstructionCount": 11,
"Comment": "0x2D",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #16",
"cmp w0, w20, lsl #16",
"sub w26, w4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"bfxil x4, x26, #0, #16",
"msr nzcv, x20"
"mov x21, x4",
"bfxil x21, x26, #0, #16",
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x21"
]
},
"sub eax, 1": {
"ExpectedInstructionCount": 6,
"Comment": "0x2D",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs w26, w4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"sub rax, 1": {
"ExpectedInstructionCount": 6,
"Comment": "0x2D",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"sub al, -1": {
@@ -1315,8 +1331,8 @@
"subs w26, w4, w20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x4, x26"
]
},
"sub rax, -1": {
@@ -1327,8 +1343,8 @@
"adds x26, x4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x4, x26"
]
},
"xor bl, cl": {
@@ -1602,13 +1618,13 @@
"Comment": "0x3C",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #24",
"cmp w0, w20, lsl #24",
"sub w26, w4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4"
]
},
"cmp ax, 1": {
@@ -1616,35 +1632,35 @@
"Comment": "0x3D",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #16",
"cmp w0, w20, lsl #16",
"sub w26, w4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4"
]
},
"cmp eax, 1": {
"ExpectedInstructionCount": 5,
"Comment": "0x3D",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs w26, w4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4"
]
},
"cmp rax, 1": {
"ExpectedInstructionCount": 5,
"Comment": "0x3D",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4"
]
},
"cmp al, -1": {
@@ -1915,14 +1931,13 @@
]
},
"xchg bl, cl": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 4,
"Comment": "0x86",
"ExpectedArm64ASM": [
"mov x20, x5",
"bfxil x20, x7, #0, #8",
"mov x21, x5",
"mov x5, x20",
"bfxil x7, x21, #0, #8"
"bfxil x7, x5, #0, #8",
"mov x5, x20"
]
},
"xchg [rax], cl": {
@@ -1934,14 +1949,13 @@
]
},
"xchg bx, cx": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 4,
"Comment": "0x87",
"ExpectedArm64ASM": [
"mov x20, x5",
"bfxil x20, x7, #0, #16",
"mov x21, x5",
"mov x5, x20",
"bfxil x7, x21, #0, #16"
"bfxil x7, x5, #0, #16",
"mov x5, x20"
]
},
"xchg [rax], cx": {
@@ -1953,13 +1967,12 @@
]
},
"xchg ebx, ecx": {
"ExpectedInstructionCount": 4,
"ExpectedInstructionCount": 3,
"Comment": "0x87",
"ExpectedArm64ASM": [
"mov w20, w7",
"mov x21, x5",
"mov x5, x20",
"mov w7, w21"
"mov w7, w5",
"mov x5, x20"
]
},
"xchg [rax], ecx": {
@@ -1973,9 +1986,9 @@
"ExpectedInstructionCount": 3,
"Comment": "0x87",
"ExpectedArm64ASM": [
"mov x20, x5",
"mov x5, x7",
"mov x7, x20"
"mov x20, x7",
"mov x7, x5",
"mov x5, x20"
]
},
"xchg [rax], rcx": {
@@ -2438,24 +2451,22 @@
]
},
"xchg ax, bx": {
"ExpectedInstructionCount": 5,
"ExpectedInstructionCount": 4,
"Comment": "0x90",
"ExpectedArm64ASM": [
"mov x20, x7",
"bfxil x20, x4, #0, #16",
"mov x21, x7",
"mov x7, x20",
"bfxil x4, x21, #0, #16"
"bfxil x4, x7, #0, #16",
"mov x7, x20"
]
},
"xchg eax, ebx": {
"ExpectedInstructionCount": 4,
"ExpectedInstructionCount": 3,
"Comment": "0x90",
"ExpectedArm64ASM": [
"mov w20, w4",
"mov x21, x7",
"mov x7, x20",
"mov w4, w21"
"mov w4, w7",
"mov x7, x20"
]
},
"xchg rax, rbx": {
@@ -2637,26 +2648,26 @@
"strb w21, [x28, #985]",
"ubfx w21, w27, #10, #1",
"sub x21, x22, x21, lsl #1",
"strb w21, [x28, #986]",
"ubfx x21, x27, #11, #1",
"orr w20, w20, w21, lsl #28",
"ubfx w21, w27, #12, #1",
"strb w21, [x28, #988]",
"ubfx w21, w27, #14, #1",
"strb w21, [x28, #990]",
"ubfx w21, w27, #16, #1",
"strb w21, [x28, #992]",
"ubfx w21, w27, #17, #1",
"strb w21, [x28, #993]",
"ubfx w21, w27, #18, #1",
"strb w21, [x28, #994]",
"ubfx w21, w27, #19, #1",
"strb w21, [x28, #995]",
"ubfx w21, w27, #20, #1",
"strb w21, [x28, #996]",
"ubfx w21, w27, #21, #1",
"strb w21, [x28, #997]",
"msr nzcv, x20"
"ubfx x22, x27, #11, #1",
"orr w20, w20, w22, lsl #28",
"ubfx w22, w27, #12, #1",
"strb w22, [x28, #988]",
"ubfx w22, w27, #14, #1",
"strb w22, [x28, #990]",
"ubfx w22, w27, #16, #1",
"strb w22, [x28, #992]",
"ubfx w22, w27, #17, #1",
"strb w22, [x28, #993]",
"ubfx w22, w27, #18, #1",
"strb w22, [x28, #994]",
"ubfx w22, w27, #19, #1",
"strb w22, [x28, #995]",
"ubfx w22, w27, #20, #1",
"strb w22, [x28, #996]",
"ubfx w22, w27, #21, #1",
"strb w22, [x28, #997]",
"msr nzcv, x20",
"strb w21, [x28, #986]"
]
},
"sahf": {
@@ -3243,10 +3254,10 @@
]
},
"repz cmpsb": {
"ExpectedInstructionCount": 29,
"ExpectedInstructionCount": 28,
"Comment": "0xa6",
"ExpectedArm64ASM": [
"cbz x5, #+0x74",
"cbz x5, #+0x70",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -3267,21 +3278,20 @@
"ccmp x27, x26, #nzcv, ne",
"b.eq #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"lsl w0, w21, #24",
"lsl w0, w27, #24",
"cmp w0, w26, lsl #24",
"sub w26, w21, w26",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"sub w26, w27, w26",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"mov x27, x20"
]
},
"repz cmpsw": {
"ExpectedInstructionCount": 29,
"ExpectedInstructionCount": 28,
"Comment": "0xa7",
"ExpectedArm64ASM": [
"cbz x5, #+0x74",
"cbz x5, #+0x70",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -3302,21 +3312,20 @@
"ccmp x27, x26, #nzcv, ne",
"b.eq #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"lsl w0, w21, #16",
"lsl w0, w27, #16",
"cmp w0, w26, lsl #16",
"sub w26, w21, w26",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"sub w26, w27, w26",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"mov x27, x20"
]
},
"repz cmpsd": {
"ExpectedInstructionCount": 27,
"ExpectedInstructionCount": 26,
"Comment": "0xa7",
"ExpectedArm64ASM": [
"cbz x5, #+0x6c",
"cbz x5, #+0x68",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -3337,19 +3346,18 @@
"ccmp x27, x26, #nzcv, ne",
"b.eq #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"subs w26, w21, w26",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"subs w26, w27, w26",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"mov x27, x20"
]
},
"repz cmpsq": {
"ExpectedInstructionCount": 27,
"ExpectedInstructionCount": 26,
"Comment": "0xa7",
"ExpectedArm64ASM": [
"cbz x5, #+0x6c",
"cbz x5, #+0x68",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -3370,19 +3378,18 @@
"ccmp x27, x26, #nzcv, ne",
"b.eq #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"subs x26, x21, x26",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"subs x26, x27, x26",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"mov x27, x20"
]
},
"repnz cmpsb": {
"ExpectedInstructionCount": 29,
"ExpectedInstructionCount": 28,
"Comment": "0xa6",
"ExpectedArm64ASM": [
"cbz x5, #+0x74",
"cbz x5, #+0x70",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -3403,21 +3410,20 @@
"ccmp x27, x26, #nZcv, ne",
"b.ne #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"lsl w0, w21, #24",
"lsl w0, w27, #24",
"cmp w0, w26, lsl #24",
"sub w26, w21, w26",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"sub w26, w27, w26",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"mov x27, x20"
]
},
"repnz cmpsw": {
"ExpectedInstructionCount": 29,
"ExpectedInstructionCount": 28,
"Comment": "0xa7",
"ExpectedArm64ASM": [
"cbz x5, #+0x74",
"cbz x5, #+0x70",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -3438,21 +3444,20 @@
"ccmp x27, x26, #nZcv, ne",
"b.ne #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"lsl w0, w21, #16",
"lsl w0, w27, #16",
"cmp w0, w26, lsl #16",
"sub w26, w21, w26",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"sub w26, w27, w26",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"mov x27, x20"
]
},
"repnz cmpsd": {
"ExpectedInstructionCount": 27,
"ExpectedInstructionCount": 26,
"Comment": "0xa7",
"ExpectedArm64ASM": [
"cbz x5, #+0x6c",
"cbz x5, #+0x68",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -3473,19 +3478,18 @@
"ccmp x27, x26, #nZcv, ne",
"b.ne #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"subs w26, w21, w26",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"subs w26, w27, w26",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"mov x27, x20"
]
},
"repnz cmpsq": {
"ExpectedInstructionCount": 27,
"ExpectedInstructionCount": 26,
"Comment": "0xa7",
"ExpectedArm64ASM": [
"cbz x5, #+0x6c",
"cbz x5, #+0x68",
"ldrsb x20, [x28, #986]",
"lsr x20, x20, #63",
"cbz x20, #+0x8",
@@ -3506,12 +3510,11 @@
"ccmp x27, x26, #nZcv, ne",
"b.ne #-0x18",
"eor w20, w27, w26",
"mov x21, x27",
"mov x27, x20",
"subs x26, x21, x26",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"subs x26, x27, x26",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"mov x27, x20"
]
},
"test al, 1": {
+142 -124
View File
@@ -16,15 +16,17 @@
],
"Instructions": {
"add al, 1": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 8,
"Comment": "GROUP1 0x80 /0",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #24",
"cmn w0, w20, lsl #24",
"add w26, w4, #0x1 (1)",
"bfxil x4, x26, #0, #8"
"mov x20, x4",
"bfxil x20, x26, #0, #8",
"mov x27, x4",
"mov x4, x20"
]
},
"or al, 1": {
@@ -37,10 +39,9 @@
]
},
"adc al, 1": {
"ExpectedInstructionCount": 17,
"ExpectedInstructionCount": 19,
"Comment": "GROUP1 0x80 /2",
"ExpectedArm64ASM": [
"mov x27, x4",
"mov w20, #0x1",
"adc w20, wzr, w20",
"add w21, w4, w20",
@@ -55,15 +56,17 @@
"bic w21, w22, w21",
"ubfx x21, x21, #7, #1",
"orr w20, w20, w21, lsl #28",
"bfxil x4, x26, #0, #8",
"msr nzcv, x20"
"mov x21, x4",
"bfxil x21, x26, #0, #8",
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x21"
]
},
"sbb al, 1": {
"ExpectedInstructionCount": 18,
"ExpectedInstructionCount": 20,
"Comment": "GROUP1 0x80 /3",
"ExpectedArm64ASM": [
"mov x27, x4",
"uxtb w20, w4",
"mov w21, #0x1",
"adc w21, wzr, w21",
@@ -79,8 +82,11 @@
"and w20, w20, w22",
"ubfx x20, x20, #7, #1",
"orr w20, w21, w20, lsl #28",
"bfxil x4, x26, #0, #8",
"msr nzcv, x20"
"mov x21, x4",
"bfxil x21, x26, #0, #8",
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x21"
]
},
"and al, 1": {
@@ -93,18 +99,20 @@
]
},
"sub al, 1": {
"ExpectedInstructionCount": 9,
"ExpectedInstructionCount": 11,
"Comment": "GROUP1 0x80 /5",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #24",
"cmp w0, w20, lsl #24",
"sub w26, w4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"bfxil x4, x26, #0, #8",
"msr nzcv, x20"
"mov x21, x4",
"bfxil x21, x26, #0, #8",
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x21"
]
},
"xor al, 1": {
@@ -121,13 +129,13 @@
"Comment": "GROUP1 0x80 /7",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #24",
"cmp w0, w20, lsl #24",
"sub w26, w4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4"
]
},
"add al, -1": {
@@ -246,23 +254,25 @@
]
},
"add ax, 256": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 8,
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov w20, #0x100",
"mov x27, x4",
"lsl w0, w4, #16",
"cmn w0, w20, lsl #16",
"add w26, w4, #0x100 (256)",
"bfxil x4, x26, #0, #16"
"mov x20, x4",
"bfxil x20, x26, #0, #16",
"mov x27, x4",
"mov x4, x20"
]
},
"add eax, 256": {
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds w26, w4, #0x100 (256)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -270,8 +280,8 @@
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds x26, x4, #0x100 (256)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -298,8 +308,8 @@
"Comment": "GROUP1 0x81 /2",
"ExpectedArm64ASM": [
"mov w20, #0x100",
"mov x27, x4",
"adcs w26, w4, w20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -308,8 +318,8 @@
"Comment": "GROUP1 0x81 /2",
"ExpectedArm64ASM": [
"mov w20, #0x100",
"mov x27, x4",
"adcs x26, x4, x20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -318,15 +328,15 @@
"Comment": "GROUP1 0x81 /3",
"ExpectedArm64ASM": [
"mov w20, #0x100",
"mov x27, x4",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"sbcs w26, w4, w20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"sbb rax, 256": {
@@ -334,15 +344,15 @@
"Comment": "GROUP1 0x81 /3",
"ExpectedArm64ASM": [
"mov w20, #0x100",
"mov x27, x4",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"sbcs x26, x4, x20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"and eax, 256": {
@@ -365,24 +375,24 @@
"ExpectedInstructionCount": 6,
"Comment": "GROUP1 0x81 /5",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs w26, w4, #0x100 (256)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"sub rax, 256": {
"ExpectedInstructionCount": 6,
"Comment": "GROUP1 0x81 /5",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x100 (256)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"xor eax, 256": {
@@ -407,34 +417,36 @@
"ExpectedInstructionCount": 5,
"Comment": "GROUP1 0x81 /7",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs w26, w4, #0x100 (256)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4"
]
},
"cmp rax, 256": {
"ExpectedInstructionCount": 5,
"Comment": "GROUP1 0x81 /7",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x100 (256)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4"
]
},
"add ax, -256": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 8,
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov w20, #0xff00",
"mov x27, x4",
"lsl w0, w4, #16",
"cmn w0, w20, lsl #16",
"add w26, w4, w20",
"bfxil x4, x26, #0, #16"
"mov x20, x4",
"bfxil x20, x26, #0, #16",
"mov x27, x4",
"mov x4, x20"
]
},
"add eax, -256": {
@@ -442,8 +454,8 @@
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov w20, #0xffffff00",
"mov x27, x4",
"adds w26, w4, w20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -451,8 +463,8 @@
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x81 /0",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x100 (256)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -479,8 +491,8 @@
"Comment": "GROUP1 0x81 /2",
"ExpectedArm64ASM": [
"mov w20, #0xffffff00",
"mov x27, x4",
"adcs w26, w4, w20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -489,8 +501,8 @@
"Comment": "GROUP1 0x81 /2",
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffff00",
"mov x27, x4",
"adcs x26, x4, x20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -499,15 +511,15 @@
"Comment": "GROUP1 0x81 /3",
"ExpectedArm64ASM": [
"mov w20, #0xffffff00",
"mov x27, x4",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"sbcs w26, w4, w20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"sbb rax, -256": {
@@ -515,15 +527,15 @@
"Comment": "GROUP1 0x81 /3",
"ExpectedArm64ASM": [
"mov x20, #0xffffffffffffff00",
"mov x27, x4",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"sbcs x26, x4, x20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"and eax, -256": {
@@ -547,24 +559,24 @@
"Comment": "GROUP1 0x81 /5",
"ExpectedArm64ASM": [
"mov w20, #0xffffff00",
"mov x27, x4",
"subs w26, w4, w20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"sub rax, -256": {
"ExpectedInstructionCount": 6,
"Comment": "GROUP1 0x81 /5",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds x26, x4, #0x100 (256)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"xor eax, -256": {
@@ -590,42 +602,44 @@
"Comment": "GROUP1 0x81 /7",
"ExpectedArm64ASM": [
"mov w20, #0xffffff00",
"mov x27, x4",
"subs w26, w4, w20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4"
]
},
"cmp rax, -256": {
"ExpectedInstructionCount": 5,
"Comment": "GROUP1 0x81 /7",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds x26, x4, #0x100 (256)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4"
]
},
"add ax, 1": {
"ExpectedInstructionCount": 6,
"ExpectedInstructionCount": 8,
"Comment": "GROUP1 0x83 /0",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"lsl w0, w4, #16",
"cmn w0, w20, lsl #16",
"add w26, w4, #0x1 (1)",
"bfxil x4, x26, #0, #16"
"mov x20, x4",
"bfxil x20, x26, #0, #16",
"mov x27, x4",
"mov x4, x20"
]
},
"add eax, 1": {
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x83 /0",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds w26, w4, #0x1 (1)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -633,8 +647,8 @@
"ExpectedInstructionCount": 3,
"Comment": "GROUP1 0x83 /0",
"ExpectedArm64ASM": [
"mov x27, x4",
"adds x26, x4, #0x1 (1)",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -661,8 +675,8 @@
"Comment": "GROUP1 0x83 /2",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"adcs w26, w4, w20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -671,8 +685,8 @@
"Comment": "GROUP1 0x83 /2",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"adcs x26, x4, x20",
"mov x27, x4",
"mov x4, x26"
]
},
@@ -681,15 +695,15 @@
"Comment": "GROUP1 0x83 /3",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"sbcs w26, w4, w20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"sbb rax, 1": {
@@ -697,15 +711,15 @@
"Comment": "GROUP1 0x83 /3",
"ExpectedArm64ASM": [
"mov w20, #0x1",
"mov x27, x4",
"mrs x21, nzcv",
"eor w21, w21, #0x20000000",
"msr nzcv, x21",
"sbcs x26, x4, x20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"and eax, 1": {
@@ -728,24 +742,24 @@
"ExpectedInstructionCount": 6,
"Comment": "GROUP1 0x83 /5",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs w26, w4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"sub rax, 1": {
"ExpectedInstructionCount": 6,
"Comment": "GROUP1 0x83 /5",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4",
"mov x4, x26"
]
},
"xor eax, 1": {
@@ -770,22 +784,22 @@
"ExpectedInstructionCount": 5,
"Comment": "GROUP1 0x83 /7",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs w26, w4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4"
]
},
"cmp rax, 1": {
"ExpectedInstructionCount": 5,
"Comment": "GROUP1 0x83 /7",
"ExpectedArm64ASM": [
"mov x27, x4",
"subs x26, x4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x4"
]
},
"add ax, -1": {
@@ -871,8 +885,8 @@
"sbcs w26, w4, w20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x4, x26"
]
},
"sbb rax, -1": {
@@ -887,8 +901,8 @@
"sbcs x26, x4, x20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x4, x26"
]
},
"and eax, -1": {
@@ -918,8 +932,8 @@
"subs w26, w4, w20",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x4, x26"
]
},
"sub rax, -1": {
@@ -930,8 +944,8 @@
"adds x26, x4, #0x1 (1)",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x4, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x4, x26"
]
},
"xor eax, -1": {
@@ -1626,8 +1640,8 @@
"eor x21, x20, x4",
"lsr x21, x21, #63",
"bfi w22, w21, #28, #1",
"mov x4, x20",
"msr nzcv, x22"
"msr nzcv, x22",
"mov x4, x20"
]
},
"rcr ax, 1": {
@@ -2427,16 +2441,18 @@
]
},
"neg bl": {
"ExpectedInstructionCount": 7,
"ExpectedInstructionCount": 9,
"Comment": "GROUP2 0xf6 /3",
"ExpectedArm64ASM": [
"mov x27, x7",
"cmp wzr, w7, lsl #24",
"neg w26, w7",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"bfxil x7, x26, #0, #8",
"msr nzcv, x20"
"mov x21, x7",
"bfxil x21, x26, #0, #8",
"msr nzcv, x20",
"mov x27, x7",
"mov x7, x21"
]
},
"mul bl": {
@@ -2564,40 +2580,42 @@
]
},
"neg bx": {
"ExpectedInstructionCount": 7,
"ExpectedInstructionCount": 9,
"Comment": "GROUP2 0xf7 /2",
"ExpectedArm64ASM": [
"mov x27, x7",
"cmp wzr, w7, lsl #16",
"neg w26, w7",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"bfxil x7, x26, #0, #16",
"msr nzcv, x20"
"mov x21, x7",
"bfxil x21, x26, #0, #16",
"msr nzcv, x20",
"mov x27, x7",
"mov x7, x21"
]
},
"neg ebx": {
"ExpectedInstructionCount": 6,
"Comment": "GROUP2 0xf7 /2",
"ExpectedArm64ASM": [
"mov x27, x7",
"negs w26, w7",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x7, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x7",
"mov x7, x26"
]
},
"neg rbx": {
"ExpectedInstructionCount": 6,
"Comment": "GROUP2 0xf7 /2",
"ExpectedArm64ASM": [
"mov x27, x7",
"negs x26, x7",
"mrs x20, nzcv",
"eor w20, w20, #0x20000000",
"mov x7, x26",
"msr nzcv, x20"
"msr nzcv, x20",
"mov x27, x7",
"mov x7, x26"
]
},
"mul bx": {
@@ -2633,9 +2651,9 @@
"ExpectedArm64ASM": [
"mul x20, x7, x4",
"umulh x6, x7, x4",
"mov x4, x20",
"cmp x6, #0x0 (0)",
"ccmn xzr, #0, #nzCV, eq"
"ccmn xzr, #0, #nzCV, eq",
"mov x4, x20"
]
},
"imul bx": {
@@ -2873,12 +2891,12 @@
"Comment": "GROUP4 0xfe /0",
"ExpectedArm64ASM": [
"cset w20, hs",
"mov x27, x4",
"adds w26, w4, #0x1 (1)",
"mrs x21, nzcv",
"bfi w21, w20, #29, #1",
"mov x4, x26",
"msr nzcv, x21"
"msr nzcv, x21",
"mov x27, x4",
"mov x4, x26"
]
},
"inc rax": {
@@ -2886,12 +2904,12 @@
"Comment": "GROUP4 0xfe /0",
"ExpectedArm64ASM": [
"cset w20, hs",
"mov x27, x4",
"adds x26, x4, #0x1 (1)",
"mrs x21, nzcv",
"bfi w21, w20, #29, #1",
"mov x4, x26",
"msr nzcv, x21"
"msr nzcv, x21",
"mov x27, x4",
"mov x4, x26"
]
},
"dec ax": {
@@ -2915,12 +2933,12 @@
"Comment": "GROUP4 0xfe /1",
"ExpectedArm64ASM": [
"cset w20, hs",
"mov x27, x4",
"subs w26, w4, #0x1 (1)",
"mrs x21, nzcv",
"bfi w21, w20, #29, #1",
"mov x4, x26",
"msr nzcv, x21"
"msr nzcv, x21",
"mov x27, x4",
"mov x4, x26"
]
},
"dec rax": {
@@ -2928,12 +2946,12 @@
"Comment": "GROUP4 0xfe /1",
"ExpectedArm64ASM": [
"cset w20, hs",
"mov x27, x4",
"subs x26, x4, #0x1 (1)",
"mrs x21, nzcv",
"bfi w21, w20, #29, #1",
"mov x4, x26",
"msr nzcv, x21"
"msr nzcv, x21",
"mov x27, x4",
"mov x4, x26"
]
},
"push ax": {
@@ -204,12 +204,12 @@
"Comment": "0x40",
"ExpectedArm64ASM": [
"cset w20, hs",
"mov x27, x4",
"adds w26, w4, #0x1 (1)",
"mrs x21, nzcv",
"bfi w21, w20, #29, #1",
"mov x4, x26",
"msr nzcv, x21"
"msr nzcv, x21",
"mov x27, x4",
"mov x4, x26"
]
},
"dec ax": {
@@ -247,12 +247,12 @@
"Comment": "0x48",
"ExpectedArm64ASM": [
"cset w20, hs",
"mov x27, x4",
"subs w26, w4, #0x1 (1)",
"mrs x21, nzcv",
"bfi w21, w20, #29, #1",
"mov x4, x26",
"msr nzcv, x21"
"msr nzcv, x21",
"mov x27, x4",
"mov x4, x26"
]
},
"pusha": {
+113 -107
View File
@@ -1464,23 +1464,14 @@
"strb w23, [x28, #1018]",
"strb w20, [x28, #1022]",
"ldrb w20, [x4, #4]",
"strb w20, [x28, #1298]",
"ldr q2, [x4, #32]",
"str q2, [x28, #1040]",
"ldr q2, [x4, #48]",
"str q2, [x28, #1056]",
"ldr q2, [x4, #64]",
"str q2, [x28, #1072]",
"ldr q2, [x4, #80]",
"str q2, [x28, #1088]",
"ldr q2, [x4, #96]",
"str q2, [x28, #1104]",
"ldr q2, [x4, #112]",
"str q2, [x28, #1120]",
"ldr q2, [x4, #128]",
"str q2, [x28, #1136]",
"ldr q2, [x4, #144]",
"str q2, [x28, #1152]",
"ldr q3, [x4, #48]",
"ldr q4, [x4, #64]",
"ldr q5, [x4, #80]",
"ldr q6, [x4, #96]",
"ldr q7, [x4, #112]",
"ldr q8, [x4, #128]",
"ldr q9, [x4, #144]",
"ldr q16, [x4, #160]",
"ldr q17, [x4, #176]",
"ldr q18, [x4, #192]",
@@ -1497,15 +1488,24 @@
"ldr q29, [x4, #368]",
"ldr q30, [x4, #384]",
"ldr q31, [x4, #400]",
"ldr w20, [x4, #24]",
"ubfx w20, w20, #13, #3",
"rbit w1, w20",
"ldr w21, [x4, #24]",
"ubfx w21, w21, #13, #3",
"rbit w1, w21",
"lsr w1, w1, #30",
"mrs x0, fpcr",
"bfi x0, x1, #22, #2",
"lsr x1, x20, #2",
"lsr x1, x21, #2",
"bfi x0, x1, #24, #1",
"msr fpcr, x0"
"msr fpcr, x0",
"strb w20, [x28, #1298]",
"str q9, [x28, #1152]",
"str q8, [x28, #1136]",
"str q7, [x28, #1120]",
"str q6, [x28, #1104]",
"str q5, [x28, #1088]",
"str q4, [x28, #1072]",
"str q3, [x28, #1056]",
"str q2, [x28, #1040]"
]
},
"rdgsbase eax": {
@@ -1697,9 +1697,10 @@
]
},
"xrstor [rax]": {
"ExpectedInstructionCount": 159,
"ExpectedInstructionCount": 165,
"Comment": "GROUP15 0x0F 0xAE /5",
"ExpectedArm64ASM": [
"sub sp, sp, #0x40 (64)",
"ldr x20, [x4, #512]",
"ubfx x20, x20, #0, #1",
"cbnz x20, #+0x8",
@@ -1718,23 +1719,23 @@
"strb w23, [x28, #1018]",
"strb w20, [x28, #1022]",
"ldrb w20, [x4, #4]",
"strb w20, [x28, #1298]",
"ldr q2, [x4, #32]",
"ldr q3, [x4, #48]",
"ldr q4, [x4, #64]",
"ldr q5, [x4, #80]",
"ldr q6, [x4, #96]",
"ldr q7, [x4, #112]",
"ldr q8, [x4, #128]",
"ldr q9, [x4, #144]",
"strb w20, [x28, #1298]",
"str q9, [x28, #1152]",
"str q8, [x28, #1136]",
"str q7, [x28, #1120]",
"str q6, [x28, #1104]",
"str q5, [x28, #1088]",
"str q4, [x28, #1072]",
"str q3, [x28, #1056]",
"str q2, [x28, #1040]",
"ldr q2, [x4, #48]",
"str q2, [x28, #1056]",
"ldr q2, [x4, #64]",
"str q2, [x28, #1072]",
"ldr q2, [x4, #80]",
"str q2, [x28, #1088]",
"ldr q2, [x4, #96]",
"str q2, [x28, #1104]",
"ldr q2, [x4, #112]",
"str q2, [x28, #1120]",
"ldr q2, [x4, #128]",
"str q2, [x28, #1136]",
"ldr q2, [x4, #144]",
"str q2, [x28, #1152]",
"b #+0x4c",
"mov w20, #0x0",
"mov w21, #0x37f",
@@ -1744,16 +1745,16 @@
"strb w20, [x28, #1017]",
"strb w20, [x28, #1018]",
"strb w20, [x28, #1022]",
"strb w20, [x28, #1298]",
"movi v2.2d, #0x0",
"str q2, [x28, #1040]",
"str q2, [x28, #1056]",
"str q2, [x28, #1072]",
"str q2, [x28, #1088]",
"str q2, [x28, #1104]",
"str q2, [x28, #1120]",
"str q2, [x28, #1136]",
"strb w20, [x28, #1298]",
"str q2, [x28, #1152]",
"str q2, [x28, #1136]",
"str q2, [x28, #1120]",
"str q2, [x28, #1104]",
"str q2, [x28, #1088]",
"str q2, [x28, #1072]",
"str q2, [x28, #1056]",
"str q2, [x28, #1040]",
"ldr x20, [x4, #512]",
"ubfx x20, x20, #1, #1",
"cbnz x20, #+0x8",
@@ -1775,76 +1776,80 @@
"ldr q30, [x4, #384]",
"ldr q31, [x4, #400]",
"b #+0x44",
"movi v16.2d, #0x0",
"mov v17.16b, v16.16b",
"mov v18.16b, v16.16b",
"mov v19.16b, v16.16b",
"mov v20.16b, v16.16b",
"mov v21.16b, v16.16b",
"mov v22.16b, v16.16b",
"mov v23.16b, v16.16b",
"mov v24.16b, v16.16b",
"mov v25.16b, v16.16b",
"mov v26.16b, v16.16b",
"mov v27.16b, v16.16b",
"mov v28.16b, v16.16b",
"mov v29.16b, v16.16b",
"mov v30.16b, v16.16b",
"mov v31.16b, v16.16b",
"movi v31.2d, #0x0",
"mov v30.16b, v31.16b",
"mov v29.16b, v31.16b",
"mov v28.16b, v31.16b",
"mov v27.16b, v31.16b",
"mov v26.16b, v31.16b",
"mov v25.16b, v31.16b",
"mov v24.16b, v31.16b",
"mov v23.16b, v31.16b",
"mov v22.16b, v31.16b",
"mov v21.16b, v31.16b",
"mov v20.16b, v31.16b",
"mov v19.16b, v31.16b",
"mov v18.16b, v31.16b",
"mov v17.16b, v31.16b",
"mov v16.16b, v31.16b",
"ldr x20, [x4, #512]",
"ubfx x20, x20, #2, #1",
"cbnz x20, #+0x8",
"b #+0x88",
"b #+0x98",
"ldr q2, [x4, #576]",
"str q2, [x28, #16]",
"ldr q2, [x4, #592]",
"str q2, [x28, #32]",
"ldr q2, [x4, #608]",
"str q2, [x28, #48]",
"ldr q2, [x4, #624]",
"str q2, [x28, #64]",
"ldr q2, [x4, #640]",
"str q2, [x28, #80]",
"ldr q2, [x4, #656]",
"str q2, [x28, #96]",
"ldr q2, [x4, #672]",
"str q2, [x28, #112]",
"ldr q2, [x4, #688]",
"str q2, [x28, #128]",
"ldr q2, [x4, #704]",
"str q2, [x28, #144]",
"ldr q2, [x4, #720]",
"str q2, [x28, #160]",
"ldr q2, [x4, #736]",
"str q2, [x28, #176]",
"ldr q2, [x4, #752]",
"str q2, [x28, #192]",
"ldr q2, [x4, #768]",
"str q2, [x28, #208]",
"ldr q2, [x4, #784]",
"str q2, [x28, #224]",
"ldr q3, [x4, #592]",
"ldr q4, [x4, #608]",
"ldr q5, [x4, #624]",
"ldr q6, [x4, #640]",
"ldr q7, [x4, #656]",
"ldr q8, [x4, #672]",
"ldr q9, [x4, #688]",
"ldr q10, [x4, #704]",
"ldr q11, [x4, #720]",
"ldr q12, [x4, #736]",
"ldr q13, [x4, #752]",
"ldr q14, [x4, #768]",
"ldr q15, [x4, #784]",
"str q2, [sp]",
"ldr q2, [x4, #800]",
"str q3, [sp, #32]",
"ldr q3, [x4, #816]",
"str q3, [x28, #256]",
"str q2, [x28, #240]",
"ldr q2, [x4, #816]",
"str q2, [x28, #256]",
"str q15, [x28, #224]",
"str q14, [x28, #208]",
"str q13, [x28, #192]",
"str q12, [x28, #176]",
"str q11, [x28, #160]",
"str q10, [x28, #144]",
"str q9, [x28, #128]",
"str q8, [x28, #112]",
"str q7, [x28, #96]",
"str q6, [x28, #80]",
"str q5, [x28, #64]",
"str q4, [x28, #48]",
"ldr q2, [sp, #32]",
"str q2, [x28, #32]",
"ldr q2, [sp]",
"str q2, [x28, #16]",
"b #+0x48",
"movi v2.2d, #0x0",
"str q2, [x28, #16]",
"str q2, [x28, #32]",
"str q2, [x28, #48]",
"str q2, [x28, #64]",
"str q2, [x28, #80]",
"str q2, [x28, #96]",
"str q2, [x28, #112]",
"str q2, [x28, #128]",
"str q2, [x28, #144]",
"str q2, [x28, #160]",
"str q2, [x28, #176]",
"str q2, [x28, #192]",
"str q2, [x28, #208]",
"str q2, [x28, #224]",
"str q2, [x28, #240]",
"str q2, [x28, #256]",
"str q2, [x28, #240]",
"str q2, [x28, #224]",
"str q2, [x28, #208]",
"str q2, [x28, #192]",
"str q2, [x28, #176]",
"str q2, [x28, #160]",
"str q2, [x28, #144]",
"str q2, [x28, #128]",
"str q2, [x28, #112]",
"str q2, [x28, #96]",
"str q2, [x28, #80]",
"str q2, [x28, #64]",
"str q2, [x28, #48]",
"str q2, [x28, #32]",
"str q2, [x28, #16]",
"ldr x20, [x4, #512]",
"ubfx x20, x20, #1, #2",
"cbnz x20, #+0x8",
@@ -1858,7 +1863,8 @@
"lsr x1, x20, #2",
"bfi x0, x1, #24, #1",
"msr fpcr, x0",
"b #+0x4"
"b #+0x4",
"add sp, sp, #0x40 (64)"
]
},
"mfence": {
+6 -6
View File
@@ -4900,9 +4900,9 @@
"Map 2 0b00 0xf2 32-bit"
],
"ExpectedArm64ASM": [
"bic w4, w5, w7",
"mov x26, x4",
"tst w4, w4"
"bic w26, w5, w7",
"tst w26, w26",
"mov x4, x26"
]
},
"andn rax, rbx, rcx": {
@@ -4911,9 +4911,9 @@
"Map 2 0b00 0xf2 64-bit"
],
"ExpectedArm64ASM": [
"bic x4, x5, x7",
"mov x26, x4",
"tst x4, x4"
"bic x26, x5, x7",
"tst x26, x26",
"mov x4, x26"
]
},
"bzhi eax, ebx, ecx": {
File diff suppressed because it is too large. Load diff
File diff suppressed because it is too large. Load diff