mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 14:00:16 +02:00
Merge pull request #4764 from alyssarosenzweig/opt/drop-constprop-2
Drop ConstProp
This commit is contained in:
55 files changed
+3824
-4043
No files matched your search
@@ -58,6 +58,7 @@ class OpDefinition:
|
||||
JITDispatch: bool
|
||||
JITDispatchOverride: str
|
||||
TiedSource: int
|
||||
Inline: list
|
||||
Arguments: list
|
||||
EmitValidation: list
|
||||
Desc: list
|
||||
@@ -278,6 +279,12 @@ def parse_ops(ops):
|
||||
if "TiedSource" in op_val:
|
||||
OpDef.TiedSource = op_val["TiedSource"]
|
||||
|
||||
# Pad Inline out to the argument count
|
||||
OpDef.Inline = [''] * len(OpDef.Arguments)
|
||||
if "Inline" in op_val:
|
||||
Value = op_val["Inline"]
|
||||
OpDef.Inline[0:len(Value)] = Value
|
||||
|
||||
# Do some fixups of the data here
|
||||
if len(OpDef.EmitValidation) != 0:
|
||||
for i in range(len(OpDef.EmitValidation)):
|
||||
@@ -773,9 +780,29 @@ def print_ir_allocator_helpers():
|
||||
output_file.write(") {\n")
|
||||
output_file.write("\t\tauto ListDataBegin = DualListData.ListBegin();\n")
|
||||
|
||||
idx = 0
|
||||
for arg in op.Arguments:
|
||||
if arg.IsSSA:
|
||||
output_file.write("\t\t{}->AddUse();\n".format(arg.Name))
|
||||
# Inline an immediate if we can
|
||||
inline = op.Inline[idx]
|
||||
idx += 1
|
||||
|
||||
if inline != '':
|
||||
Sized = "Size" in [x.Name for x in op.Arguments]
|
||||
P = ["Size" if Sized else "OpSize::i64Bit", arg.Name]
|
||||
|
||||
# A few cases need extra info plumbed.
|
||||
if inline == "SubtractZero":
|
||||
P += ["Src2"]
|
||||
elif inline == "Mem":
|
||||
P += ["OffsetType", "OffsetScale"]
|
||||
elif inline == "Memtso":
|
||||
P += ["OffsetType", "OffsetScale", "true /* TSO */"]
|
||||
inline = "Mem"
|
||||
|
||||
output_file.write(f"\t\t{arg.Name} = Inline{inline}({', '.join(P)});\n")
|
||||
|
||||
output_file.write(f"\t\t{arg.Name}->AddUse();\n")
|
||||
|
||||
# Insert validation here. This is skipped for the
|
||||
# OrderedNodeWrapper version because validation can depend on
|
||||
|
||||
@@ -65,7 +65,6 @@ set (SRCS
|
||||
Interface/IR/IRDumper.cpp
|
||||
Interface/IR/IREmitter.cpp
|
||||
Interface/IR/PassManager.cpp
|
||||
Interface/IR/Passes/ConstProp.cpp
|
||||
Interface/IR/Passes/IRDumperPass.cpp
|
||||
Interface/IR/Passes/IRValidation.cpp
|
||||
Interface/IR/Passes/RedundantFlagCalculationElimination.cpp
|
||||
|
||||
@@ -11,8 +11,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
Ref Tmp = A.Base;
|
||||
|
||||
if (A.Offset) {
|
||||
Ref Offset = IREmit->Constant(A.Offset);
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, Offset) : Offset;
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Offset) : IREmit->Constant(A.Offset);
|
||||
}
|
||||
|
||||
if (A.Index) {
|
||||
@@ -25,7 +24,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
Tmp = IREmit->_Lshl(GPRSize, A.Index, IREmit->Constant(Log2));
|
||||
}
|
||||
} else {
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Index) : A.Index;
|
||||
}
|
||||
}
|
||||
|
||||
@@ -46,7 +45,7 @@ Ref LoadEffectiveAddress(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSize, b
|
||||
}
|
||||
|
||||
if (A.Segment && AddSegmentBase) {
|
||||
Tmp = Tmp ? IREmit->_Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
Tmp = Tmp ? IREmit->Add(GPRSize, Tmp, A.Segment) : A.Segment;
|
||||
}
|
||||
|
||||
return Tmp ?: IREmit->Constant(0);
|
||||
@@ -115,7 +114,7 @@ AddressMode SelectAddressMode(IREmitter* IREmit, AddressMode A, IR::OpSize GPRSi
|
||||
|
||||
auto ScaledRegisterLoadstore = [IREmit, GPRSize](AddressMode A) -> AddressMode {
|
||||
if (A.Index && A.Segment) {
|
||||
A.Base = IREmit->_Add(GPRSize, A.Base, A.Segment);
|
||||
A.Base = IREmit->Add(GPRSize, A.Base, A.Segment);
|
||||
} else if (A.Segment) {
|
||||
A.Index = A.Segment;
|
||||
A.IndexScale = 1;
|
||||
|
||||
@@ -613,7 +613,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
if (InstsInBlock == 0) {
|
||||
// Special case for an empty instruction block.
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry - GuestRIP));
|
||||
}
|
||||
|
||||
for (size_t i = 0; i < InstsInBlock; ++i) {
|
||||
@@ -657,7 +657,7 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
Thread->OpDispatcher->SetCurrentCodeBlock(CodeWasChangedBlock);
|
||||
Thread->OpDispatcher->_ThreadRemoveCodeEntry();
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, InstAddress - GuestRIP));
|
||||
|
||||
auto NextOpBlock = Thread->OpDispatcher->CreateNewCodeBlockAfter(CurrentBlock);
|
||||
|
||||
@@ -723,7 +723,8 @@ ContextImpl::GenerateIR(FEXCore::Core::InternalThreadState* Thread, uint64_t Gue
|
||||
|
||||
if (NeedsBlockEnd) {
|
||||
// We had some instructions. Early exit
|
||||
Thread->OpDispatcher->ExitFunction(Thread->OpDispatcher->_EntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
Thread->OpDispatcher->ExitFunction(
|
||||
Thread->OpDispatcher->_InlineEntrypointOffset(GPRSize, Block.Entry + BlockInstructionsLength - GuestRIP));
|
||||
break;
|
||||
}
|
||||
|
||||
|
||||
@@ -166,7 +166,7 @@ void OpDispatchBuilder::RETOp(OpcodeArgs) {
|
||||
|
||||
if (Op->OP == 0xC2) {
|
||||
auto Offset = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
SP = _Add(GPRSize, SP, Offset);
|
||||
SP = Add(GPRSize, SP, Offset);
|
||||
}
|
||||
|
||||
// Store the new stack pointer
|
||||
@@ -453,7 +453,7 @@ void OpDispatchBuilder::POPAOp(OpcodeArgs) {
|
||||
StoreGPRRegister(X86State::REG_RBP, Pop(Size, SP), Size);
|
||||
|
||||
// Skip loading RSP because it'll be correct at the end
|
||||
SP = _RMWHandle(_Add(OpSize::i64Bit, SP, _InlineConstant(IR::OpSizeToSize(Size))));
|
||||
SP = _RMWHandle(Add(OpSize::i64Bit, SP, IR::OpSizeToSize(Size)));
|
||||
|
||||
StoreGPRRegister(X86State::REG_RBX, Pop(Size, SP), Size);
|
||||
StoreGPRRegister(X86State::REG_RDX, Pop(Size, SP), Size);
|
||||
@@ -524,7 +524,7 @@ void OpDispatchBuilder::CALLOp(OpcodeArgs) {
|
||||
int64_t TargetOffset = Op->Src[0].Literal();
|
||||
|
||||
auto NewRIP = GetRelocatedPC(Op, TargetOffset);
|
||||
auto ConstantPC = _Sub(GPRSize, NewRIP, Constant(TargetOffset));
|
||||
auto ConstantPC = Sub(GPRSize, NewRIP, TargetOffset);
|
||||
|
||||
// Push the return address.
|
||||
Push(GPRSize, ConstantPC);
|
||||
@@ -534,7 +534,7 @@ void OpDispatchBuilder::CALLOp(OpcodeArgs) {
|
||||
|
||||
if (NextRIP != TargetRIP) {
|
||||
// Store the RIP
|
||||
ExitFunction(NewRIP, BranchHint::Call, ConstantPC, [&]() {
|
||||
ExitRelocatedPC(Op, TargetOffset, BranchHint::Call, ConstantPC, [&]() {
|
||||
auto CallReturnJumpTarget = JumpTargets.find(NextRIP);
|
||||
if (CallReturnJumpTarget != JumpTargets.end() && CallReturnJumpTarget->second.IsEntryPoint) {
|
||||
return CallReturnJumpTarget->second.BlockEntry;
|
||||
@@ -567,27 +567,6 @@ void OpDispatchBuilder::CALLAbsoluteOp(OpcodeArgs) {
|
||||
}());
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::SelectPF(bool Invert, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue) {
|
||||
uint64_t TrueConst, FalseConst;
|
||||
if (IsValueConstant(WrapNode(TrueValue), &TrueConst) && IsValueConstant(WrapNode(FalseValue), &FalseConst) && FalseConst == 0) {
|
||||
if (TrueConst == 1) {
|
||||
return LoadPFRaw(true, Invert);
|
||||
} else if (TrueConst == 0xffffffff) {
|
||||
return _Sbfe(OpSize::i32Bit, 1, 0, LoadPFRaw(false, Invert));
|
||||
} else if (TrueConst == 0xffffffffffffffffull) {
|
||||
return _Sbfe(OpSize::i64Bit, 1, 0, LoadPFRaw(false, Invert));
|
||||
}
|
||||
}
|
||||
|
||||
Ref Cmp = LoadPFRaw(false, Invert);
|
||||
SaveNZCV();
|
||||
|
||||
// Because we're only clobbering NZCV internally, we ignore all carry flag
|
||||
// shenanigans and just use the raw test and raw select.
|
||||
_TestNZ(OpSize::i32Bit, Cmp, _InlineConstant(1));
|
||||
return _NZCVSelect(ResultSize, {COND_NEQ}, TrueValue, FalseValue);
|
||||
}
|
||||
|
||||
std::optional<CondClassType> OpDispatchBuilder::DecodeNZCVCondition(uint8_t OP) {
|
||||
switch (OP) {
|
||||
case 0x0: { // JO - Jump if OF == 1
|
||||
@@ -643,51 +622,64 @@ std::optional<CondClassType> OpDispatchBuilder::DecodeNZCVCondition(uint8_t OP)
|
||||
}
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::SelectCC(uint8_t OP, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue) {
|
||||
auto Cond = DecodeNZCVCondition(OP);
|
||||
if (Cond) {
|
||||
// Use raw select since DecodeNZCVCondition handles the carry invert
|
||||
return _NZCVSelect(ResultSize, *Cond, TrueValue, FalseValue);
|
||||
}
|
||||
static bool ParityJumpIsJP(uint8_t OP) {
|
||||
LOGMAN_THROW_A_FMT(OP == 0xA || OP == 0xB, "JP or JNP");
|
||||
return OP == 0xA;
|
||||
}
|
||||
|
||||
switch (OP) {
|
||||
case 0xA: // JP - Jump if PF == 1
|
||||
case 0xB: { // JNP - Jump if PF == 0
|
||||
Ref OpDispatchBuilder::SelectCC0All1(uint8_t OP) {
|
||||
if (auto Cond = DecodeNZCVCondition(OP); Cond) {
|
||||
// Use raw select since DecodeNZCVCondition handles the carry invert
|
||||
return _NZCVSelect(OpSize::i64Bit, *Cond, _InlineConstant(~0ULL), _InlineConstant(0));
|
||||
} else {
|
||||
// Raw value contains inverted PF in bottom bit
|
||||
return SelectPF(OP == 0xA, ResultSize, TrueValue, FalseValue);
|
||||
}
|
||||
default: LOGMAN_MSG_A_FMT("Unknown CC Op: 0x{:x}\n", OP); return nullptr;
|
||||
return _Sbfe(OpSize::i64Bit, 1, 0, LoadPFRaw(false, ParityJumpIsJP(OP)));
|
||||
}
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::SETccOp(OpcodeArgs) {
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
auto OneConst = Constant(1);
|
||||
|
||||
auto SrcCond = SelectCC(Op->OP & 0xF, OpSize::i64Bit, OneConst, ZeroConst);
|
||||
Ref SrcCond;
|
||||
if (auto Cond = DecodeNZCVCondition(Op->OP & 0xf); Cond) {
|
||||
// Use raw select since DecodeNZCVCondition handles the carry invert
|
||||
SrcCond = _NZCVSelect01(*Cond);
|
||||
} else {
|
||||
SrcCond = LoadPFRaw(true, ParityJumpIsJP(Op->OP & 0xf));
|
||||
}
|
||||
|
||||
StoreResult(GPRClass, Op, SrcCond, OpSize::iInvalid);
|
||||
}
|
||||
|
||||
void OpDispatchBuilder::CMOVOp(OpcodeArgs) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
const auto OP = Op->OP & 0xF;
|
||||
const auto ResultSize = std::max(OpSize::i32Bit, OpSizeFromSrc(Op));
|
||||
|
||||
// Calculate flags early.
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// Destination is always a GPR.
|
||||
Ref Dest = LoadSource_WithOpSize(GPRClass, Op, Op->Dest, GPRSize, Op->Flags);
|
||||
Ref Src {};
|
||||
Ref Src {}, SrcCond {};
|
||||
if (Op->Src[0].IsGPR()) {
|
||||
Src = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], GPRSize, Op->Flags);
|
||||
} else {
|
||||
Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags);
|
||||
}
|
||||
|
||||
auto SrcCond = SelectCC(Op->OP & 0xF, std::max(OpSize::i32Bit, OpSizeFromSrc(Op)), Src, Dest);
|
||||
if (auto Cond = DecodeNZCVCondition(OP); Cond) {
|
||||
// Use raw select since DecodeNZCVCondition handles the carry invert
|
||||
SrcCond = _NZCVSelect(ResultSize, *Cond, Src, Dest);
|
||||
} else {
|
||||
// Raw value contains inverted PF in bottom bit
|
||||
Ref Cmp = LoadPFRaw(false, ParityJumpIsJP(OP));
|
||||
SaveNZCV();
|
||||
|
||||
// Because we're only clobbering NZCV internally, we ignore all carry flag
|
||||
// shenanigans and just use the raw test and raw select.
|
||||
_TestNZ(OpSize::i32Bit, Cmp, _InlineConstant(1));
|
||||
SrcCond = _NZCVSelect(ResultSize, {COND_NEQ}, Src, Dest);
|
||||
}
|
||||
|
||||
StoreResult(GPRClass, Op, SrcCond, OpSize::iInvalid);
|
||||
}
|
||||
@@ -742,10 +734,8 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
StartNewBlock();
|
||||
|
||||
auto NewRIP = GetRelocatedPC(Op, TargetOffset);
|
||||
|
||||
// Store the new RIP
|
||||
ExitFunction(NewRIP);
|
||||
ExitRelocatedPC(Op, TargetOffset);
|
||||
}
|
||||
|
||||
// Failure to take branch
|
||||
@@ -759,11 +749,8 @@ void OpDispatchBuilder::CondJUMPOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
StartNewBlock();
|
||||
|
||||
// Leave block
|
||||
auto RIPTargetConst = GetRelocatedPC(Op);
|
||||
|
||||
// Store the new RIP
|
||||
ExitFunction(RIPTargetConst);
|
||||
// Leave block & store the new RIP
|
||||
ExitRelocatedPC(Op);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -798,10 +785,8 @@ void OpDispatchBuilder::CondJUMPRCXOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
StartNewBlock();
|
||||
|
||||
auto NewRIP = GetRelocatedPC(Op, Op->Src[0].Data.Literal.Value);
|
||||
|
||||
// Store the new RIP
|
||||
ExitFunction(NewRIP);
|
||||
ExitRelocatedPC(Op, Op->Src[0].Data.Literal.Value);
|
||||
}
|
||||
|
||||
// Failure to take branch
|
||||
@@ -815,11 +800,8 @@ void OpDispatchBuilder::CondJUMPRCXOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
StartNewBlock();
|
||||
|
||||
// Leave block
|
||||
auto RIPTargetConst = GetRelocatedPC(Op);
|
||||
|
||||
// Store the new RIP
|
||||
ExitFunction(RIPTargetConst);
|
||||
// Leave block & store the new RIP
|
||||
ExitRelocatedPC(Op);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -844,7 +826,7 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
uint64_t Target = Op->PC + Op->InstSize + Op->Src[1].Literal();
|
||||
|
||||
Ref CondReg = LoadSource_WithOpSize(GPRClass, Op, Op->Src[0], SrcSize, Op->Flags);
|
||||
CondReg = _Sub(OpSize, CondReg, _InlineConstant(1));
|
||||
CondReg = Sub(OpSize, CondReg, 1);
|
||||
StoreResult(GPRClass, Op, Op->Src[0], CondReg, OpSize::iInvalid);
|
||||
|
||||
// If LOOPE then jumps to target if RCX != 0 && ZF == 1
|
||||
@@ -872,10 +854,8 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
StartNewBlock();
|
||||
|
||||
auto NewRIP = GetRelocatedPC(Op, Op->Src[1].Data.Literal.Value);
|
||||
|
||||
// Store the new RIP
|
||||
ExitFunction(NewRIP);
|
||||
ExitRelocatedPC(Op, Op->Src[1].Data.Literal.Value);
|
||||
}
|
||||
|
||||
// Failure to take branch
|
||||
@@ -889,11 +869,8 @@ void OpDispatchBuilder::LoopOp(OpcodeArgs) {
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
StartNewBlock();
|
||||
|
||||
// Leave block
|
||||
auto RIPTargetConst = GetRelocatedPC(Op);
|
||||
|
||||
// Store the new RIP
|
||||
ExitFunction(RIPTargetConst);
|
||||
// Leave block & store the new RIP
|
||||
ExitRelocatedPC(Op);
|
||||
}
|
||||
}
|
||||
}
|
||||
@@ -937,10 +914,10 @@ void OpDispatchBuilder::JUMPOp(OpcodeArgs) {
|
||||
SetJumpTarget(Jump_, JumpTarget);
|
||||
SetCurrentCodeBlock(JumpTarget);
|
||||
StartNewBlock();
|
||||
ExitFunction(GetRelocatedPC(Op, TargetOffset));
|
||||
ExitRelocatedPC(Op, TargetOffset);
|
||||
}
|
||||
} else {
|
||||
ExitFunction(GetRelocatedPC(Op, TargetOffset));
|
||||
ExitRelocatedPC(Op, TargetOffset);
|
||||
}
|
||||
}
|
||||
|
||||
@@ -1416,7 +1393,7 @@ void OpDispatchBuilder::SHLDOp(OpcodeArgs) {
|
||||
|
||||
// a64 masks the bottom bits, so if we're using a native 32/64-bit shift, we
|
||||
// can negate to do the subtract (it's congruent), which saves a constant.
|
||||
auto ShiftRight = Size >= 32 ? _Neg(OpSize::i64Bit, Shift) : _Sub(OpSize::i64Bit, Constant(Size), Shift);
|
||||
auto ShiftRight = Size >= 32 ? _Neg(OpSize::i64Bit, Shift) : Sub(OpSize::i64Bit, Constant(Size), Shift);
|
||||
|
||||
auto Tmp1 = _Lshl(OpSize::i64Bit, Dest, Shift);
|
||||
auto Tmp2 = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Src, ShiftRight);
|
||||
@@ -1487,7 +1464,7 @@ void OpDispatchBuilder::SHRDOp(OpcodeArgs) {
|
||||
Shift = _And(OpSize::i64Bit, Shift, _InlineConstant(0x1F));
|
||||
}
|
||||
|
||||
auto ShiftLeft = _Sub(OpSize::i64Bit, Constant(Size), Shift);
|
||||
auto ShiftLeft = Sub(OpSize::i64Bit, Constant(Size), Shift);
|
||||
|
||||
auto Tmp1 = _Lshr(Size == 64 ? OpSize::i64Bit : OpSize::i32Bit, Dest, Shift);
|
||||
auto Tmp2 = _Lshl(OpSize::i64Bit, Src, ShiftLeft);
|
||||
@@ -1701,15 +1678,13 @@ void OpDispatchBuilder::BLSMSKBMIOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
auto* Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Result = _Xor(Size, _Sub(Size, Src, _InlineConstant(1)), Src);
|
||||
auto Result = _Xor(Size, Sub(Size, Src, 1), Src);
|
||||
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
InvalidatePF_AF();
|
||||
|
||||
// CF set according to the Src
|
||||
auto Zero = Constant(0);
|
||||
auto One = Constant(1);
|
||||
auto CFInv = _Select(IR::COND_NEQ, Src, Zero, One, Zero);
|
||||
auto CFInv = To01(OpSize::i64Bit, Src);
|
||||
|
||||
// The output of BLSMSK is always nonzero, so TST will clear Z (along with C
|
||||
// and O) while setting S.
|
||||
@@ -1723,13 +1698,11 @@ void OpDispatchBuilder::BLSRBMIOp(OpcodeArgs) {
|
||||
const auto Size = OpSizeFromSrc(Op);
|
||||
|
||||
auto* Src = LoadSource(GPRClass, Op, Op->Src[0], Op->Flags, {.AllowUpperGarbage = true});
|
||||
auto Result = _And(Size, _Sub(Size, Src, _InlineConstant(1)), Src);
|
||||
auto Result = _And(Size, Sub(Size, Src, 1), Src);
|
||||
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
|
||||
auto Zero = Constant(0);
|
||||
auto One = Constant(1);
|
||||
auto CFInv = _Select(IR::COND_NEQ, Src, Zero, One, Zero);
|
||||
auto CFInv = To01(OpSize::i64Bit, Src);
|
||||
|
||||
SetNZ_ZeroCV(Size, Result);
|
||||
SetCFInverted(CFInv);
|
||||
@@ -1788,9 +1761,7 @@ void OpDispatchBuilder::BZHI(OpcodeArgs) {
|
||||
auto Result = _NZCVSelect(Size, {COND_NEQ}, Src, MaskResult);
|
||||
StoreResult(GPRClass, Op, Result, OpSize::iInvalid);
|
||||
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto CFInv = _NZCVSelect(OpSize::i32Bit, {COND_EQ}, One, Zero);
|
||||
auto CFInv = _NZCVSelect01({COND_EQ});
|
||||
|
||||
InvalidatePF_AF();
|
||||
SetNZ_ZeroCV(Size, Result);
|
||||
@@ -2050,7 +2021,7 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
|
||||
|
||||
// Our new CF will be bit (Shift - 1) of the source. this is hoisted up to
|
||||
// avoid the need to copy the source. Again, the Lshr absorbs the masking.
|
||||
auto NewCF = _Lshr(OpSize, Dest, _Sub(OpSize, Src, _InlineConstant(1)));
|
||||
auto NewCF = _Lshr(OpSize, Dest, Sub(OpSize, Src, 1));
|
||||
SetCFDirect(NewCF, 0, true);
|
||||
|
||||
// Since shift != 0 we can inject the CF
|
||||
@@ -2154,7 +2125,7 @@ void OpDispatchBuilder::RCRSmallerOp(OpcodeArgs) {
|
||||
if (Src.IsConstant) {
|
||||
SetCFDirect(Tmp, (Src.C & 0x1f) - 1, true);
|
||||
} else {
|
||||
auto NewCF = _Lshr(OpSize::i32Bit, Tmp, _Sub(OpSize::i32Bit, Src.Ref(), _InlineConstant(1)));
|
||||
auto NewCF = _Lshr(OpSize::i32Bit, Tmp, Sub(OpSize::i32Bit, Src.Ref(), 1));
|
||||
SetCFDirect(NewCF, 0, true);
|
||||
}
|
||||
|
||||
@@ -2264,7 +2235,7 @@ void OpDispatchBuilder::RCLOp(OpcodeArgs) {
|
||||
SetCFDirect(NewCF, 0, true);
|
||||
|
||||
// Since Shift != 0 we can inject the CF. Shift absorbs the masking.
|
||||
Ref CFShl = _Sub(OpSize, Src, _InlineConstant(1));
|
||||
Ref CFShl = Sub(OpSize, Src, 1);
|
||||
auto TmpCF = _Lshl(OpSize, CF, CFShl);
|
||||
Res = _Or(OpSize, Res, TmpCF);
|
||||
|
||||
@@ -2762,7 +2733,7 @@ void OpDispatchBuilder::PopcountOp(OpcodeArgs) {
|
||||
|
||||
Ref OpDispatchBuilder::CalculateAFForDecimal(Ref A) {
|
||||
auto Nibble = _And(OpSize::i64Bit, A, Constant(0xF));
|
||||
auto Greater = _Select(FEXCore::IR::COND_UGT, Nibble, Constant(9), Constant(1), Constant(0));
|
||||
auto Greater = Select01(OpSize::i64Bit, CondClassType {COND_UGT}, Nibble, Constant(9));
|
||||
|
||||
return _Or(OpSize::i64Bit, LoadAF(), Greater);
|
||||
}
|
||||
@@ -2774,13 +2745,13 @@ void OpDispatchBuilder::DAAOp(OpcodeArgs) {
|
||||
auto AF = CalculateAFForDecimal(AL);
|
||||
|
||||
// CF |= (AL > 0x99);
|
||||
CFInv = _And(OpSize::i64Bit, CFInv, _Select(FEXCore::IR::COND_ULE, AL, Constant(0x99), Constant(1), Constant(0)));
|
||||
CFInv = _And(OpSize::i64Bit, CFInv, Select01(OpSize::i64Bit, CondClassType {COND_ULE}, AL, Constant(0x99)));
|
||||
|
||||
// AL = AF ? (AL + 0x6) : AL;
|
||||
AL = _Select(FEXCore::IR::COND_NEQ, AF, Constant(0), _Add(OpSize::i64Bit, AL, Constant(0x6)), AL);
|
||||
AL = _Select(FEXCore::IR::COND_NEQ, AF, Constant(0), Add(OpSize::i64Bit, AL, 0x6), AL);
|
||||
|
||||
// AL = CF ? (AL + 0x60) : AL;
|
||||
AL = _Select(FEXCore::IR::COND_EQ, CFInv, Constant(0), _Add(OpSize::i64Bit, AL, Constant(0x60)), AL);
|
||||
AL = _Select(FEXCore::IR::COND_EQ, CFInv, Constant(0), Add(OpSize::i64Bit, AL, 0x60), AL);
|
||||
|
||||
// SF, ZF, PF set according to result. CF set per above. OF undefined.
|
||||
StoreGPRRegister(X86State::REG_RAX, AL, OpSize::i8Bit);
|
||||
@@ -2797,16 +2768,16 @@ void OpDispatchBuilder::DASOp(OpcodeArgs) {
|
||||
auto AF = CalculateAFForDecimal(AL);
|
||||
|
||||
// CF |= (AL > 0x99);
|
||||
CF = _Or(OpSize::i64Bit, CF, _Select(FEXCore::IR::COND_UGT, AL, Constant(0x99), Constant(1), Constant(0)));
|
||||
CF = _Or(OpSize::i64Bit, CF, Select01(OpSize::i64Bit, CondClassType {COND_UGT}, AL, Constant(0x99)));
|
||||
|
||||
// NewCF = CF | (AF && (Borrow from AL - 6))
|
||||
auto NewCF = _Or(OpSize::i32Bit, CF, _Select(FEXCore::IR::COND_ULT, AL, Constant(6), AF, CF));
|
||||
|
||||
// AL = AF ? (AL - 0x6) : AL;
|
||||
AL = _Select(FEXCore::IR::COND_NEQ, AF, Constant(0), _Sub(OpSize::i64Bit, AL, Constant(0x6)), AL);
|
||||
AL = _Select(FEXCore::IR::COND_NEQ, AF, Constant(0), Sub(OpSize::i64Bit, AL, 0x6), AL);
|
||||
|
||||
// AL = CF ? (AL - 0x60) : AL;
|
||||
AL = _Select(FEXCore::IR::COND_NEQ, CF, Constant(0), _Sub(OpSize::i64Bit, AL, Constant(0x60)), AL);
|
||||
AL = _Select(FEXCore::IR::COND_NEQ, CF, Constant(0), Sub(OpSize::i64Bit, AL, 0x60), AL);
|
||||
|
||||
// SF, ZF, PF set according to result. CF set per above. OF undefined.
|
||||
StoreGPRRegister(X86State::REG_RAX, AL, OpSize::i8Bit);
|
||||
@@ -2826,7 +2797,7 @@ void OpDispatchBuilder::AAAOp(OpcodeArgs) {
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// AX = CF ? (AX + 0x106) : 0
|
||||
A = NZCVSelect(OpSize::i32Bit, {COND_UGE} /* CF = 1 */, _Add(OpSize::i32Bit, A, Constant(0x106)), A);
|
||||
A = NZCVSelect(OpSize::i32Bit, {COND_UGE} /* CF = 1 */, Add(OpSize::i32Bit, A, 0x106), A);
|
||||
|
||||
// AL = AL & 0x0F
|
||||
A = _And(OpSize::i32Bit, A, Constant(0xFF0F));
|
||||
@@ -2843,7 +2814,7 @@ void OpDispatchBuilder::AASOp(OpcodeArgs) {
|
||||
CalculateDeferredFlags();
|
||||
|
||||
// AX = CF ? (AX - 0x106) : 0
|
||||
A = NZCVSelect(OpSize::i32Bit, {COND_UGE} /* CF = 1 */, _Sub(OpSize::i32Bit, A, Constant(0x106)), A);
|
||||
A = NZCVSelect(OpSize::i32Bit, {COND_UGE} /* CF = 1 */, Sub(OpSize::i32Bit, A, 0x106), A);
|
||||
|
||||
// AL = AL & 0x0F
|
||||
A = _And(OpSize::i32Bit, A, Constant(0xFF0F));
|
||||
@@ -2868,7 +2839,7 @@ void OpDispatchBuilder::AADOp(OpcodeArgs) {
|
||||
auto A = LoadGPRRegister(X86State::REG_RAX);
|
||||
auto AH = _Lshr(OpSize::i32Bit, A, Constant(8));
|
||||
auto Imm8 = Constant(Op->Src[0].Data.Literal.Value & 0xFF);
|
||||
auto NewAL = _Add(OpSize::i64Bit, A, _Mul(OpSize::i64Bit, AH, Imm8));
|
||||
auto NewAL = Add(OpSize::i64Bit, A, _Mul(OpSize::i64Bit, AH, Imm8));
|
||||
auto Result = _And(OpSize::i64Bit, NewAL, Constant(0xFF));
|
||||
StoreGPRRegister(X86State::REG_RAX, Result, OpSize::i16Bit);
|
||||
|
||||
@@ -2937,14 +2908,13 @@ void OpDispatchBuilder::EnterOp(OpcodeArgs) {
|
||||
|
||||
if (Level > 0) {
|
||||
for (uint8_t i = 1; i < Level; ++i) {
|
||||
auto Offset = Constant(i * IR::OpSizeToSize(GPRSize));
|
||||
auto MemLoc = _Sub(GPRSize, OldBP, Offset);
|
||||
auto MemLoc = Sub(GPRSize, OldBP, i * IR::OpSizeToSize(GPRSize));
|
||||
auto Mem = _LoadMem(GPRClass, GPRSize, MemLoc, GPRSize);
|
||||
NewSP = PushValue(GPRSize, Mem);
|
||||
}
|
||||
NewSP = PushValue(GPRSize, temp_RBP);
|
||||
}
|
||||
NewSP = _Sub(GPRSize, NewSP, Constant(AllocSpace));
|
||||
NewSP = Sub(GPRSize, NewSP, AllocSpace);
|
||||
StoreGPRRegister(X86State::REG_RSP, NewSP);
|
||||
StoreGPRRegister(X86State::REG_RBP, temp_RBP);
|
||||
}
|
||||
@@ -3074,7 +3044,7 @@ void OpDispatchBuilder::INCOp(OpcodeArgs) {
|
||||
|
||||
if (Size < 32 && CTX->HostFeatures.SupportsFlagM) {
|
||||
// Addition producing upper garbage
|
||||
Result = _Add(OpSize::i32Bit, Dest, _InlineConstant(1));
|
||||
Result = Add(OpSize::i32Bit, Dest, 1);
|
||||
CalculatePF(Result);
|
||||
CalculateAF(Dest, Constant(1));
|
||||
|
||||
@@ -3115,7 +3085,7 @@ void OpDispatchBuilder::DECOp(OpcodeArgs) {
|
||||
|
||||
if (Size < 32 && CTX->HostFeatures.SupportsFlagM) {
|
||||
// Subtraction producing upper garbage
|
||||
Result = _Sub(OpSize::i32Bit, Dest, _InlineConstant(1));
|
||||
Result = Sub(OpSize::i32Bit, Dest, 1);
|
||||
CalculatePF(Result);
|
||||
CalculateAF(Dest, Constant(1));
|
||||
|
||||
@@ -3195,11 +3165,11 @@ void OpDispatchBuilder::MOVSOp(OpcodeArgs) {
|
||||
auto SrcSegment = GetSegment(Op->Flags, FEXCore::X86Tables::DecodeFlags::FLAG_DS_PREFIX);
|
||||
|
||||
if (DstSegment) {
|
||||
DstAddr = _Add(OpSize::i64Bit, DstAddr, DstSegment);
|
||||
DstAddr = Add(OpSize::i64Bit, DstAddr, DstSegment);
|
||||
}
|
||||
|
||||
if (SrcSegment) {
|
||||
SrcAddr = _Add(OpSize::i64Bit, SrcAddr, SrcSegment);
|
||||
SrcAddr = Add(OpSize::i64Bit, SrcAddr, SrcSegment);
|
||||
}
|
||||
|
||||
Ref Result_Src = _AllocateGPR(false);
|
||||
@@ -3207,11 +3177,11 @@ void OpDispatchBuilder::MOVSOp(OpcodeArgs) {
|
||||
_MemCpy(CTX->IsAtomicTSOEnabled(), Size, DstAddr, SrcAddr, Counter, LoadDir(1), Result_Dst, Result_Src);
|
||||
|
||||
if (DstSegment) {
|
||||
Result_Dst = _Sub(OpSize::i64Bit, Result_Dst, DstSegment);
|
||||
Result_Dst = Sub(OpSize::i64Bit, Result_Dst, DstSegment);
|
||||
}
|
||||
|
||||
if (SrcSegment) {
|
||||
Result_Src = _Sub(OpSize::i64Bit, Result_Src, SrcSegment);
|
||||
Result_Src = Sub(OpSize::i64Bit, Result_Src, SrcSegment);
|
||||
}
|
||||
|
||||
StoreGPRRegister(X86State::REG_RCX, Constant(0));
|
||||
@@ -3227,8 +3197,8 @@ void OpDispatchBuilder::MOVSOp(OpcodeArgs) {
|
||||
_StoreMemAutoTSO(GPRClass, Size, RDI, Src, Size);
|
||||
|
||||
auto PtrDir = LoadDir(IR::OpSizeToSize(Size));
|
||||
RSI = _Add(OpSize::i64Bit, RSI, PtrDir);
|
||||
RDI = _Add(OpSize::i64Bit, RDI, PtrDir);
|
||||
RSI = Add(OpSize::i64Bit, RSI, PtrDir);
|
||||
RDI = Add(OpSize::i64Bit, RDI, PtrDir);
|
||||
|
||||
StoreGPRRegister(X86State::REG_RSI, RSI);
|
||||
StoreGPRRegister(X86State::REG_RDI, RDI);
|
||||
@@ -3259,11 +3229,11 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
auto PtrDir = LoadDir(IR::OpSizeToSize(Size));
|
||||
|
||||
// Offset the pointer
|
||||
Dest_RDI = _Add(OpSize::i64Bit, Dest_RDI, PtrDir);
|
||||
Dest_RDI = Add(OpSize::i64Bit, Dest_RDI, PtrDir);
|
||||
StoreGPRRegister(X86State::REG_RDI, Dest_RDI);
|
||||
|
||||
// Offset second pointer
|
||||
Dest_RSI = _Add(OpSize::i64Bit, Dest_RSI, PtrDir);
|
||||
Dest_RSI = Add(OpSize::i64Bit, Dest_RSI, PtrDir);
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI);
|
||||
} else {
|
||||
// Calculate flags early.
|
||||
@@ -3307,17 +3277,17 @@ void OpDispatchBuilder::CMPSOp(OpcodeArgs) {
|
||||
Ref TailCounter = LoadGPRRegister(X86State::REG_RCX);
|
||||
|
||||
// Decrement counter
|
||||
TailCounter = _SubWithFlags(OpSize::i64Bit, TailCounter, Constant(1));
|
||||
TailCounter = SubWithFlags(OpSize::i64Bit, TailCounter, 1);
|
||||
|
||||
// Store the counter since we don't have phis
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
Dest_RDI = _Add(OpSize::i64Bit, Dest_RDI, Constant(PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
Dest_RDI = Add(OpSize::i64Bit, Dest_RDI, PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size)));
|
||||
StoreGPRRegister(X86State::REG_RDI, Dest_RDI);
|
||||
|
||||
// Offset second pointer
|
||||
Dest_RSI = _Add(OpSize::i64Bit, Dest_RSI, Constant(PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
Dest_RSI = Add(OpSize::i64Bit, Dest_RSI, PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size)));
|
||||
StoreGPRRegister(X86State::REG_RSI, Dest_RSI);
|
||||
|
||||
// If TailCounter != 0, compare sources.
|
||||
@@ -3417,13 +3387,13 @@ void OpDispatchBuilder::LODSOp(OpcodeArgs) {
|
||||
Ref TailDest_RSI = LoadGPRRegister(X86State::REG_RSI);
|
||||
|
||||
// Decrement counter
|
||||
TailCounter = _Sub(OpSize::i64Bit, TailCounter, Constant(1));
|
||||
TailCounter = Sub(OpSize::i64Bit, TailCounter, 1);
|
||||
|
||||
// Store the counter since we don't have phis
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
TailDest_RSI = _Add(OpSize::i64Bit, TailDest_RSI, Constant(PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
TailDest_RSI = Add(OpSize::i64Bit, TailDest_RSI, PtrDir * static_cast<int32_t>(IR::OpSizeToSize(Size)));
|
||||
StoreGPRRegister(X86State::REG_RSI, TailDest_RSI);
|
||||
|
||||
// Jump back to the start, we have more work to do
|
||||
@@ -3501,13 +3471,13 @@ void OpDispatchBuilder::SCASOp(OpcodeArgs) {
|
||||
Ref TailDest_RDI = LoadGPRRegister(X86State::REG_RDI);
|
||||
|
||||
// Decrement counter
|
||||
TailCounter = _Sub(OpSize::i64Bit, TailCounter, Constant(1));
|
||||
TailCounter = Sub(OpSize::i64Bit, TailCounter, 1);
|
||||
|
||||
// Store the counter since we don't have phis
|
||||
StoreGPRRegister(X86State::REG_RCX, TailCounter);
|
||||
|
||||
// Offset the pointer
|
||||
TailDest_RDI = _Add(OpSize::i64Bit, TailDest_RDI, Constant(Dir * static_cast<int32_t>(IR::OpSizeToSize(Size))));
|
||||
TailDest_RDI = Add(OpSize::i64Bit, TailDest_RDI, Dir * static_cast<int32_t>(IR::OpSizeToSize(Size)));
|
||||
StoreGPRRegister(X86State::REG_RDI, TailDest_RDI);
|
||||
|
||||
CalculateDeferredFlags();
|
||||
@@ -3895,7 +3865,7 @@ void OpDispatchBuilder::Finalize() {
|
||||
|
||||
// We haven't emitted. Dump out to the dispatcher
|
||||
SetCurrentCodeBlock(Handler.second.BlockEntry);
|
||||
ExitFunction(_EntrypointOffset(GPRSize, Handler.first - Entry));
|
||||
ExitFunction(_InlineEntrypointOffset(GPRSize, Handler.first - Entry));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -3970,7 +3940,7 @@ Ref OpDispatchBuilder::GetSegment(uint32_t Flags, uint32_t DefaultPrefix, bool O
|
||||
Ref OpDispatchBuilder::AppendSegmentOffset(Ref Value, uint32_t Flags, uint32_t DefaultPrefix, bool Override) {
|
||||
auto Segment = GetSegment(Flags, DefaultPrefix, Override);
|
||||
if (Segment) {
|
||||
Value = _Add(std::max(OpSize::i32Bit, std::max(GetOpSize(Value), GetOpSize(Segment))), Value, Segment);
|
||||
Value = Add(std::max(OpSize::i32Bit, std::max(GetOpSize(Value), GetOpSize(Segment))), Value, Segment);
|
||||
}
|
||||
|
||||
return Value;
|
||||
@@ -4198,7 +4168,7 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
|
||||
} else {
|
||||
// For X87 extended doubles, Split the load.
|
||||
auto Res = _LoadMem(Class, OpSize::i64Bit, MemSrc, Align == OpSize::iInvalid ? OpSize : Align);
|
||||
return _VLoadVectorElement(OpSize::i128Bit, OpSize::i16Bit, Res, 4, _Add(OpSize::i64Bit, MemSrc, _InlineConstant(8)));
|
||||
return _VLoadVectorElement(OpSize::i128Bit, OpSize::i16Bit, Res, 4, Add(OpSize::i64Bit, MemSrc, 8));
|
||||
}
|
||||
}
|
||||
|
||||
@@ -4208,11 +4178,6 @@ Ref OpDispatchBuilder::LoadSource_WithOpSize(RegisterClassType Class, const X86T
|
||||
}
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
return _EntrypointOffset(GPRSize, Op->PC + Op->InstSize + Offset - Entry);
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::LoadGPRRegister(uint32_t GPR, IR::OpSize Size, uint8_t Offset, bool AllowUpperGarbage) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
if (Size == OpSize::iInvalid) {
|
||||
@@ -4351,7 +4316,7 @@ void OpDispatchBuilder::StoreResult(FEXCore::IR::RegisterClassType Class, FEXCor
|
||||
}
|
||||
|
||||
OpDispatchBuilder::OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx)
|
||||
: IREmitter {ctx->OpDispatcherAllocator}
|
||||
: IREmitter {ctx->OpDispatcherAllocator, ctx->HostFeatures.SupportsTSOImm9}
|
||||
, CTX {ctx} {
|
||||
ResetWorkingList();
|
||||
InstallHostSpecificOpcodeHandlers();
|
||||
@@ -4366,9 +4331,6 @@ OpDispatchBuilder::OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx)
|
||||
DefaultAVXStateFunc = &OpDispatchBuilder::AVX128_DefaultAVXState;
|
||||
}
|
||||
}
|
||||
OpDispatchBuilder::OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator& Allocator)
|
||||
: IREmitter {Allocator}
|
||||
, CTX {nullptr} {}
|
||||
|
||||
void OpDispatchBuilder::ResetWorkingList() {
|
||||
IREmitter::ResetWorkingList();
|
||||
@@ -4409,7 +4371,7 @@ void OpDispatchBuilder::ALUOp(OpcodeArgs, FEXCore::IR::IROps ALUIROp, FEXCore::I
|
||||
auto Zero = LoadGPR(Op->Dest.Data.GPR.GPR);
|
||||
HandleNZ00Write();
|
||||
InvalidateAF();
|
||||
CalculatePF(_SubWithFlags(OpSize::i32Bit, Zero, Zero));
|
||||
CalculatePF(SubWithFlags(OpSize::i32Bit, Zero, Zero));
|
||||
CFInverted = true;
|
||||
FlushRegisterCache();
|
||||
|
||||
|
||||
@@ -201,8 +201,7 @@ public:
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
// If we don't have a jump target to a new block then we have to leave
|
||||
// Set the RIP to the next instruction and leave
|
||||
auto RelocatedNextRIP = _EntrypointOffset(GPRSize, NextRIP - Entry);
|
||||
ExitFunction(RelocatedNextRIP);
|
||||
ExitFunction(_InlineEntrypointOffset(GPRSize, NextRIP - Entry));
|
||||
} else if (it != JumpTargets.end()) {
|
||||
Jump(it->second.BlockEntry);
|
||||
return true;
|
||||
@@ -266,7 +265,6 @@ public:
|
||||
}
|
||||
|
||||
OpDispatchBuilder(FEXCore::Context::ContextImpl* ctx);
|
||||
OpDispatchBuilder(FEXCore::Utils::IntrusivePooledAllocator& Allocator);
|
||||
|
||||
void ResetWorkingList();
|
||||
void ResetDecodeFailure() {
|
||||
@@ -1189,6 +1187,33 @@ public:
|
||||
}
|
||||
}
|
||||
|
||||
void StoreContextHelper(IR::OpSize Size, RegisterClassType Class, Ref Value, uint32_t Offset) {
|
||||
// For i128Bit, we won't see a normal Constant to inline, but as a special
|
||||
// case we can replace with a 2x64-bit store which can use inline zeroes.
|
||||
if (Size == OpSize::i128Bit) {
|
||||
auto Header = GetOpHeader(WrapNode(Value));
|
||||
const auto MAX_STP_OFFSET = (252 * 4);
|
||||
|
||||
if (Offset <= MAX_STP_OFFSET && Header->Op == OP_LOADNAMEDVECTORCONSTANT) {
|
||||
auto Const = Header->C<IR::IROp_LoadNamedVectorConstant>();
|
||||
|
||||
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
|
||||
Ref Zero = _Constant(0);
|
||||
Ref STP = _StoreContextPair(IR::OpSize::i64Bit, GPRClass, Zero, Zero, Offset);
|
||||
|
||||
// XXX: This works around InlineConstant not having an associated
|
||||
// register class, else we'd just do InlineConstant above.
|
||||
Ref InlineZero = _InlineConstant(0);
|
||||
ReplaceNodeArgument(STP, 0, InlineZero);
|
||||
ReplaceNodeArgument(STP, 1, InlineZero);
|
||||
return;
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
_StoreContext(Size, Class, Value, Offset);
|
||||
}
|
||||
|
||||
void FlushRegisterCache(bool SRAOnly = false) {
|
||||
// At block boundaries, fix up the carry flag.
|
||||
if (!SRAOnly) {
|
||||
@@ -1254,7 +1279,7 @@ public:
|
||||
_StoreContextPair(Size, Class, ValueNext, Value, Offset - SizeInt);
|
||||
Bits &= ~NextBit;
|
||||
} else {
|
||||
_StoreContext(Size, Class, Value, Offset);
|
||||
StoreContextHelper(Size, Class, Value, Offset);
|
||||
// If Partial and MMX register, then we need to store all 1s in bits 64-80
|
||||
if (Partial && Index >= MM0Index && Index <= MM7Index) {
|
||||
_StoreContext(OpSize::i16Bit, IR::GPRClass, Constant(0xFFFF), Offset + 8);
|
||||
@@ -1521,7 +1546,23 @@ private:
|
||||
void StoreGPRRegister(uint32_t GPR, const Ref Src, IR::OpSize Size = OpSize::iInvalid, uint8_t Offset = 0);
|
||||
void StoreXMMRegister(uint32_t XMM, const Ref Src);
|
||||
|
||||
Ref GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0);
|
||||
Ref _GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, bool Inline) {
|
||||
const auto GPRSize = GetGPROpSize();
|
||||
const auto Offs = Op->PC + Op->InstSize + Offset - Entry;
|
||||
return Inline ? _InlineEntrypointOffset(GPRSize, Offs) : _EntrypointOffset(GPRSize, Offs);
|
||||
}
|
||||
|
||||
Ref GetRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
|
||||
return _GetRelocatedPC(Op, Offset, false);
|
||||
}
|
||||
|
||||
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset = 0) {
|
||||
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */));
|
||||
}
|
||||
|
||||
void ExitRelocatedPC(const FEXCore::X86Tables::DecodedOp& Op, int64_t Offset, BranchHint Hint, Ref CallReturnAddress, Ref CallReturnBlock) {
|
||||
ExitFunction(_GetRelocatedPC(Op, Offset, true /* Inline */), Hint, CallReturnAddress, CallReturnBlock);
|
||||
}
|
||||
|
||||
[[nodiscard]]
|
||||
static bool IsOperandMem(const X86Tables::DecodedOperand& Operand, bool Load) {
|
||||
@@ -1650,7 +1691,7 @@ private:
|
||||
// This is currently worse for 8/16-bit, but that should be optimized. TODO
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
if (SetPF) {
|
||||
CalculatePF(_SubWithFlags(SrcSize, Res, Constant(0)));
|
||||
CalculatePF(SubWithFlags(SrcSize, Res, (uint64_t)0));
|
||||
} else {
|
||||
_SubNZCV(SrcSize, Res, Constant(0));
|
||||
}
|
||||
@@ -2036,7 +2077,7 @@ private:
|
||||
} else {
|
||||
// Because we explicitly inverted for CF above, we use the unsafe
|
||||
// _NZCVSelect rather than the safe CF-aware version.
|
||||
return _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(BitOffset, Invert), Constant(1), Constant(0));
|
||||
return _NZCVSelect01(CondForNZCVBit(BitOffset, Invert));
|
||||
}
|
||||
} else if (BitOffset == FEXCore::X86State::RFLAG_PF_RAW_LOC) {
|
||||
return LoadGPR(Core::CPUState::PF_AS_GREG);
|
||||
@@ -2133,7 +2174,7 @@ private:
|
||||
void ConvertNZCVToX87() {
|
||||
LOGMAN_THROW_A_FMT(NZCVDirty && CachedNZCV, "NZCV must be saved");
|
||||
|
||||
Ref V = _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(FEXCore::X86State::RFLAG_OF_RAW_LOC, false), Constant(1), Constant(0));
|
||||
Ref V = _NZCVSelect01(CondForNZCVBit(FEXCore::X86State::RFLAG_OF_RAW_LOC, false));
|
||||
|
||||
if (CTX->HostFeatures.SupportsFlagM2) {
|
||||
// Convert to x86 flags, saves us from or'ing after.
|
||||
@@ -2141,8 +2182,8 @@ private:
|
||||
}
|
||||
|
||||
// CF is inverted after FCMP
|
||||
Ref C = _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(FEXCore::X86State::RFLAG_CF_RAW_LOC, true), Constant(1), Constant(0));
|
||||
Ref Z = _NZCVSelect(OpSize::i32Bit, CondForNZCVBit(FEXCore::X86State::RFLAG_ZF_RAW_LOC, false), Constant(1), Constant(0));
|
||||
Ref C = _NZCVSelect01(CondForNZCVBit(FEXCore::X86State::RFLAG_CF_RAW_LOC, true));
|
||||
Ref Z = _NZCVSelect01(CondForNZCVBit(FEXCore::X86State::RFLAG_ZF_RAW_LOC, false));
|
||||
|
||||
if (!CTX->HostFeatures.SupportsFlagM2) {
|
||||
C = _Or(OpSize::i32Bit, C, V);
|
||||
@@ -2238,7 +2279,7 @@ private:
|
||||
|
||||
std::optional<CondClassType> DecodeNZCVCondition(uint8_t OP);
|
||||
Ref SelectBit(Ref Cmp, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
Ref SelectCC(uint8_t OP, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
Ref SelectCC0All1(uint8_t OP);
|
||||
|
||||
/**
|
||||
* @brief Flushes NZCV. Mostly vestigial.
|
||||
@@ -2313,7 +2354,6 @@ private:
|
||||
* @name These functions are used by the deferred flag handling while it is calculating and storing flags in to RFLAGs.
|
||||
* @{ */
|
||||
Ref LoadPFRaw(bool Mask, bool Invert);
|
||||
Ref SelectPF(bool Invert, IR::OpSize ResultSize, Ref TrueValue, Ref FalseValue);
|
||||
Ref LoadAF();
|
||||
void FixupAF();
|
||||
void SetAFAndFixup(Ref AF);
|
||||
@@ -2551,7 +2591,7 @@ private:
|
||||
}
|
||||
|
||||
ArithRef Presub(uint64_t K) {
|
||||
return IsConstant ? ArithRef(E, K - C) : ArithRef(E, E->_Sub(OpSize::i64Bit, E->Constant(K), R));
|
||||
return IsConstant ? ArithRef(E, K - C) : ArithRef(E, E->Sub(OpSize::i64Bit, E->Constant(K), R));
|
||||
}
|
||||
|
||||
ArithRef Lshl(uint64_t Shift) {
|
||||
|
||||
@@ -2212,8 +2212,6 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs) {
|
||||
// For 256-bit, we need to split up the operation. This is nontrivial.
|
||||
// Let's go the simple route here.
|
||||
Ref ZF, CFInv;
|
||||
Ref ZeroConst = Constant(0);
|
||||
Ref OneConst = Constant(1);
|
||||
|
||||
const auto ElementSizeInBits = IR::OpSizeAsBits(ElementSize);
|
||||
|
||||
@@ -2255,7 +2253,7 @@ void OpDispatchBuilder::AVX128_VTESTP(OpcodeArgs) {
|
||||
|
||||
// ExtGPR will either be [0, 8] or [0, 16] If 0 then set Flag.
|
||||
auto ExtGPR = _VExtractToGPR(OpSize::i128Bit, ElementSize, AddWide, 0);
|
||||
CFInv = _Select(IR::COND_NEQ, ExtGPR, ZeroConst, OneConst, ZeroConst);
|
||||
CFInv = To01(OpSize::i64Bit, ExtGPR);
|
||||
}
|
||||
|
||||
// As in PTest, this sets Z appropriately while zeroing the rest of NZCV.
|
||||
@@ -2294,10 +2292,7 @@ void OpDispatchBuilder::AVX128_PTest(OpcodeArgs) {
|
||||
Test1 = _VExtractToGPR(OpSize::i128Bit, OpSize::i16Bit, Test1, 0);
|
||||
Test2 = _VExtractToGPR(OpSize::i128Bit, OpSize::i16Bit, Test2, 0);
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
auto OneConst = Constant(1);
|
||||
|
||||
Test2 = _Select(FEXCore::IR::COND_NEQ, Test2, ZeroConst, OneConst, ZeroConst);
|
||||
Test2 = To01(OpSize::i64Bit, Test2);
|
||||
|
||||
// Careful, these flags are different between {V,}PTEST and VTESTP{S,D}
|
||||
// Set ZF according to Test1. SF will be zeroed since we do a 32-bit test on
|
||||
@@ -2518,7 +2513,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherImpl(OpSize Size, O
|
||||
///< BaseAddr doesn't need to exist, calculate that here.
|
||||
Ref BaseAddr = VSIB.BaseAddr;
|
||||
if (BaseAddr && VSIB.Displacement) {
|
||||
BaseAddr = _Add(OpSize::i64Bit, BaseAddr, Constant(VSIB.Displacement));
|
||||
BaseAddr = Add(OpSize::i64Bit, BaseAddr, VSIB.Displacement);
|
||||
} else if (VSIB.Displacement) {
|
||||
BaseAddr = Constant(VSIB.Displacement);
|
||||
} else if (!BaseAddr) {
|
||||
@@ -2613,7 +2608,7 @@ OpDispatchBuilder::RefPair OpDispatchBuilder::AVX128_VPGatherQPSImpl(Ref Dest, R
|
||||
///< BaseAddr doesn't need to exist, calculate that here.
|
||||
Ref BaseAddr = VSIB.BaseAddr;
|
||||
if (BaseAddr && VSIB.Displacement) {
|
||||
BaseAddr = _Add(OpSize::i64Bit, BaseAddr, Constant(VSIB.Displacement));
|
||||
BaseAddr = Add(OpSize::i64Bit, BaseAddr, VSIB.Displacement);
|
||||
} else if (VSIB.Displacement) {
|
||||
BaseAddr = Constant(VSIB.Displacement);
|
||||
} else if (!BaseAddr) {
|
||||
|
||||
@@ -201,7 +201,7 @@ void OpDispatchBuilder::FixupAF() {
|
||||
auto PFRaw = GetRFLAG(FEXCore::X86State::RFLAG_PF_RAW_LOC);
|
||||
auto AFRaw = GetRFLAG(FEXCore::X86State::RFLAG_AF_RAW_LOC);
|
||||
|
||||
// Again 64-bit as masking is more expensive given our ConstProp design.
|
||||
// Again 64-bit as masking is more expensive.
|
||||
Ref XorRes = _Xor(OpSize::i64Bit, AFRaw, PFRaw);
|
||||
SetRFLAG<FEXCore::X86State::RFLAG_AF_RAW_LOC>(XorRes);
|
||||
}
|
||||
@@ -267,8 +267,6 @@ Ref OpDispatchBuilder::IncrementByCarry(OpSize OpSize, Ref Src) {
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
Ref Res;
|
||||
|
||||
@@ -288,11 +286,11 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
Ref Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
// Need to zero-extend for the comparison.
|
||||
Res = _Add(OpSize, Src1, Src2PlusCF);
|
||||
Res = Add(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
// TODO: We can fold that second Bfe in (cmp uxth).
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Res, Src2PlusCF, One, Zero);
|
||||
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Res, Src2PlusCF);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetCFInverted(SelectCFInv);
|
||||
@@ -304,8 +302,6 @@ Ref OpDispatchBuilder::CalculateFlags_ADC(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
}
|
||||
|
||||
Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2) {
|
||||
auto Zero = _InlineConstant(0);
|
||||
auto One = _InlineConstant(1);
|
||||
auto OpSize = SrcSize == OpSize::i64Bit ? OpSize::i64Bit : OpSize::i32Bit;
|
||||
|
||||
CalculateAF(Src1, Src2);
|
||||
@@ -325,10 +321,10 @@ Ref OpDispatchBuilder::CalculateFlags_SBB(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
|
||||
auto Src2PlusCF = IncrementByCarry(OpSize, Src2);
|
||||
|
||||
Res = _Sub(OpSize, Src1, Src2PlusCF);
|
||||
Res = Sub(OpSize, Src1, Src2PlusCF);
|
||||
Res = _Bfe(OpSize, IR::OpSizeAsBits(SrcSize), 0, Res);
|
||||
|
||||
auto SelectCFInv = _Select(FEXCore::IR::COND_UGE, Src1, Src2PlusCF, One, Zero);
|
||||
auto SelectCFInv = Select01(OpSize, CondClassType {COND_UGE}, Src1, Src2PlusCF);
|
||||
|
||||
SetNZ_ZeroCV(SrcSize, Res);
|
||||
SetCFInverted(SelectCFInv);
|
||||
@@ -349,10 +345,10 @@ Ref OpDispatchBuilder::CalculateFlags_SUB(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
Res = _SubWithFlags(SrcSize, Src1, Src2);
|
||||
Res = SubWithFlags(SrcSize, Src1, Src2);
|
||||
} else {
|
||||
_SubNZCV(SrcSize, Src1, Src2);
|
||||
Res = _Sub(OpSize::i32Bit, Src1, Src2);
|
||||
Res = Sub(OpSize::i32Bit, Src1, Src2);
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
@@ -379,10 +375,10 @@ Ref OpDispatchBuilder::CalculateFlags_ADD(IR::OpSize SrcSize, Ref Src1, Ref Src2
|
||||
|
||||
Ref Res;
|
||||
if (SrcSize >= OpSize::i32Bit) {
|
||||
Res = _AddWithFlags(SrcSize, Src1, Src2);
|
||||
Res = AddWithFlags(SrcSize, Src1, Src2);
|
||||
} else {
|
||||
_AddNZCV(SrcSize, Src1, Src2);
|
||||
Res = _Add(OpSize::i32Bit, Src1, Src2);
|
||||
Res = Add(OpSize::i32Bit, Src1, Src2);
|
||||
}
|
||||
|
||||
CalculatePF(Res);
|
||||
|
||||
@@ -3950,10 +3950,7 @@ void OpDispatchBuilder::PTestOpImpl(OpSize Size, Ref Dest, Ref Src) {
|
||||
Test1 = _VExtractToGPR(Size, OpSize::i16Bit, Test1, 0);
|
||||
Test2 = _VExtractToGPR(Size, OpSize::i16Bit, Test2, 0);
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
auto OneConst = Constant(1);
|
||||
|
||||
Test2 = _Select(FEXCore::IR::COND_NEQ, Test2, ZeroConst, OneConst, ZeroConst);
|
||||
Test2 = To01(OpSize::i64Bit, Test2);
|
||||
|
||||
// Careful, these flags are different between {V,}PTEST and VTESTP{S,D}
|
||||
// Set ZF according to Test1. SF will be zeroed since we do a 32-bit test on
|
||||
@@ -3990,10 +3987,7 @@ void OpDispatchBuilder::VTESTOpImpl(OpSize SrcSize, IR::OpSize ElementSize, Ref
|
||||
Ref AndGPR = _VExtractToGPR(SrcSize, OpSize::i16Bit, MaxAnd, 0);
|
||||
Ref AndNotGPR = _VExtractToGPR(SrcSize, OpSize::i16Bit, MaxAndNot, 0);
|
||||
|
||||
Ref ZeroConst = Constant(0);
|
||||
Ref OneConst = Constant(1);
|
||||
|
||||
Ref CFInv = _Select(IR::COND_NEQ, AndNotGPR, ZeroConst, OneConst, ZeroConst);
|
||||
Ref CFInv = To01(OpSize::i64Bit, AndNotGPR);
|
||||
|
||||
// As in PTest, this sets Z appropriately while zeroing the rest of NZCV.
|
||||
SetNZ_ZeroCV(OpSize::i32Bit, AndGPR);
|
||||
@@ -5084,7 +5078,7 @@ void OpDispatchBuilder::VPGATHER(OpcodeArgs) {
|
||||
///< BaseAddr doesn't need to exist, calculate that here.
|
||||
Ref BaseAddr = VSIB.BaseAddr;
|
||||
if (BaseAddr && VSIB.Displacement) {
|
||||
BaseAddr = _Add(OpSize::i64Bit, BaseAddr, Constant(VSIB.Displacement));
|
||||
BaseAddr = Add(OpSize::i64Bit, BaseAddr, VSIB.Displacement);
|
||||
} else if (VSIB.Displacement) {
|
||||
BaseAddr = Constant(VSIB.Displacement);
|
||||
} else if (!BaseAddr) {
|
||||
|
||||
@@ -114,10 +114,10 @@ void OpDispatchBuilder::FILD(OpcodeArgs) {
|
||||
auto absolute = _Neg(OpSize::i64Bit, Data, CondClassType {COND_MI});
|
||||
|
||||
// left justify the absolute integer
|
||||
auto shift = _Sub(OpSize::i64Bit, Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
|
||||
auto shift = Sub(OpSize::i64Bit, Constant(63), _FindMSB(IR::OpSize::i64Bit, absolute));
|
||||
auto shifted = _Lshl(OpSize::i64Bit, absolute, shift);
|
||||
|
||||
auto adjusted_exponent = _Sub(OpSize::i64Bit, Constant(0x3fff + 63), shift);
|
||||
auto adjusted_exponent = Sub(OpSize::i64Bit, Constant(0x3fff + 63), shift);
|
||||
auto zeroed_exponent = _Select(COND_EQ, absolute, zero, zero, adjusted_exponent);
|
||||
auto upper = _Or(OpSize::i64Bit, sign, zeroed_exponent);
|
||||
|
||||
@@ -164,12 +164,12 @@ void OpDispatchBuilder::FIST(OpcodeArgs, bool Truncate) {
|
||||
// Check for NaN/Infinity: exponent = 0x7fff
|
||||
SaveNZCV();
|
||||
_TestNZ(OpSize::i64Bit, Exponent, Constant(0x7fff));
|
||||
Ref IsSpecial = _NZCVSelect(OpSize::i64Bit, {COND_EQ}, Constant(1), Constant(0));
|
||||
Ref IsSpecial = _NZCVSelect01({COND_EQ});
|
||||
|
||||
// For overflow detection, check if exponent indicates a value >= 2^15
|
||||
// Biased exponent for 2^15 is 0x3fff + 15 = 0x400e
|
||||
_SubWithFlags(OpSize::i64Bit, Exponent, Constant(0x400e));
|
||||
Ref IsOverflow = _NZCVSelect(OpSize::i64Bit, {COND_UGE}, Constant(1), Constant(0));
|
||||
SubWithFlags(OpSize::i64Bit, Exponent, 0x400e);
|
||||
Ref IsOverflow = _NZCVSelect01({COND_UGE});
|
||||
|
||||
// Set Invalid Operation flag if overflow or special value
|
||||
Ref InvalidFlag = _Or(OpSize::i64Bit, IsSpecial, IsOverflow);
|
||||
@@ -443,13 +443,13 @@ void OpDispatchBuilder::X87LDENV(OpcodeArgs) {
|
||||
auto NewFCW = _LoadMem(GPRClass, OpSize::i16Bit, Mem, OpSize::i16Bit);
|
||||
_StoreContext(OpSize::i16Bit, GPRClass, NewFCW, offsetof(FEXCore::Core::CPUState, FCW));
|
||||
|
||||
Ref MemLocation = _Add(OpSize::i64Bit, Mem, Constant(IR::OpSizeToSize(Size) * 1));
|
||||
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 1);
|
||||
auto NewFSW = _LoadMem(GPRClass, Size, MemLocation, Size);
|
||||
ReconstructX87StateFromFSW_Helper(NewFSW);
|
||||
|
||||
{
|
||||
// FTW
|
||||
Ref MemLocation = _Add(OpSize::i64Bit, Mem, Constant(IR::OpSizeToSize(Size) * 2));
|
||||
Ref MemLocation = Add(OpSize::i64Bit, Mem, IR::OpSizeToSize(Size) * 2);
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, MemLocation, Size));
|
||||
}
|
||||
}
|
||||
@@ -512,7 +512,6 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
_StoreMem(GPRClass, Size, ZeroConst, Mem, Constant(IR::OpSizeToSize(Size) * 6), Size, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
auto OneConst = Constant(1);
|
||||
auto SevenConst = Constant(7);
|
||||
const auto LoadSize = ReducedPrecisionMode ? OpSize::i64Bit : OpSize::i128Bit;
|
||||
for (int i = 0; i < 7; ++i) {
|
||||
@@ -521,7 +520,7 @@ void OpDispatchBuilder::X87FNSAVE(OpcodeArgs) {
|
||||
data = _F80CVTTo(data, OpSize::i64Bit);
|
||||
}
|
||||
_StoreMem(FPRClass, OpSize::i128Bit, data, Mem, Constant((IR::OpSizeToSize(Size) * 7) + (10 * i)), OpSize::i8Bit, MEM_OFFSET_SXTX, 1);
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
@@ -565,9 +564,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
SetX87FTW(_LoadMem(GPRClass, Size, Mem, Constant(IR::OpSizeToSize(Size) * 2), Size, MEM_OFFSET_SXTX, 1));
|
||||
}
|
||||
|
||||
auto OneConst = Constant(1);
|
||||
auto SevenConst = Constant(7);
|
||||
|
||||
auto low = Constant(~0ULL);
|
||||
auto high = Constant(0xFFFF);
|
||||
Ref Mask = _VLoadTwoGPRs(low, high);
|
||||
@@ -582,7 +579,7 @@ void OpDispatchBuilder::X87FRSTOR(OpcodeArgs) {
|
||||
}
|
||||
_StoreContextIndexed(Reg, Top, StoreSize, MMBaseOffset(), IR::OpSizeToSize(OpSize::i128Bit), FPRClass);
|
||||
|
||||
Top = _And(OpSize::i32Bit, _Add(OpSize::i32Bit, Top, OneConst), SevenConst);
|
||||
Top = _And(OpSize::i32Bit, Add(OpSize::i32Bit, Top, 1), SevenConst);
|
||||
}
|
||||
|
||||
// The final st(7) needs a bit of special handling here
|
||||
@@ -831,11 +828,7 @@ void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
|
||||
default: LOGMAN_MSG_A_FMT("Unhandled FCMOV op: 0x{:x}", Opcode); break;
|
||||
}
|
||||
|
||||
auto ZeroConst = Constant(0);
|
||||
auto AllOneConst = Constant(0xffff'ffff'ffff'ffffull);
|
||||
|
||||
Ref SrcCond = SelectCC(CC, OpSize::i64Bit, AllOneConst, ZeroConst);
|
||||
Ref VecCond = _VDupFromGPR(OpSize::i128Bit, OpSize::i64Bit, SrcCond);
|
||||
Ref VecCond = _VDupFromGPR(OpSize::i128Bit, OpSize::i64Bit, SelectCC0All1(CC));
|
||||
_F80VBSLStack(OpSize::i128Bit, VecCond, Op->OP & 7, 0);
|
||||
}
|
||||
|
||||
@@ -851,11 +844,9 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
|
||||
// Claim this is a normal number
|
||||
// We don't support anything else
|
||||
auto TopValid = _StackValidTag(0);
|
||||
auto ZeroConst = Constant(0);
|
||||
auto OneConst = Constant(1);
|
||||
|
||||
// In the case of top being invalid then C3:C2:C0 is 0b101
|
||||
auto C3 = _Select(FEXCore::IR::COND_NEQ, TopValid, OneConst, OneConst, ZeroConst);
|
||||
auto C3 = Select01(OpSize::i32Bit, CondClassType {COND_NEQ}, TopValid, Constant(1));
|
||||
|
||||
auto C2 = TopValid;
|
||||
auto C0 = C3; // Mirror C3 until something other than zero is supported
|
||||
|
||||
@@ -382,7 +382,7 @@ void OpDispatchBuilder::X87FXTRACTF64(OpcodeArgs) {
|
||||
|
||||
// non zero case
|
||||
Ref ExpNZ = _Bfe(OpSize::i64Bit, 11, 52, Gpr);
|
||||
ExpNZ = _Sub(OpSize::i64Bit, ExpNZ, Constant(1023));
|
||||
ExpNZ = Sub(OpSize::i64Bit, ExpNZ, Constant(1023));
|
||||
Ref ExpNZV = _Float_FromGPR_S(OpSize::i64Bit, OpSize::i64Bit, ExpNZ);
|
||||
|
||||
Ref SigNZ = _And(OpSize::i64Bit, Gpr, Constant(0x800f'ffff'ffff'ffffLL));
|
||||
|
||||
@@ -308,12 +308,14 @@
|
||||
"RAOverride": "0"
|
||||
},
|
||||
"CondJump SSA:$Cmp1, SSA:$Cmp2, SSA:$TrueBlock, SSA:$FalseBlock, CondClass:$Cond{{COND_NEQ}}, OpSize:$CompareSize{OpSize::iInvalid}, i1:$FromNZCV{false}": {
|
||||
"Inline": ["", "AddSub"],
|
||||
"HasSideEffects": true,
|
||||
"RAOverride": "2"
|
||||
},
|
||||
"ExitFunction OpSize:#Size, GPR:$NewRIP, BranchHint:$Hint, GPR:$CallReturnAddress, SSA:$CallReturnBlock": {
|
||||
"Desc": ["Exits the current JIT function with a target RIP"
|
||||
],
|
||||
"Inline": ["Any"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"RAOverride": "2"
|
||||
@@ -448,10 +450,10 @@
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"Inline": ["Zero", ""],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "ByteSize",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class",
|
||||
"($Class == GPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit)) || $Class == FPRClass",
|
||||
"($Class == FPRClass && (#ByteSize == IR::OpSize::i8Bit || #ByteSize == IR::OpSize::i16Bit || #ByteSize == IR::OpSize::i32Bit || #ByteSize == IR::OpSize::i64Bit || #ByteSize == IR::OpSize::i128Bit || #ByteSize == IR::OpSize::i256Bit)) || $Class == GPRClass",
|
||||
"!($Offset >= offsetof(Core::CPUState, gregs[0]) && $Offset < offsetof(Core::CPUState, gregs[16])) && \"Can't StoreContext to GPR\"",
|
||||
@@ -543,6 +545,7 @@
|
||||
},
|
||||
|
||||
"SSA = LoadMem RegisterClass:$Class, OpSize:#Size, GPR:$Addr, GPR:$Offset, OpSize:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Inline": ["", "Mem"],
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
@@ -557,11 +560,9 @@
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"Inline": ["Zero", "", "Mem"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class"
|
||||
]
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"StoreMemPair RegisterClass:$Class, OpSize:#Size, SSA:$Value1, SSA:$Value2, GPR:$Addr, u32:$Offset": {
|
||||
@@ -569,11 +570,9 @@
|
||||
"Zero Extends if value's type is too small",
|
||||
"Truncates if value's type is too large"
|
||||
],
|
||||
"Inline": ["Zero", "Zero"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value1) == $Class"
|
||||
]
|
||||
"DestSize": "Size"
|
||||
},
|
||||
|
||||
"StoreMemX87SVEOptPredicate OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Value, GPR:$Addr": {
|
||||
@@ -593,6 +592,7 @@
|
||||
"SSA = LoadMemTSO RegisterClass:$Class, OpSize:#Size, GPR:$Addr, GPR:$Offset, OpSize:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a x86 TSO compatible load from memory. Offset must be Invalid()."
|
||||
],
|
||||
"Inline": ["", "Memtso"],
|
||||
"DestSize": "Size",
|
||||
"DynamicDispatch": true
|
||||
},
|
||||
@@ -600,12 +600,10 @@
|
||||
"StoreMemTSO RegisterClass:$Class, OpSize:#Size, SSA:$Value, GPR:$Addr, GPR:$Offset, OpSize:$Align, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a x86 TSO compatible store to memory. Offset must be Invalid()."
|
||||
],
|
||||
"Inline": ["Zero", "", "Memtso"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"DynamicDispatch": true,
|
||||
"EmitValidation": [
|
||||
"WalkFindRegClass($Value) == $Class"
|
||||
]
|
||||
"DynamicDispatch": true
|
||||
},
|
||||
|
||||
"FPR = VLoadVectorMasked OpSize:#RegisterSize, OpSize:#ElementSize, FPR:$Mask, GPR:$Addr, GPR:$Offset, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
@@ -716,6 +714,7 @@
|
||||
"Desc": ["Duplicates behaviour of x86 STOS repeat",
|
||||
"Returns the final address that gets generated without the prefix appended."
|
||||
],
|
||||
"Inline": ["", "", "Zero", "", "Any"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
@@ -723,6 +722,7 @@
|
||||
"Desc": ["Duplicates behaviour of x86 MOVS repeat",
|
||||
"Returns the final addresses after they have been incremented or decremented"
|
||||
],
|
||||
"Inline": ["", "", "", "Any"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
@@ -760,9 +760,8 @@
|
||||
"Prefetch i1:$ForStore, i1:$Stream, i8:$CacheLevel, GPR:$Addr, GPR:$Offset, MemOffsetType:$OffsetType, u8:$OffsetScale": {
|
||||
"Desc": ["Does a cacheline prefetch operation"
|
||||
],
|
||||
"EmitValidation": [
|
||||
"CacheLevel > 0 && CacheLevel < 4"
|
||||
],
|
||||
"Inline": ["", "Mem"],
|
||||
"EmitValidation": ["CacheLevel > 0 && CacheLevel < 4"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "OpSize::i64Bit"
|
||||
},
|
||||
@@ -1080,6 +1079,7 @@
|
||||
"Desc": [ "Integer Add",
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"Inline": ["", "LargeAddSub"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
@@ -1089,6 +1089,7 @@
|
||||
"Desc": [ "Integer Add with carry",
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"Inline": ["Zero", ""],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
@@ -1116,6 +1117,7 @@
|
||||
},
|
||||
"GPR = AddWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": [ "Integer add. Truncates and sets NZCV per AddNZCV"],
|
||||
"Inline": ["", "LargeAddSub"],
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true,
|
||||
"EmitValidation": [
|
||||
@@ -1124,6 +1126,7 @@
|
||||
},
|
||||
"AddNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the sum of two GPRs"],
|
||||
"Inline": ["", "LargeAddSub"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size"
|
||||
},
|
||||
@@ -1150,10 +1153,12 @@
|
||||
},
|
||||
"RmifNZCV GPR:$Src, u8:$Rotate, u8:$Mask": {
|
||||
"Desc": ["Rotate, mask, and insert into NZCV on FlagM platforms"],
|
||||
"Inline": ["Zero", ""],
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"CondAddNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2, CondClass:$Cond, u8:$FalseNZCV": {
|
||||
"Desc": ["If condition is true, set NZCV per sum of GPRs, else force NZCV to a constant."],
|
||||
"Inline": ["Zero", "AddSub"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -1162,6 +1167,7 @@
|
||||
},
|
||||
"CondSubNZCV OpSize:#Size, GPR:$Src1, GPR:$Src2, CondClass:$Cond, u8:$FalseNZCV": {
|
||||
"Desc": ["If condition is true, set NZCV per difference of GPRs, else force NZCV to a constant."],
|
||||
"Inline": ["Zero", "AddSub"],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -1170,6 +1176,7 @@
|
||||
},
|
||||
"GPR = AdcWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Adds and set NZCV for the sum of two GPRs and carry-in given as NZCV"],
|
||||
"Inline": ["Zero", ""],
|
||||
"HasSideEffects": true,
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
@@ -1219,6 +1226,7 @@
|
||||
"Desc": [ "Integer Sub",
|
||||
"Will truncate to 64 or 32bits"
|
||||
],
|
||||
"Inline": ["SubtractZero", "LargeAddSub"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
@@ -1236,6 +1244,7 @@
|
||||
},
|
||||
"GPR = SubWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": [ "Integer Sub. Truncates and sets NZCV per SubNZCV"],
|
||||
"Inline": ["SubtractZero", "LargeAddSub"],
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true,
|
||||
"EmitValidation": [
|
||||
@@ -1252,6 +1261,7 @@
|
||||
"Desc": ["Set NZCV for the difference of two GPRs. ",
|
||||
"Carry flag uses arm64 definition, inverted x86.",
|
||||
""],
|
||||
"Inline": ["Zero", "LargeAddSub"],
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true
|
||||
},
|
||||
@@ -1259,6 +1269,7 @@
|
||||
"Desc": ["Integer binary or"
|
||||
],
|
||||
"DestSize": "Size",
|
||||
"Inline": ["", "Logical"],
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
@@ -1288,8 +1299,8 @@
|
||||
]
|
||||
},
|
||||
"GPR = Xor OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer binary exclusive or"
|
||||
],
|
||||
"Desc": ["Integer binary exclusive or"],
|
||||
"Inline": ["", "Logical"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
@@ -1310,8 +1321,8 @@
|
||||
]
|
||||
},
|
||||
"GPR = And OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer binary and"
|
||||
],
|
||||
"Desc": ["Integer binary and"],
|
||||
"Inline": ["", "Logical"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
@@ -1327,6 +1338,7 @@
|
||||
"GPR = AndWithFlags OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer binary and"
|
||||
],
|
||||
"Inline": ["", "Logical"],
|
||||
"DestSize": "Size",
|
||||
"TiedSource": 0,
|
||||
"HasSideEffects": true
|
||||
@@ -1334,12 +1346,14 @@
|
||||
"GPR = Andn OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer binary AND NOT. Performs the equivalent of Src1 & ~Src2"],
|
||||
"DestSize": "Size",
|
||||
"Inline": ["", "Logical"],
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"TestNZ OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Set NZCV for the binary AND of two GPRs, setting N and Z accordingly and zeroing C and V"],
|
||||
"Inline": ["", "Logical"],
|
||||
"DestSize": "Size",
|
||||
"HasSideEffects": true
|
||||
},
|
||||
@@ -1349,24 +1363,24 @@
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Lshl OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer logical shift left"
|
||||
],
|
||||
"Desc": ["Integer logical shift left"],
|
||||
"Inline": ["", "Any"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Lshr OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer logical shift right"
|
||||
],
|
||||
"Desc": ["Integer logical shift right"],
|
||||
"Inline": ["", "Any"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Ashr OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer arithmetic shift right"
|
||||
],
|
||||
"Desc": ["Integer arithmetic shift right"],
|
||||
"Inline": ["", "Any"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
@@ -1383,8 +1397,8 @@
|
||||
"HasSideEffects": true
|
||||
},
|
||||
"GPR = Ror OpSize:#Size, GPR:$Src1, GPR:$Src2": {
|
||||
"Desc": ["Integer rotate right"
|
||||
],
|
||||
"Desc": ["Integer rotate right"],
|
||||
"Inline": ["", "Any"],
|
||||
"DestSize": "Size",
|
||||
"EmitValidation": [
|
||||
"Size == FEXCore::IR::OpSize::i32Bit || Size == FEXCore::IR::OpSize::i64Bit"
|
||||
@@ -1520,12 +1534,12 @@
|
||||
"op:",
|
||||
"Dest = Cmp1 <Cond> Cmp2 ? TrueVal : FalseVal"
|
||||
],
|
||||
"Inline": ["", "AddSub", "", ""],
|
||||
"DestSize": "ResultSize",
|
||||
"ImplicitFlagClobber": true,
|
||||
"EmitValidation": [
|
||||
"CompareSize == FEXCore::IR::OpSize::i32Bit || CompareSize == FEXCore::IR::OpSize::i64Bit || CompareSize == FEXCore::IR::OpSize::i128Bit",
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit",
|
||||
"WalkFindRegClass($Cmp1) == WalkFindRegClass($Cmp2)"
|
||||
"ResultSize == FEXCore::IR::OpSize::i32Bit || ResultSize == FEXCore::IR::OpSize::i64Bit"
|
||||
]
|
||||
},
|
||||
"GPR = Extr OpSize:#Size, GPR:$Upper, GPR:$Lower, u8:$LSB": {
|
||||
|
||||
@@ -1,6 +1,7 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
#pragma once
|
||||
|
||||
#include "CodeEmitter/Emitter.h"
|
||||
#include "Interface/IR/IR.h"
|
||||
#include "Interface/IR/IntrusiveIRList.h"
|
||||
|
||||
@@ -23,8 +24,9 @@ class IREmitter {
|
||||
friend class FEXCore::IR::PassManager;
|
||||
|
||||
public:
|
||||
IREmitter(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator)
|
||||
: DualListData {ThreadAllocator, 8 * 1024 * 1024} {
|
||||
IREmitter(FEXCore::Utils::IntrusivePooledAllocator& ThreadAllocator, bool SupportsTSOImm9)
|
||||
: DualListData {ThreadAllocator, 8 * 1024 * 1024}
|
||||
, SupportsTSOImm9(SupportsTSOImm9) {
|
||||
ReownOrClaimBuffer();
|
||||
ResetWorkingList();
|
||||
}
|
||||
@@ -51,6 +53,56 @@ public:
|
||||
|
||||
FEXCore::IR::RegisterClassType WalkFindRegClass(Ref Node);
|
||||
|
||||
// These inlining helpers are used by IRDefines.inc so define first.
|
||||
Ref InlineMem(OpSize Size, Ref Offset, MemOffsetType OffsetType, uint8_t& OffsetScale, bool TSO = false) {
|
||||
uint64_t Imm {};
|
||||
if (OffsetType != MEM_OFFSET_SXTX || !IsValueConstant(WrapNode(Offset), &Imm)) {
|
||||
return Offset;
|
||||
}
|
||||
|
||||
// The immediate may be scaled in the IR, we need to correct for that.
|
||||
Imm *= OffsetScale;
|
||||
|
||||
// Signed immediate unscaled 9-bit range for both regular and LRCPC2 ops.
|
||||
bool IsSIMM9 = ((int64_t)Imm >= -256) && ((int64_t)Imm <= 255);
|
||||
IsSIMM9 &= (SupportsTSOImm9 || !TSO);
|
||||
|
||||
// Extended offsets for regular loadstore only.
|
||||
LOGMAN_THROW_A_FMT(Size >= IR::OpSize::i8Bit && Size <= IR::OpSize::i256Bit, "Must be sized");
|
||||
|
||||
bool IsExtended = (Imm & (IR::OpSizeToSize(Size) - 1)) == 0 && Imm / IR::OpSizeToSize(Size) <= 4095;
|
||||
IsExtended &= !TSO;
|
||||
|
||||
if (IsSIMM9 || IsExtended) {
|
||||
OffsetScale = 1;
|
||||
return _InlineConstant(Imm);
|
||||
} else {
|
||||
return Offset;
|
||||
}
|
||||
}
|
||||
|
||||
#define DEF_INLINE(Type, Variable, Filter) \
|
||||
Ref Inline##Type(OpSize Size, Ref Source) { \
|
||||
uint64_t Variable; \
|
||||
if (IsValueConstant(WrapNode(Source), &Variable) && (Filter)) { \
|
||||
return _InlineConstant(Variable); \
|
||||
} else { \
|
||||
return Source; \
|
||||
} \
|
||||
}
|
||||
|
||||
DEF_INLINE(Any, _, true)
|
||||
DEF_INLINE(Zero, X, X == 0)
|
||||
DEF_INLINE(AddSub, X, ARMEmitter::IsImmAddSub(X))
|
||||
DEF_INLINE(LargeAddSub, X, ARMEmitter::IsImmAddSub(X) && Size >= OpSize::i32Bit);
|
||||
DEF_INLINE(Logical, X, ARMEmitter::Emitter::IsImmLogical(X, std::max((int)IR::OpSizeAsBits(Size), 32)));
|
||||
|
||||
Ref InlineSubtractZero(OpSize Size, Ref Src1, Ref Src2) {
|
||||
// Only inline a zero if we won't inline the other source.
|
||||
return IsValueConstant(WrapNode(Src2)) ? Src1 : InlineZero(Size, Src1);
|
||||
}
|
||||
#undef DEF_INLINE
|
||||
|
||||
// These handlers add cost to the constructor and destructor
|
||||
// If it becomes an issue then blow them away
|
||||
// GCC also generates some pretty atrocious code around these
|
||||
@@ -82,6 +134,66 @@ public:
|
||||
return _StoreMem(Class, Size, Value, Addr, Invalid(), Align, MEM_OFFSET_SXTX, 1);
|
||||
}
|
||||
|
||||
IRPair<IROp_Select> Select01(FEXCore::IR::OpSize CompareSize, CondClassType Cond, OrderedNode* Cmp1, OrderedNode* Cmp2) {
|
||||
return _Select(OpSize::i64Bit, CompareSize, Cond, Cmp1, Cmp2, _InlineConstant(1), _InlineConstant(0));
|
||||
}
|
||||
|
||||
IRPair<IROp_Select> To01(FEXCore::IR::OpSize CompareSize, OrderedNode* Cmp1) {
|
||||
return Select01(CompareSize, CondClassType {COND_NEQ}, Cmp1, Constant(0));
|
||||
}
|
||||
|
||||
IRPair<IROp_NZCVSelect> _NZCVSelect01(CondClassType Cond) {
|
||||
return _NZCVSelect(OpSize::i64Bit, Cond, _InlineConstant(1), _InlineConstant(0));
|
||||
}
|
||||
|
||||
Ref Addsub(IR::OpSize Size, IROps Op, IROps NegatedOp, Ref Src1, uint64_t Src2) {
|
||||
// Sign-extend the constant
|
||||
if (Size == OpSize::i32Bit) {
|
||||
Src2 = (int64_t)(int32_t)Src2;
|
||||
}
|
||||
|
||||
// Negative constants need to be negated to inline.
|
||||
if (Src2 & (1ull << 63) && ARMEmitter::IsImmAddSub(-Src2)) {
|
||||
Op = NegatedOp;
|
||||
Src2 = -Src2;
|
||||
}
|
||||
|
||||
auto Dest = _Add(Size, Src1, Constant(Src2));
|
||||
Dest.first->Header.Op = Op;
|
||||
return Dest;
|
||||
}
|
||||
|
||||
Ref Add(IR::OpSize Size, Ref Src1, uint64_t Src2) {
|
||||
return Addsub(Size, OP_ADD, OP_SUB, Src1, Src2);
|
||||
}
|
||||
|
||||
Ref Sub(IR::OpSize Size, Ref Src1, uint64_t Src2) {
|
||||
return Addsub(Size, OP_SUB, OP_ADD, Src1, Src2);
|
||||
}
|
||||
|
||||
Ref AddWithFlags(IR::OpSize Size, Ref Src1, uint64_t Src2) {
|
||||
return Addsub(Size, OP_ADDWITHFLAGS, OP_SUBWITHFLAGS, Src1, Src2);
|
||||
}
|
||||
|
||||
Ref SubWithFlags(IR::OpSize Size, Ref Src1, uint64_t Src2) {
|
||||
return Addsub(Size, OP_SUBWITHFLAGS, OP_ADDWITHFLAGS, Src1, Src2);
|
||||
}
|
||||
|
||||
#define DEF_ADDSUB(Op) \
|
||||
Ref Op(IR::OpSize Size, Ref Src1, Ref Src2) { \
|
||||
uint64_t Constant; \
|
||||
if (IsValueConstant(WrapNode(Src2), &Constant)) { \
|
||||
return Op(Size, Src1, Constant); \
|
||||
} else { \
|
||||
return _##Op(Size, Src1, Src2); \
|
||||
} \
|
||||
}
|
||||
|
||||
DEF_ADDSUB(Add)
|
||||
DEF_ADDSUB(Sub)
|
||||
DEF_ADDSUB(AddWithFlags)
|
||||
DEF_ADDSUB(SubWithFlags)
|
||||
|
||||
int64_t Constants[32];
|
||||
Ref ConstantRefs[32];
|
||||
uint32_t NrConstants;
|
||||
@@ -348,6 +460,7 @@ protected:
|
||||
Ref CurrentCodeBlock {};
|
||||
fextl::vector<Ref> CodeBlocks;
|
||||
uint64_t Entry {};
|
||||
bool SupportsTSOImm9 {};
|
||||
};
|
||||
|
||||
} // namespace FEXCore::IR
|
||||
@@ -71,7 +71,6 @@ void PassManager::AddDefaultPasses(FEXCore::Context::ContextImpl* ctx) {
|
||||
|
||||
if (!DisablePasses()) {
|
||||
InsertPass(CreateX87StackOptimizationPass(ctx->HostFeatures, ctx->Config.Is64BitMode ? IR::OpSize::i64Bit : IR::OpSize::i32Bit));
|
||||
InsertPass(CreateConstProp(ctx->HostFeatures.SupportsTSOImm9));
|
||||
InsertPass(CreateDeadFlagCalculationEliminination());
|
||||
}
|
||||
}
|
||||
|
||||
@@ -43,7 +43,6 @@ protected:
|
||||
};
|
||||
|
||||
class PassManager final {
|
||||
friend class ConstProp;
|
||||
public:
|
||||
void AddDefaultPasses(FEXCore::Context::ContextImpl* ctx);
|
||||
void AddDefaultValidationPasses();
|
||||
|
||||
@@ -16,7 +16,6 @@ namespace FEXCore::IR {
|
||||
class Pass;
|
||||
class RegisterAllocationPass;
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateDeadFlagCalculationEliminination();
|
||||
fextl::unique_ptr<FEXCore::IR::RegisterAllocationPass> CreateRegisterAllocationPass(const FEXCore::CPUIDEmu* CPUID);
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateX87StackOptimizationPass(const FEXCore::HostFeatures&, OpSize GPROpSize);
|
||||
|
||||
@@ -1,341 +0,0 @@
|
||||
// SPDX-License-Identifier: MIT
|
||||
/*
|
||||
$info$
|
||||
tags: ir|opts
|
||||
desc: ConstProp, ZExt elim, const pooling, fcmp reduction, const inlining
|
||||
$end_info$
|
||||
*/
|
||||
#include <CodeEmitter/Emitter.h>
|
||||
|
||||
#include "Interface/IR/IREmitter.h"
|
||||
#include "Interface/IR/PassManager.h"
|
||||
|
||||
#include <FEXCore/IR/IR.h>
|
||||
#include <FEXCore/HLE/SyscallHandler.h>
|
||||
#include <FEXCore/Utils/LogManager.h>
|
||||
#include <FEXCore/Utils/Profiler.h>
|
||||
#include <FEXCore/fextl/map.h>
|
||||
#include <FEXCore/fextl/unordered_map.h>
|
||||
|
||||
#include <cstdint>
|
||||
#include <string.h>
|
||||
|
||||
namespace FEXCore::IR {
|
||||
|
||||
// aarch64 heuristics
|
||||
static bool IsImmLogical(uint64_t imm, unsigned width) {
|
||||
if (width < 32) {
|
||||
width = 32;
|
||||
}
|
||||
return ARMEmitter::Emitter::IsImmLogical(imm, width);
|
||||
}
|
||||
|
||||
class ConstProp final : public FEXCore::IR::Pass {
|
||||
public:
|
||||
explicit ConstProp(bool SupportsTSOImm9)
|
||||
: SupportsTSOImm9 {SupportsTSOImm9} {}
|
||||
|
||||
void Run(IREmitter* IREmit) override;
|
||||
|
||||
private:
|
||||
void ConstantPropagation(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp);
|
||||
|
||||
bool SupportsTSOImm9 {};
|
||||
|
||||
template<class F>
|
||||
bool InlineIf(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index, F Filter) {
|
||||
uint64_t Constant;
|
||||
if (!IREmit->IsValueConstant(IROp->Args[Index], &Constant) || !Filter(Constant)) {
|
||||
return false;
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[Index]));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Index, IREmit->_InlineConstant(Constant));
|
||||
return true;
|
||||
}
|
||||
|
||||
bool Inline(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index) {
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, [](uint64_t _) { return true; });
|
||||
}
|
||||
|
||||
bool InlineIfZero(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index) {
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, [](uint64_t X) { return X == 0; });
|
||||
}
|
||||
|
||||
bool InlineIfLargeAddSub(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp, unsigned Index) {
|
||||
// We don't allow 8/16-bit operations to have constants, since no
|
||||
// constant would be in bounds after the JIT's 24/16 shift.
|
||||
auto Filter = [&IROp](uint64_t X) {
|
||||
return ARMEmitter::IsImmAddSub(X) && IROp->Size >= OpSize::i32Bit;
|
||||
};
|
||||
|
||||
return InlineIf(IREmit, CurrentIR, CodeNode, IROp, Index, Filter);
|
||||
}
|
||||
|
||||
void InlineMemImmediate(IREmitter* IREmit, const IRListView& IR, Ref CodeNode, IR::RegisterClassType RegisterClass, IROp_Header* IROp,
|
||||
OrderedNodeWrapper Offset, MemOffsetType OffsetType, const size_t Offset_Index, uint8_t& OffsetScale, bool TSO) {
|
||||
uint64_t Imm {};
|
||||
if (OffsetType != MEM_OFFSET_SXTX || !IREmit->IsValueConstant(Offset, &Imm)) {
|
||||
return;
|
||||
}
|
||||
|
||||
// The immediate may be scaled in the IR, we need to correct for that.
|
||||
Imm *= OffsetScale;
|
||||
|
||||
// Signed immediate unscaled 9-bit range for both regular and LRCPC2 ops.
|
||||
bool IsSIMM9 = ((int64_t)Imm >= -256) && ((int64_t)Imm <= 255);
|
||||
IsSIMM9 &= (SupportsTSOImm9 || !TSO);
|
||||
|
||||
// Extended offsets for regular loadstore only.
|
||||
LOGMAN_THROW_A_FMT(IROp->Size >= IR::OpSize::i8Bit && IROp->Size <= (RegisterClass == GPRClass ? IR::OpSize::i64Bit : IR::OpSize::i256Bit),
|
||||
"Invalid "
|
||||
"size");
|
||||
bool IsExtended = (Imm & (IR::OpSizeToSize(IROp->Size) - 1)) == 0 && Imm / IR::OpSizeToSize(IROp->Size) <= 4095;
|
||||
IsExtended &= !TSO;
|
||||
|
||||
if (IsSIMM9 || IsExtended) {
|
||||
IREmit->SetWriteCursor(IR.GetNode(Offset));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Offset_Index, IREmit->_InlineConstant(Imm));
|
||||
OffsetScale = 1;
|
||||
}
|
||||
}
|
||||
};
|
||||
|
||||
// constprop + some more per instruction logic
|
||||
void ConstProp::ConstantPropagation(IREmitter* IREmit, const IRListView& CurrentIR, Ref CodeNode, IROp_Header* IROp) {
|
||||
switch (IROp->Op) {
|
||||
case OP_ADD:
|
||||
case OP_SUB:
|
||||
case OP_ADDWITHFLAGS:
|
||||
case OP_SUBWITHFLAGS: {
|
||||
auto Op = IROp->C<IR::IROp_Add>();
|
||||
uint64_t Constant1 {};
|
||||
uint64_t Constant2 {};
|
||||
bool IsConstant2 = IREmit->IsValueConstant(IROp->Args[1], &Constant2);
|
||||
|
||||
/* IsImmAddSub assumes the constants are sign-extended, take care of that
|
||||
* here so we get the optimization for 32-bit adds too.
|
||||
*/
|
||||
if (Op->Header.Size == OpSize::i32Bit) {
|
||||
Constant1 = (int64_t)(int32_t)Constant1;
|
||||
Constant2 = (int64_t)(int32_t)Constant2;
|
||||
}
|
||||
|
||||
if (IsConstant2 && !ARMEmitter::IsImmAddSub(Constant2) && ARMEmitter::IsImmAddSub(-Constant2)) {
|
||||
// If the second argument is constant, the immediate is not ImmAddSub, but when negated is.
|
||||
// So, negate the operation to negate (and inline) the constant.
|
||||
if (IROp->Op == OP_ADD) {
|
||||
IROp->Op = OP_SUB;
|
||||
} else if (IROp->Op == OP_SUB) {
|
||||
IROp->Op = OP_ADD;
|
||||
} else if (IROp->Op == OP_ADDWITHFLAGS) {
|
||||
IROp->Op = OP_SUBWITHFLAGS;
|
||||
} else if (IROp->Op == OP_SUBWITHFLAGS) {
|
||||
IROp->Op = OP_ADDWITHFLAGS;
|
||||
}
|
||||
|
||||
IREmit->SetWriteCursorBefore(CodeNode);
|
||||
|
||||
// Negate the constant.
|
||||
auto NegConstant = IREmit->_Constant(-Constant2);
|
||||
|
||||
// Replace the second source with the negated constant.
|
||||
IREmit->ReplaceNodeArgument(CodeNode, Op->Src2_Index, NegConstant);
|
||||
}
|
||||
|
||||
if (!InlineIfLargeAddSub(IREmit, CurrentIR, CodeNode, IROp, 1) && (IROp->Op == OP_SUB || IROp->Op == OP_SUBWITHFLAGS)) {
|
||||
// TODO: Generalize this
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_ADDNZCV: {
|
||||
InlineIfLargeAddSub(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_SUBNZCV: {
|
||||
if (!InlineIfLargeAddSub(IREmit, CurrentIR, CodeNode, IROp, 1)) {
|
||||
// TODO: Generalize this
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_AND:
|
||||
case OP_OR:
|
||||
case OP_XOR: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
break;
|
||||
}
|
||||
case OP_ANDWITHFLAGS:
|
||||
case OP_ANDN:
|
||||
case OP_TESTNZ: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, [&IROp](uint64_t X) { return IsImmLogical(X, IR::OpSizeAsBits(IROp->Size)); });
|
||||
break;
|
||||
}
|
||||
case OP_ASHR:
|
||||
case OP_ROR: {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_LSHL: {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_LSHR: {
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, 1);
|
||||
break;
|
||||
}
|
||||
case OP_ADC:
|
||||
case OP_ADCWITHFLAGS:
|
||||
case OP_RMIFNZCV: {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
break;
|
||||
}
|
||||
case OP_STORECONTEXT: {
|
||||
// For i128Bit, we won't see a normal Constant to inline, but as a special
|
||||
// case we can replace with a 2x64-bit store which can use inline zeroes.
|
||||
if (IROp->Size == OpSize::i128Bit) {
|
||||
auto Op = IROp->C<IR::IROp_StoreContext>();
|
||||
auto Header = IREmit->GetOpHeader(IROp->Args[0]);
|
||||
const auto MAX_STP_OFFSET = (252 * 4);
|
||||
|
||||
if (Op->Offset <= MAX_STP_OFFSET && Header->Op == OP_LOADNAMEDVECTORCONSTANT) {
|
||||
auto Const = Header->C<IR::IROp_LoadNamedVectorConstant>();
|
||||
|
||||
if (Const->Constant == IR::NamedVectorConstant::NAMED_VECTOR_ZERO) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
Ref Zero = IREmit->_Constant(0);
|
||||
Ref STP = IREmit->_StoreContextPair(IR::OpSize::i64Bit, GPRClass, Zero, Zero, Op->Offset);
|
||||
IREmit->Remove(CodeNode);
|
||||
|
||||
// XXX: This works around InlineConstant not having an associated
|
||||
// register class, else we'd just do InlineConstant above.
|
||||
Ref InlineZero = IREmit->_InlineConstant(0);
|
||||
IREmit->ReplaceNodeArgument(STP, 0, InlineZero);
|
||||
IREmit->ReplaceNodeArgument(STP, 1, InlineZero);
|
||||
}
|
||||
}
|
||||
} else {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_CONDADDNZCV:
|
||||
case OP_CONDSUBNZCV: {
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 0);
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, ARMEmitter::IsImmAddSub);
|
||||
break;
|
||||
}
|
||||
case OP_SELECT: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, ARMEmitter::IsImmAddSub);
|
||||
|
||||
uint64_t AllOnes = IROp->Size == OpSize::i64Bit ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
|
||||
uint64_t Constant2 {};
|
||||
uint64_t Constant3 {};
|
||||
if (IREmit->IsValueConstant(IROp->Args[2], &Constant2) && IREmit->IsValueConstant(IROp->Args[3], &Constant3) &&
|
||||
(Constant2 == 1 || Constant2 == AllOnes) && Constant3 == 0) {
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(IROp->Args[2]));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 2, IREmit->_InlineConstant(Constant2));
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 3, IREmit->_InlineConstant(Constant3));
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case OP_NZCVSELECT: {
|
||||
// We always allow source 1 to be zero, but source 0 can only be a
|
||||
// special 1/~0 constant if source 1 is 0.
|
||||
if (InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, 1)) {
|
||||
uint64_t AllOnes = IROp->Size == OpSize::i64Bit ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 0, [&AllOnes](uint64_t X) { return X == 1 || X == AllOnes; });
|
||||
}
|
||||
break;
|
||||
}
|
||||
case OP_CONDJUMP: {
|
||||
InlineIf(IREmit, CurrentIR, CodeNode, IROp, 1, ARMEmitter::IsImmAddSub);
|
||||
break;
|
||||
}
|
||||
case OP_EXITFUNCTION: {
|
||||
auto Op = IROp->C<IR::IROp_ExitFunction>();
|
||||
|
||||
if (!Inline(IREmit, CurrentIR, CodeNode, IROp, Op->NewRIP_Index)) {
|
||||
auto NewRIP = IREmit->GetOpHeader(Op->NewRIP);
|
||||
if (NewRIP->Op == OP_ENTRYPOINTOFFSET) {
|
||||
auto EO = NewRIP->C<IR::IROp_EntrypointOffset>();
|
||||
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->NewRIP));
|
||||
|
||||
IREmit->ReplaceNodeArgument(CodeNode, 0, IREmit->_InlineEntrypointOffset(EO->Header.Size, EO->Offset));
|
||||
}
|
||||
}
|
||||
break;
|
||||
}
|
||||
|
||||
case OP_LOADMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEM: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMem>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
case OP_PREFETCH: {
|
||||
auto Op = IROp->CW<IR::IROp_Prefetch>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, GPRClass, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, false);
|
||||
break;
|
||||
}
|
||||
case OP_LOADMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_LoadMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMTSO: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemTSO>();
|
||||
InlineMemImmediate(IREmit, CurrentIR, CodeNode, Op->Class, IROp, Op->Offset, Op->OffsetType, Op->Offset_Index, Op->OffsetScale, true);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
case OP_STOREMEMPAIR: {
|
||||
auto Op = IROp->CW<IR::IROp_StoreMemPair>();
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value1_Index);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value2_Index);
|
||||
break;
|
||||
}
|
||||
case OP_MEMCPY: {
|
||||
auto Op = IROp->CW<IR::IROp_MemCpy>();
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, Op->Direction_Index);
|
||||
break;
|
||||
}
|
||||
case OP_MEMSET: {
|
||||
auto Op = IROp->CW<IR::IROp_MemSet>();
|
||||
Inline(IREmit, CurrentIR, CodeNode, IROp, Op->Direction_Index);
|
||||
InlineIfZero(IREmit, CurrentIR, CodeNode, IROp, Op->Value_Index);
|
||||
break;
|
||||
}
|
||||
|
||||
default: break;
|
||||
}
|
||||
}
|
||||
|
||||
void ConstProp::Run(IREmitter* IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::ConstProp");
|
||||
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
for (auto [BlockNode, BlockIROp] : CurrentIR.GetBlocks()) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
ConstantPropagation(IREmit, CurrentIR, CodeNode, IROp);
|
||||
}
|
||||
}
|
||||
}
|
||||
|
||||
fextl::unique_ptr<FEXCore::IR::Pass> CreateConstProp(bool SupportsTSOImm9) {
|
||||
return fextl::make_unique<ConstProp>(SupportsTSOImm9);
|
||||
}
|
||||
} // namespace FEXCore::IR
|
||||
@@ -410,7 +410,7 @@ inline Ref X87StackOptimization::GetOffsetTopWithCache_Slow(uint8_t Offset) {
|
||||
|
||||
auto* OffsetTop = GetTopWithCache_Slow();
|
||||
if (Offset != 0) {
|
||||
OffsetTop = IREmit->_And(OpSize::i32Bit, IREmit->_Add(OpSize::i32Bit, OffsetTop, GetConstant(Offset)), GetConstant(7));
|
||||
OffsetTop = IREmit->_And(OpSize::i32Bit, IREmit->Add(OpSize::i32Bit, OffsetTop, Offset), GetConstant(7));
|
||||
// GetTopWithCache_Slow already sets the cache so we don't need to set it here for offset == 0
|
||||
TopOffsetCache[Offset] = OffsetTop;
|
||||
}
|
||||
@@ -542,7 +542,7 @@ void X87StackOptimization::HandleBinopStack(IROps Op64, bool VFOp64, IROps Op80,
|
||||
inline void X87StackOptimization::UpdateTopForPop_Slow() {
|
||||
// Pop the top of the x87 stack
|
||||
auto* TopOffset = GetTopWithCache_Slow();
|
||||
TopOffset = IREmit->_Add(OpSize::i32Bit, TopOffset, GetConstant(1));
|
||||
TopOffset = IREmit->Add(OpSize::i32Bit, TopOffset, 1);
|
||||
TopOffset = IREmit->_And(OpSize::i32Bit, TopOffset, GetConstant(7));
|
||||
SetTopWithCache_Slow(TopOffset);
|
||||
}
|
||||
@@ -550,7 +550,7 @@ inline void X87StackOptimization::UpdateTopForPop_Slow() {
|
||||
inline void X87StackOptimization::UpdateTopForPush_Slow() {
|
||||
// Pop the top of the x87 stack
|
||||
auto* TopOffset = GetTopWithCache_Slow();
|
||||
TopOffset = IREmit->_Sub(OpSize::i32Bit, TopOffset, GetConstant(1));
|
||||
TopOffset = IREmit->Sub(OpSize::i32Bit, TopOffset, 1);
|
||||
TopOffset = IREmit->_And(OpSize::i32Bit, TopOffset, GetConstant(7));
|
||||
SetTopWithCache_Slow(TopOffset);
|
||||
}
|
||||
@@ -569,12 +569,7 @@ Ref X87StackOptimization::SynchronizeStackValues() {
|
||||
|
||||
if (TopOffset != 0) {
|
||||
auto* OrigTop = GetTopWithCache_Slow();
|
||||
Ref NewTop {};
|
||||
if (TopOffset > 0) {
|
||||
NewTop = IREmit->_And(OpSize::i32Bit, IREmit->_Sub(OpSize::i32Bit, OrigTop, GetConstant(TopOffset)), GetConstant(0x7));
|
||||
} else {
|
||||
NewTop = IREmit->_And(OpSize::i32Bit, IREmit->_Add(OpSize::i32Bit, OrigTop, GetConstant(-TopOffset)), GetConstant(0x7));
|
||||
}
|
||||
Ref NewTop = IREmit->_And(OpSize::i32Bit, IREmit->Sub(OpSize::i32Bit, OrigTop, TopOffset), GetConstant(0x7));
|
||||
SetTopWithCache_Slow(NewTop);
|
||||
}
|
||||
StackData.TopOffset = 0;
|
||||
@@ -832,7 +827,7 @@ void X87StackOptimization::Run(IREmitter* Emit) {
|
||||
auto* TopValue = GetTopWithCache_Slow();
|
||||
if (Offset != 0) {
|
||||
auto* Mask = GetConstant(7);
|
||||
TopValue = IREmit->_And(OpSize::i32Bit, IREmit->_Add(OpSize::i32Bit, TopValue, GetConstant(Offset)), Mask);
|
||||
TopValue = IREmit->_And(OpSize::i32Bit, IREmit->Add(OpSize::i32Bit, TopValue, Offset), Mask);
|
||||
}
|
||||
SetX87ValidTag(TopValue, false);
|
||||
} else {
|
||||
|
||||
@@ -105,7 +105,6 @@ C++ Functions to generate IR. See IR.json for spec.
|
||||
IR to IR Optimization
|
||||
- [PassManager.cpp](../FEXCore/Source/Interface/IR/PassManager.cpp): Defines which passes are run, and runs them
|
||||
- [PassManager.h](../FEXCore/Source/Interface/IR/PassManager.h)
|
||||
- [ConstProp.cpp](../FEXCore/Source/Interface/IR/Passes/ConstProp.cpp): ConstProp, ZExt elim, const pooling, fcmp reduction, const inlining
|
||||
- [IRValidation.cpp](../FEXCore/Source/Interface/IR/Passes/IRValidation.cpp): Sanity checking pass
|
||||
- [RedundantFlagCalculationElimination.cpp](../FEXCore/Source/Interface/IR/Passes/RedundantFlagCalculationElimination.cpp): This is not used right now, possibly broken
|
||||
- [RegisterAllocationPass.cpp](../FEXCore/Source/Interface/IR/Passes/RegisterAllocationPass.cpp)
|
||||
|
||||
@@ -2702,8 +2702,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"csetm x0, eq",
|
||||
"ccmn x26, x0, #nzCv, le"
|
||||
]
|
||||
@@ -2715,8 +2715,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"csetm x0, eq",
|
||||
"ccmn x26, x0, #nzCv, le"
|
||||
]
|
||||
@@ -2728,8 +2728,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"csetm x0, eq",
|
||||
"ccmn x26, x0, #nzCv, le"
|
||||
]
|
||||
@@ -2741,8 +2741,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"csetm x0, eq",
|
||||
"ccmn x26, x0, #nzCv, le"
|
||||
]
|
||||
|
||||
@@ -19,8 +19,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"axflag"
|
||||
]
|
||||
},
|
||||
@@ -31,8 +31,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"axflag"
|
||||
]
|
||||
},
|
||||
@@ -43,8 +43,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"axflag"
|
||||
]
|
||||
},
|
||||
@@ -55,8 +55,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"axflag"
|
||||
]
|
||||
}
|
||||
|
||||
@@ -512,12 +512,12 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
@@ -529,8 +529,6 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #32]",
|
||||
"ldr q3, [x28, #48]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"and v4.16b, v17.16b, v16.16b",
|
||||
"and v5.16b, v3.16b, v2.16b",
|
||||
"ushr v4.4s, v4.4s, #31",
|
||||
@@ -545,11 +543,13 @@
|
||||
"add v2.4s, v2.4s, v4.4s",
|
||||
"addv s2, v2.4s",
|
||||
"mov w21, v2.s[0]",
|
||||
"mov w27, #0x0",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
@@ -570,12 +570,12 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
@@ -587,8 +587,6 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #32]",
|
||||
"ldr q3, [x28, #48]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"and v4.16b, v17.16b, v16.16b",
|
||||
"and v5.16b, v3.16b, v2.16b",
|
||||
"ushr v4.2d, v4.2d, #63",
|
||||
@@ -603,11 +601,13 @@
|
||||
"add v2.2d, v2.2d, v4.2d",
|
||||
"addp v2.2d, v2.2d, v2.2d",
|
||||
"mov x21, v2.d[0]",
|
||||
"mov w27, #0x0",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
@@ -676,12 +676,12 @@
|
||||
"umov w20, v2.h[0]",
|
||||
"umov w21, v3.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
@@ -704,12 +704,12 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -30,11 +30,11 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"vtestps ymm0, ymm1": {
|
||||
@@ -45,8 +45,6 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #32]",
|
||||
"ldr q3, [x28, #48]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"and v4.16b, v17.16b, v16.16b",
|
||||
"and v5.16b, v3.16b, v2.16b",
|
||||
"ushr v4.4s, v4.4s, #31",
|
||||
@@ -61,10 +59,12 @@
|
||||
"add v2.4s, v2.4s, v4.4s",
|
||||
"addv s2, v2.4s",
|
||||
"mov w21, v2.s[0]",
|
||||
"mov w27, #0x0",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"vtestpd xmm0, xmm1": {
|
||||
@@ -84,11 +84,11 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"vtestpd ymm0, ymm1": {
|
||||
@@ -99,8 +99,6 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr q2, [x28, #32]",
|
||||
"ldr q3, [x28, #48]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"and v4.16b, v17.16b, v16.16b",
|
||||
"and v5.16b, v3.16b, v2.16b",
|
||||
"ushr v4.2d, v4.2d, #63",
|
||||
@@ -115,10 +113,12 @@
|
||||
"add v2.2d, v2.2d, v4.2d",
|
||||
"addp v2.2d, v2.2d, v2.2d",
|
||||
"mov x21, v2.d[0]",
|
||||
"mov w27, #0x0",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"vptest xmm0, xmm1": {
|
||||
@@ -134,11 +134,11 @@
|
||||
"umov w20, v2.h[0]",
|
||||
"umov w21, v3.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"vptest ymm0, ymm1": {
|
||||
@@ -160,11 +160,11 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"vmaskmovps xmm0, xmm1, [rax]": {
|
||||
@@ -642,7 +642,7 @@
|
||||
"bic w20, w6, w20",
|
||||
"tst x7, #0xe0",
|
||||
"csel w4, w6, w20, ne",
|
||||
"cset w20, eq",
|
||||
"cset x20, eq",
|
||||
"cmp w4, #0x0 (0)",
|
||||
"rmif x20, #63, #nzCv"
|
||||
]
|
||||
@@ -658,7 +658,7 @@
|
||||
"bic x20, x6, x20",
|
||||
"tst x7, #0xc0",
|
||||
"csel x4, x6, x20, ne",
|
||||
"cset w20, eq",
|
||||
"cset x20, eq",
|
||||
"cmp x4, #0x0 (0)",
|
||||
"rmif x20, #63, #nzCv"
|
||||
]
|
||||
|
||||
@@ -1477,7 +1477,7 @@
|
||||
"mov w20, #0xff",
|
||||
"ldaddalb w20, w27, [x4]",
|
||||
"mov w20, #0x1",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"lsl w0, w27, #24",
|
||||
"cmp w0, w20, lsl #24",
|
||||
"sub w26, w27, #0x1 (1)",
|
||||
@@ -1575,7 +1575,7 @@
|
||||
"mov w20, #0xffff",
|
||||
"ldaddalh w20, w27, [x4]",
|
||||
"mov w20, #0x1",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"lsl w0, w27, #16",
|
||||
"cmp w0, w20, lsl #16",
|
||||
"sub w26, w27, #0x1 (1)",
|
||||
@@ -1590,7 +1590,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xffffffff",
|
||||
"ldaddal w20, w27, [x4]",
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"subs w26, w27, #0x1 (1)",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
@@ -1603,7 +1603,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov x20, #0xffffffffffffffff",
|
||||
"ldaddal x20, x27, [x4]",
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"subs x26, x27, #0x1 (1)",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
@@ -1616,7 +1616,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldaddalb w20, w27, [x4]",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"lsl w0, w27, #24",
|
||||
"cmn w0, w20, lsl #24",
|
||||
"add w26, w27, #0x1 (1)",
|
||||
@@ -1631,7 +1631,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldaddalh w20, w27, [x4]",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"lsl w0, w27, #16",
|
||||
"cmn w0, w20, lsl #16",
|
||||
"add w26, w27, #0x1 (1)",
|
||||
@@ -1646,7 +1646,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldaddal w20, w27, [x4]",
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"adds w26, w27, #0x1 (1)",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
@@ -1659,7 +1659,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldaddal x20, x27, [x4]",
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"adds x26, x27, #0x1 (1)",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
|
||||
@@ -171,7 +171,7 @@
|
||||
"mov w20, #0x5",
|
||||
"movk w20, #0x1, lsl #16",
|
||||
"str x20, [x28, #24]",
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x27, x26",
|
||||
"ubfx w21, w21, #4, #1",
|
||||
"orr x20, x20, x21, lsl #4",
|
||||
@@ -182,7 +182,7 @@
|
||||
"ldrsb x21, [x28, #986]",
|
||||
"lsr x21, x21, #63",
|
||||
"orr x20, x20, x21, lsl #10",
|
||||
"cset w21, vs",
|
||||
"cset x21, vs",
|
||||
"orr x20, x20, x21, lsl #11",
|
||||
"ldrb w21, [x28, #988]",
|
||||
"orr x20, x20, x21, lsl #12",
|
||||
@@ -1574,7 +1574,7 @@
|
||||
"mov v16.s[0], v0.s[0]",
|
||||
"mov w7, #0x0",
|
||||
"fcmp s16, s18",
|
||||
"cset w26, vc",
|
||||
"cset x26, vc",
|
||||
"axflag",
|
||||
"mov x27, x7"
|
||||
]
|
||||
|
||||
@@ -1389,7 +1389,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0xffffffff",
|
||||
"ldaddal w20, w27, [x4]",
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"subs w26, w27, #0x1 (1)",
|
||||
"rmif x20, #63, #nzCv"
|
||||
]
|
||||
@@ -1400,7 +1400,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov x20, #0xffffffffffffffff",
|
||||
"ldaddal x20, x27, [x4]",
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"subs x26, x27, #0x1 (1)",
|
||||
"rmif x20, #63, #nzCv"
|
||||
]
|
||||
@@ -1435,7 +1435,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldaddal w20, w27, [x4]",
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"adds w26, w27, #0x1 (1)",
|
||||
"rmif x20, #63, #nzCv"
|
||||
]
|
||||
@@ -1446,7 +1446,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, #0x1",
|
||||
"ldaddal x20, x27, [x4]",
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"adds x26, x27, #0x1 (1)",
|
||||
"rmif x20, #63, #nzCv"
|
||||
]
|
||||
|
||||
@@ -94,7 +94,7 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"adds x4, x4, x6",
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"adds x26, x4, #0x1 (1)",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"mov x27, x4",
|
||||
@@ -125,7 +125,7 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"subs x4, x4, x6",
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"subs x26, x4, #0x1 (1)",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"mov x27, x4",
|
||||
@@ -212,7 +212,7 @@
|
||||
"and x20, x7, #0x3f",
|
||||
"cbz x20, #+0x20",
|
||||
"lsr x20, x4, x7",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"neg x22, x7",
|
||||
"lsl x23, x4, x22",
|
||||
"orr x20, x20, x23, lsl #1",
|
||||
@@ -304,7 +304,7 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset w26, vc",
|
||||
"cset x26, vc",
|
||||
"bfxil x7, x26, #0, #8",
|
||||
"subs x26, x4, #0x0 (0)"
|
||||
]
|
||||
|
||||
@@ -25,11 +25,11 @@
|
||||
"umov w20, v2.h[0]",
|
||||
"umov w21, v3.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"adcx eax, ebx": {
|
||||
|
||||
@@ -1665,7 +1665,7 @@
|
||||
"ExpectedInstructionCount": 39,
|
||||
"Comment": "0x9c",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x27, x26",
|
||||
"ubfx w21, w21, #4, #1",
|
||||
"orr x20, x20, x21, lsl #4",
|
||||
@@ -1676,7 +1676,7 @@
|
||||
"ldrsb x21, [x28, #986]",
|
||||
"lsr x21, x21, #63",
|
||||
"orr x20, x20, x21, lsl #10",
|
||||
"cset w21, vs",
|
||||
"cset x21, vs",
|
||||
"orr x20, x20, x21, lsl #11",
|
||||
"ldrb w21, [x28, #988]",
|
||||
"orr x20, x20, x21, lsl #12",
|
||||
@@ -1710,7 +1710,7 @@
|
||||
"ExpectedInstructionCount": 39,
|
||||
"Comment": "0x9c",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x27, x26",
|
||||
"ubfx w21, w21, #4, #1",
|
||||
"orr x20, x20, x21, lsl #4",
|
||||
@@ -1721,7 +1721,7 @@
|
||||
"ldrsb x21, [x28, #986]",
|
||||
"lsr x21, x21, #63",
|
||||
"orr x20, x20, x21, lsl #10",
|
||||
"cset w21, vs",
|
||||
"cset x21, vs",
|
||||
"orr x20, x20, x21, lsl #11",
|
||||
"ldrb w21, [x28, #988]",
|
||||
"orr x20, x20, x21, lsl #12",
|
||||
@@ -1815,7 +1815,7 @@
|
||||
"ExpectedInstructionCount": 14,
|
||||
"Comment": "0x9f",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x27, x26",
|
||||
"ubfx w21, w21, #4, #1",
|
||||
"orr x20, x20, x21, lsl #4",
|
||||
|
||||
@@ -908,22 +908,22 @@
|
||||
"Comment": "GROUP2 0xC0 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxtb w20, w4",
|
||||
"mov w21, #0x0",
|
||||
"cset w22, lo",
|
||||
"bfi x21, x20, #55, #8",
|
||||
"bfi x21, x22, #63, #1",
|
||||
"bfi x21, x20, #46, #8",
|
||||
"bfi x21, x22, #54, #1",
|
||||
"bfi x21, x20, #37, #8",
|
||||
"bfi x21, x22, #45, #1",
|
||||
"bfi x21, x20, #28, #8",
|
||||
"bfi x21, x22, #36, #1",
|
||||
"bfi x21, x20, #19, #8",
|
||||
"bfi x21, x22, #27, #1",
|
||||
"bfxil x21, x20, #0, #8",
|
||||
"ror x20, x21, #62",
|
||||
"cset x21, lo",
|
||||
"mov w22, #0x0",
|
||||
"bfi x22, x20, #55, #8",
|
||||
"bfi x22, x21, #63, #1",
|
||||
"bfi x22, x20, #46, #8",
|
||||
"bfi x22, x21, #54, #1",
|
||||
"bfi x22, x20, #37, #8",
|
||||
"bfi x22, x21, #45, #1",
|
||||
"bfi x22, x20, #28, #8",
|
||||
"bfi x22, x21, #36, #1",
|
||||
"bfi x22, x20, #19, #8",
|
||||
"bfi x22, x21, #27, #1",
|
||||
"bfxil x22, x20, #0, #8",
|
||||
"ror x20, x22, #62",
|
||||
"bfxil x4, x20, #0, #8",
|
||||
"ror x20, x21, #61",
|
||||
"ror x20, x22, #61",
|
||||
"eor x20, x20, #0x1",
|
||||
"rmif x20, #63, #nzCv"
|
||||
]
|
||||
@@ -932,7 +932,7 @@
|
||||
"ExpectedInstructionCount": 10,
|
||||
"Comment": "GROUP2 0xC0 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"uxtb w21, w4",
|
||||
"bfi x21, x20, #8, #1",
|
||||
"bfi x21, x21, #9, #9",
|
||||
@@ -1044,18 +1044,18 @@
|
||||
"Comment": "GROUP2 0xC1 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxth w20, w4",
|
||||
"mov w21, #0x0",
|
||||
"cset w22, lo",
|
||||
"bfi x21, x20, #47, #16",
|
||||
"bfi x21, x22, #63, #1",
|
||||
"bfi x21, x20, #30, #16",
|
||||
"bfi x21, x22, #46, #1",
|
||||
"bfi x21, x20, #13, #16",
|
||||
"bfi x21, x22, #29, #1",
|
||||
"bfxil x21, x20, #0, #16",
|
||||
"ror x20, x21, #62",
|
||||
"cset x21, lo",
|
||||
"mov w22, #0x0",
|
||||
"bfi x22, x20, #47, #16",
|
||||
"bfi x22, x21, #63, #1",
|
||||
"bfi x22, x20, #30, #16",
|
||||
"bfi x22, x21, #46, #1",
|
||||
"bfi x22, x20, #13, #16",
|
||||
"bfi x22, x21, #29, #1",
|
||||
"bfxil x22, x20, #0, #16",
|
||||
"ror x20, x22, #62",
|
||||
"bfxil x4, x20, #0, #16",
|
||||
"ror x20, x21, #61",
|
||||
"ror x20, x22, #61",
|
||||
"eor x20, x20, #0x1",
|
||||
"rmif x20, #63, #nzCv"
|
||||
]
|
||||
@@ -1065,7 +1065,7 @@
|
||||
"Comment": "GROUP2 0xC1 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"lsl w20, w4, #2",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr w20, w20, w4, lsr #31",
|
||||
"eor x22, x4, #0x40000000",
|
||||
"rmif x22, #29, #nzCv",
|
||||
@@ -1077,7 +1077,7 @@
|
||||
"Comment": "GROUP2 0xC1 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"lsl x20, x4, #2",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr x20, x20, x4, lsr #63",
|
||||
"eor x22, x4, #0x4000000000000000",
|
||||
"rmif x22, #61, #nzCv",
|
||||
@@ -1088,7 +1088,7 @@
|
||||
"ExpectedInstructionCount": 9,
|
||||
"Comment": "GROUP2 0xC1 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"uxth w21, w4",
|
||||
"bfi x21, x20, #16, #1",
|
||||
"bfi x21, x21, #17, #17",
|
||||
@@ -1104,7 +1104,7 @@
|
||||
"Comment": "GROUP2 0xC1 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"lsr w20, w4, #2",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr w20, w20, w4, lsl #31",
|
||||
"eor x22, x4, #0x2",
|
||||
"rmif x22, #0, #nzCv",
|
||||
@@ -1116,7 +1116,7 @@
|
||||
"Comment": "GROUP2 0xC1 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"lsr x20, x4, #2",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr x20, x20, x4, lsl #63",
|
||||
"eor x22, x4, #0x2",
|
||||
"rmif x22, #0, #nzCv",
|
||||
@@ -1257,7 +1257,7 @@
|
||||
"Comment": "GROUP2 0xd0 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxtb w20, w4",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr w21, w21, w20, lsl #1",
|
||||
"eor x22, x20, #0x80",
|
||||
"rmif x22, #6, #nzCv",
|
||||
@@ -1271,7 +1271,7 @@
|
||||
"Comment": "GROUP2 0xd0 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxtb w20, w4",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"eor x22, x20, #0x1",
|
||||
"rmif x22, #63, #nzCv",
|
||||
"ubfx w20, w20, #1, #7",
|
||||
@@ -1396,7 +1396,7 @@
|
||||
"Comment": "GROUP2 0xd1 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxth w20, w4",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr w21, w21, w20, lsl #1",
|
||||
"eor x22, x20, #0x8000",
|
||||
"rmif x22, #14, #nzCv",
|
||||
@@ -1410,7 +1410,7 @@
|
||||
"Comment": "GROUP2 0xd1 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, w4",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr w4, w21, w20, lsl #1",
|
||||
"eor x21, x20, #0x80000000",
|
||||
"rmif x21, #30, #nzCv",
|
||||
@@ -1422,7 +1422,7 @@
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Comment": "GROUP2 0xd1 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"orr x20, x20, x4, lsl #1",
|
||||
"eor x21, x4, #0x8000000000000000",
|
||||
"rmif x21, #62, #nzCv",
|
||||
@@ -1435,7 +1435,7 @@
|
||||
"ExpectedInstructionCount": 8,
|
||||
"Comment": "GROUP2 0xd1 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x4, #0x1",
|
||||
"rmif x21, #63, #nzCv",
|
||||
"ubfx w21, w4, #1, #15",
|
||||
@@ -1449,7 +1449,7 @@
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Comment": "GROUP2 0xd1 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x4, #0x1",
|
||||
"rmif x21, #63, #nzCv",
|
||||
"extr w4, w20, w4, #1",
|
||||
@@ -1461,7 +1461,7 @@
|
||||
"ExpectedInstructionCount": 6,
|
||||
"Comment": "GROUP2 0xd1 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x4, #0x1",
|
||||
"rmif x21, #63, #nzCv",
|
||||
"extr x4, x20, x4, #1",
|
||||
@@ -1622,25 +1622,25 @@
|
||||
"cbz x20, #+0x68",
|
||||
"and x20, x7, #0x1f",
|
||||
"uxtb w21, w4",
|
||||
"mov w22, #0x0",
|
||||
"cset w23, lo",
|
||||
"bfi x22, x21, #55, #8",
|
||||
"bfi x22, x23, #63, #1",
|
||||
"bfi x22, x21, #46, #8",
|
||||
"bfi x22, x23, #54, #1",
|
||||
"bfi x22, x21, #37, #8",
|
||||
"bfi x22, x23, #45, #1",
|
||||
"bfi x22, x21, #28, #8",
|
||||
"bfi x22, x23, #36, #1",
|
||||
"bfi x22, x21, #19, #8",
|
||||
"bfi x22, x23, #27, #1",
|
||||
"bfxil x22, x21, #0, #8",
|
||||
"cset x22, lo",
|
||||
"mov w23, #0x0",
|
||||
"bfi x23, x21, #55, #8",
|
||||
"bfi x23, x22, #63, #1",
|
||||
"bfi x23, x21, #46, #8",
|
||||
"bfi x23, x22, #54, #1",
|
||||
"bfi x23, x21, #37, #8",
|
||||
"bfi x23, x22, #45, #1",
|
||||
"bfi x23, x21, #28, #8",
|
||||
"bfi x23, x22, #36, #1",
|
||||
"bfi x23, x21, #19, #8",
|
||||
"bfi x23, x22, #27, #1",
|
||||
"bfxil x23, x21, #0, #8",
|
||||
"neg x21, x20",
|
||||
"ror x21, x22, x21",
|
||||
"ror x21, x23, x21",
|
||||
"bfxil x4, x21, #0, #8",
|
||||
"mov w23, #0x3f",
|
||||
"sub x20, x23, x20",
|
||||
"ror x20, x22, x20",
|
||||
"mov w22, #0x3f",
|
||||
"sub x20, x22, x20",
|
||||
"ror x20, x23, x20",
|
||||
"eor x22, x20, #0x1",
|
||||
"rmif x22, #63, #nzCv",
|
||||
"eor x20, x20, x21, lsr #7",
|
||||
@@ -1654,7 +1654,7 @@
|
||||
"and x20, x7, #0x1f",
|
||||
"cbz x20, #+0x40",
|
||||
"and x20, x7, #0x1f",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"uxtb w22, w4",
|
||||
"bfi x22, x21, #8, #1",
|
||||
"bfi x22, x22, #9, #9",
|
||||
@@ -1817,21 +1817,21 @@
|
||||
"cbz x20, #+0x58",
|
||||
"and x20, x7, #0x1f",
|
||||
"uxth w21, w4",
|
||||
"mov w22, #0x0",
|
||||
"cset w23, lo",
|
||||
"bfi x22, x21, #47, #16",
|
||||
"bfi x22, x23, #63, #1",
|
||||
"bfi x22, x21, #30, #16",
|
||||
"bfi x22, x23, #46, #1",
|
||||
"bfi x22, x21, #13, #16",
|
||||
"bfi x22, x23, #29, #1",
|
||||
"bfxil x22, x21, #0, #16",
|
||||
"cset x22, lo",
|
||||
"mov w23, #0x0",
|
||||
"bfi x23, x21, #47, #16",
|
||||
"bfi x23, x22, #63, #1",
|
||||
"bfi x23, x21, #30, #16",
|
||||
"bfi x23, x22, #46, #1",
|
||||
"bfi x23, x21, #13, #16",
|
||||
"bfi x23, x22, #29, #1",
|
||||
"bfxil x23, x21, #0, #16",
|
||||
"neg x21, x20",
|
||||
"ror x21, x22, x21",
|
||||
"ror x21, x23, x21",
|
||||
"bfxil x4, x21, #0, #16",
|
||||
"mov w23, #0x3f",
|
||||
"sub x20, x23, x20",
|
||||
"ror x20, x22, x20",
|
||||
"mov w22, #0x3f",
|
||||
"sub x20, x22, x20",
|
||||
"ror x20, x23, x20",
|
||||
"eor x22, x20, #0x1",
|
||||
"rmif x22, #63, #nzCv",
|
||||
"eor x20, x20, x21, lsr #15",
|
||||
@@ -1845,7 +1845,7 @@
|
||||
"and w20, w7, #0x1f",
|
||||
"cbz x20, #+0x3c",
|
||||
"lsl w20, w4, w7",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"neg w22, w7",
|
||||
"lsr w23, w4, w22",
|
||||
"orr w20, w20, w23, lsr #1",
|
||||
@@ -1868,7 +1868,7 @@
|
||||
"and x20, x7, #0x3f",
|
||||
"cbz x20, #+0x38",
|
||||
"lsl x20, x4, x7",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"neg x22, x7",
|
||||
"lsr x23, x4, x22",
|
||||
"orr x20, x20, x23, lsr #1",
|
||||
@@ -1889,7 +1889,7 @@
|
||||
"and x20, x7, #0x1f",
|
||||
"cbz x20, #+0x3c",
|
||||
"and x20, x7, #0x1f",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"uxth w22, w4",
|
||||
"bfi x22, x21, #16, #1",
|
||||
"bfi x22, x22, #17, #17",
|
||||
@@ -1911,7 +1911,7 @@
|
||||
"and w20, w7, #0x1f",
|
||||
"cbz x20, #+0x3c",
|
||||
"lsr w20, w4, w7",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"neg w22, w7",
|
||||
"lsl w23, w4, w22",
|
||||
"orr w20, w20, w23, lsl #1",
|
||||
@@ -1934,7 +1934,7 @@
|
||||
"and x20, x7, #0x3f",
|
||||
"cbz x20, #+0x38",
|
||||
"lsr x20, x4, x7",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"neg x22, x7",
|
||||
"lsl x23, x4, x22",
|
||||
"orr x20, x20, x23, lsl #1",
|
||||
@@ -2386,7 +2386,7 @@
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Comment": "GROUP4 0xfe /0",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"adds w26, w4, #0x1 (1)",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"mov x27, x4",
|
||||
@@ -2397,7 +2397,7 @@
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Comment": "GROUP4 0xfe /0",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"adds x26, x4, #0x1 (1)",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"mov x27, x4",
|
||||
@@ -2420,7 +2420,7 @@
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Comment": "GROUP4 0xfe /1",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"subs w26, w4, #0x1 (1)",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"mov x27, x4",
|
||||
@@ -2431,7 +2431,7 @@
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Comment": "GROUP4 0xfe /1",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"subs x26, x4, #0x1 (1)",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"mov x27, x4",
|
||||
|
||||
@@ -107,7 +107,7 @@
|
||||
"Comment": "0x27",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxtb w20, w4",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"and x22, x20, #0xf",
|
||||
"cmp x22, #0x9 (9)",
|
||||
"cset x22, hi",
|
||||
@@ -134,7 +134,7 @@
|
||||
"Comment": "0x2f",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxtb w20, w4",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"and x22, x20, #0xf",
|
||||
"cmp x22, #0x9 (9)",
|
||||
"cset x22, hi",
|
||||
@@ -214,7 +214,7 @@
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Comment": "0x40",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"adds w26, w4, #0x1 (1)",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"mov x27, x4",
|
||||
@@ -251,7 +251,7 @@
|
||||
"ExpectedInstructionCount": 5,
|
||||
"Comment": "0x48",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"subs w26, w4, #0x1 (1)",
|
||||
"rmif x20, #63, #nzCv",
|
||||
"mov x27, x4",
|
||||
@@ -310,7 +310,7 @@
|
||||
"ExpectedInstructionCount": 39,
|
||||
"Comment": "0x9c",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x27, x26",
|
||||
"ubfx w21, w21, #4, #1",
|
||||
"orr x20, x20, x21, lsl #4",
|
||||
@@ -321,7 +321,7 @@
|
||||
"ldrsb x21, [x28, #986]",
|
||||
"lsr x21, x21, #63",
|
||||
"orr x20, x20, x21, lsl #10",
|
||||
"cset w21, vs",
|
||||
"cset x21, vs",
|
||||
"orr x20, x20, x21, lsl #11",
|
||||
"ldrb w21, [x28, #988]",
|
||||
"orr x20, x20, x21, lsl #12",
|
||||
@@ -355,7 +355,7 @@
|
||||
"ExpectedInstructionCount": 39,
|
||||
"Comment": "0x9c",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x27, x26",
|
||||
"ubfx w21, w21, #4, #1",
|
||||
"orr x20, x20, x21, lsl #4",
|
||||
@@ -366,7 +366,7 @@
|
||||
"ldrsb x21, [x28, #986]",
|
||||
"lsr x21, x21, #63",
|
||||
"orr x20, x20, x21, lsl #10",
|
||||
"cset w21, vs",
|
||||
"cset x21, vs",
|
||||
"orr x20, x20, x21, lsl #11",
|
||||
"ldrb w21, [x28, #988]",
|
||||
"orr x20, x20, x21, lsl #12",
|
||||
|
||||
@@ -18,8 +18,8 @@
|
||||
"Comment": "0x0f 0x2e",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"axflag"
|
||||
]
|
||||
},
|
||||
@@ -28,8 +28,8 @@
|
||||
"Comment": "0x0f 0x2f",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"axflag"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -739,9 +739,9 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mrs x20, rndr",
|
||||
"bfxil x4, x20, #0, #16",
|
||||
"cset x20, eq",
|
||||
"mov w26, #0x1",
|
||||
"mov w27, #0x0",
|
||||
"cset w20, eq",
|
||||
"rmif x20, #63, #NZCV"
|
||||
]
|
||||
},
|
||||
@@ -751,9 +751,9 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mrs x20, rndr",
|
||||
"mov w4, w20",
|
||||
"cset x20, eq",
|
||||
"mov w26, #0x1",
|
||||
"mov w27, #0x0",
|
||||
"cset w20, eq",
|
||||
"rmif x20, #63, #NZCV"
|
||||
]
|
||||
},
|
||||
@@ -762,9 +762,9 @@
|
||||
"Comment": "GROUP9 0x0F 0xC7 /6",
|
||||
"ExpectedArm64ASM": [
|
||||
"mrs x4, rndr",
|
||||
"cset x20, eq",
|
||||
"mov w26, #0x1",
|
||||
"mov w27, #0x0",
|
||||
"cset w20, eq",
|
||||
"rmif x20, #63, #NZCV"
|
||||
]
|
||||
},
|
||||
@@ -774,9 +774,9 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mrs x20, rndrrs",
|
||||
"bfxil x4, x20, #0, #16",
|
||||
"cset x20, eq",
|
||||
"mov w26, #0x1",
|
||||
"mov w27, #0x0",
|
||||
"cset w20, eq",
|
||||
"rmif x20, #63, #NZCV"
|
||||
]
|
||||
},
|
||||
@@ -786,9 +786,9 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mrs x20, rndrrs",
|
||||
"mov w4, w20",
|
||||
"cset x20, eq",
|
||||
"mov w26, #0x1",
|
||||
"mov w27, #0x0",
|
||||
"cset w20, eq",
|
||||
"rmif x20, #63, #NZCV"
|
||||
]
|
||||
},
|
||||
@@ -797,9 +797,9 @@
|
||||
"Comment": "GROUP9 0x0F 0xC7 /7",
|
||||
"ExpectedArm64ASM": [
|
||||
"mrs x4, rndrrs",
|
||||
"cset x20, eq",
|
||||
"mov w26, #0x1",
|
||||
"mov w27, #0x0",
|
||||
"cset w20, eq",
|
||||
"rmif x20, #63, #NZCV"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -18,8 +18,8 @@
|
||||
"Comment": "0x66 0x0f 0x2e",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"axflag"
|
||||
]
|
||||
},
|
||||
@@ -28,8 +28,8 @@
|
||||
"Comment": "0x66 0x0f 0x2f",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"axflag"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -21,8 +21,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"axflag"
|
||||
]
|
||||
},
|
||||
@@ -33,8 +33,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"axflag"
|
||||
]
|
||||
},
|
||||
@@ -45,8 +45,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"axflag"
|
||||
]
|
||||
},
|
||||
@@ -57,8 +57,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"axflag"
|
||||
]
|
||||
},
|
||||
|
||||
@@ -29,11 +29,11 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"vtestps ymm0, ymm1": {
|
||||
@@ -53,11 +53,11 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"vtestpd xmm0, xmm1": {
|
||||
@@ -77,11 +77,11 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"vtestpd ymm0, ymm1": {
|
||||
@@ -101,11 +101,11 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"vptest xmm0, xmm1": {
|
||||
@@ -121,11 +121,11 @@
|
||||
"umov w20, v2.h[0]",
|
||||
"umov w21, v3.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"vptest ymm0, ymm1": {
|
||||
@@ -141,11 +141,11 @@
|
||||
"umov w20, v2.h[0]",
|
||||
"umov w21, v3.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"rmif x21, #63, #nzCv"
|
||||
"rmif x21, #63, #nzCv",
|
||||
"mov w26, #0x1"
|
||||
]
|
||||
},
|
||||
"vmaskmovps xmm0, xmm1, [rax]": {
|
||||
@@ -375,7 +375,7 @@
|
||||
"bic w20, w6, w20",
|
||||
"tst x7, #0xe0",
|
||||
"csel w4, w6, w20, ne",
|
||||
"cset w20, eq",
|
||||
"cset x20, eq",
|
||||
"cmp w4, #0x0 (0)",
|
||||
"rmif x20, #63, #nzCv"
|
||||
]
|
||||
@@ -391,7 +391,7 @@
|
||||
"bic x20, x6, x20",
|
||||
"tst x7, #0xc0",
|
||||
"csel x4, x6, x20, ne",
|
||||
"cset w20, eq",
|
||||
"cset x20, eq",
|
||||
"cmp x4, #0x0 (0)",
|
||||
"rmif x20, #63, #nzCv"
|
||||
]
|
||||
|
||||
@@ -76,7 +76,7 @@
|
||||
"neg w20, w6",
|
||||
"and w4, w6, w20",
|
||||
"cmp w4, #0x0 (0)",
|
||||
"cset w20, eq",
|
||||
"cset x20, eq",
|
||||
"rmif x20, #63, #nzCv"
|
||||
]
|
||||
},
|
||||
@@ -89,7 +89,7 @@
|
||||
"neg x20, x6",
|
||||
"and x4, x6, x20",
|
||||
"cmp x4, #0x0 (0)",
|
||||
"cset w20, eq",
|
||||
"cset x20, eq",
|
||||
"rmif x20, #63, #nzCv"
|
||||
]
|
||||
}
|
||||
|
||||
@@ -481,26 +481,26 @@
|
||||
"strb w20, [x28, #1019]",
|
||||
"add x0, x28, x20, lsl #4",
|
||||
"str q4, [x0, #1040]",
|
||||
"mov w21, #0x1",
|
||||
"add w22, w20, #0x1 (1)",
|
||||
"and w22, w22, #0x7",
|
||||
"add x0, x28, x22, lsl #4",
|
||||
"add w21, w20, #0x1 (1)",
|
||||
"and w21, w21, #0x7",
|
||||
"add x0, x28, x21, lsl #4",
|
||||
"str q7, [x0, #1040]",
|
||||
"add w22, w20, #0x2 (2)",
|
||||
"and w22, w22, #0x7",
|
||||
"add x0, x28, x22, lsl #4",
|
||||
"add w21, w20, #0x2 (2)",
|
||||
"and w21, w21, #0x7",
|
||||
"add x0, x28, x21, lsl #4",
|
||||
"str q2, [x0, #1040]",
|
||||
"add w22, w20, #0x3 (3)",
|
||||
"and w22, w22, #0x7",
|
||||
"mov w23, #0x8",
|
||||
"sub w20, w23, w20",
|
||||
"ldrb w23, [x28, #1426]",
|
||||
"mov w12, #0x707",
|
||||
"lsr w20, w12, w20",
|
||||
"orr w20, w23, w20",
|
||||
"add w21, w20, #0x3 (3)",
|
||||
"and w21, w21, #0x7",
|
||||
"mov w22, #0x8",
|
||||
"sub w20, w22, w20",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"mov w23, #0x707",
|
||||
"lsr w20, w23, w20",
|
||||
"orr w20, w22, w20",
|
||||
"strb w20, [x28, #1426]",
|
||||
"ldrb w20, [x28, #1426]",
|
||||
"lsl w21, w21, w22",
|
||||
"mov w22, #0x1",
|
||||
"lsl w21, w22, w21",
|
||||
"bic w20, w20, w21",
|
||||
"strb w20, [x28, #1426]"
|
||||
]
|
||||
@@ -743,22 +743,22 @@
|
||||
"strb w20, [x28, #1019]",
|
||||
"add x0, x28, x20, lsl #4",
|
||||
"str q3, [x0, #1040]",
|
||||
"mov w21, #0x1",
|
||||
"add w22, w20, #0x1 (1)",
|
||||
"and w22, w22, #0x7",
|
||||
"add x0, x28, x22, lsl #4",
|
||||
"add w21, w20, #0x1 (1)",
|
||||
"and w21, w21, #0x7",
|
||||
"add x0, x28, x21, lsl #4",
|
||||
"str q2, [x0, #1040]",
|
||||
"add w22, w20, #0x6 (6)",
|
||||
"and w22, w22, #0x7",
|
||||
"mov w23, #0x8",
|
||||
"sub w20, w23, w20",
|
||||
"ldrb w23, [x28, #1426]",
|
||||
"mov w12, #0x303",
|
||||
"lsr w20, w12, w20",
|
||||
"orr w20, w23, w20",
|
||||
"add w21, w20, #0x6 (6)",
|
||||
"and w21, w21, #0x7",
|
||||
"mov w22, #0x8",
|
||||
"sub w20, w22, w20",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"mov w23, #0x303",
|
||||
"lsr w20, w23, w20",
|
||||
"orr w20, w22, w20",
|
||||
"strb w20, [x28, #1426]",
|
||||
"ldrb w20, [x28, #1426]",
|
||||
"lsl w21, w21, w22",
|
||||
"mov w22, #0x1",
|
||||
"lsl w21, w22, w21",
|
||||
"bic w20, w20, w21",
|
||||
"strb w20, [x28, #1426]"
|
||||
]
|
||||
@@ -1003,27 +1003,26 @@
|
||||
"ldr x30, [sp], #16",
|
||||
"mov v2.16b, v0.16b",
|
||||
"ldrb w20, [x28, #1019]",
|
||||
"mov w21, #0x1",
|
||||
"sub w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
"strb w20, [x28, #1019]",
|
||||
"add x0, x28, x20, lsl #4",
|
||||
"str q2, [x0, #1040]",
|
||||
"add w22, w20, #0x2 (2)",
|
||||
"and w22, w22, #0x7",
|
||||
"ldrb w23, [x28, #1426]",
|
||||
"lsl w20, w21, w20",
|
||||
"orr w20, w23, w20",
|
||||
"add w21, w20, #0x2 (2)",
|
||||
"and w21, w21, #0x7",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"mov w23, #0x1",
|
||||
"lsl w20, w23, w20",
|
||||
"orr w20, w22, w20",
|
||||
"strb w20, [x28, #1426]",
|
||||
"ldrb w20, [x28, #1426]",
|
||||
"lsl w21, w21, w22",
|
||||
"lsl w21, w23, w21",
|
||||
"bic w20, w20, w21",
|
||||
"strb w20, [x28, #1426]",
|
||||
"ldrb w20, [x28, #1019]",
|
||||
"mov w21, #0x1",
|
||||
"add w22, w20, #0x1 (1)",
|
||||
"and w22, w22, #0x7",
|
||||
"add x0, x28, x22, lsl #4",
|
||||
"add w21, w20, #0x1 (1)",
|
||||
"and w21, w21, #0x7",
|
||||
"add x0, x28, x21, lsl #4",
|
||||
"ldr q2, [x0, #1040]",
|
||||
"add x0, x28, x20, lsl #4",
|
||||
"ldr q3, [x0, #1040]",
|
||||
@@ -1035,12 +1034,13 @@
|
||||
"blr x0",
|
||||
"ldr x30, [sp], #16",
|
||||
"mov v2.16b, v0.16b",
|
||||
"add x0, x28, x22, lsl #4",
|
||||
"add x0, x28, x21, lsl #4",
|
||||
"str q2, [x0, #1040]",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"lsl w23, w21, w20",
|
||||
"bic w22, w22, w23",
|
||||
"strb w22, [x28, #1426]",
|
||||
"ldrb w21, [x28, #1426]",
|
||||
"mov w22, #0x1",
|
||||
"lsl w23, w22, w20",
|
||||
"bic w21, w21, w23",
|
||||
"strb w21, [x28, #1426]",
|
||||
"add w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
"strb w20, [x28, #1019]",
|
||||
@@ -1054,9 +1054,9 @@
|
||||
"ldr x30, [sp], #16",
|
||||
"fmov s2, s0",
|
||||
"str s2, [x10]",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"lsl w21, w21, w20",
|
||||
"bic w21, w22, w21",
|
||||
"ldrb w21, [x28, #1426]",
|
||||
"lsl w22, w22, w20",
|
||||
"bic w21, w21, w22",
|
||||
"strb w21, [x28, #1426]",
|
||||
"add w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
@@ -1876,30 +1876,30 @@
|
||||
"strb w20, [x28, #1019]",
|
||||
"add x0, x28, x20, lsl #4",
|
||||
"str q5, [x0, #1040]",
|
||||
"mov w21, #0x1",
|
||||
"add w22, w20, #0x1 (1)",
|
||||
"and w22, w22, #0x7",
|
||||
"add x0, x28, x22, lsl #4",
|
||||
"add w21, w20, #0x1 (1)",
|
||||
"and w21, w21, #0x7",
|
||||
"add x0, x28, x21, lsl #4",
|
||||
"str q4, [x0, #1040]",
|
||||
"add w22, w20, #0x2 (2)",
|
||||
"and w22, w22, #0x7",
|
||||
"add x0, x28, x22, lsl #4",
|
||||
"add w21, w20, #0x2 (2)",
|
||||
"and w21, w21, #0x7",
|
||||
"add x0, x28, x21, lsl #4",
|
||||
"str q3, [x0, #1040]",
|
||||
"add w22, w20, #0x3 (3)",
|
||||
"and w22, w22, #0x7",
|
||||
"add x0, x28, x22, lsl #4",
|
||||
"add w21, w20, #0x3 (3)",
|
||||
"and w21, w21, #0x7",
|
||||
"add x0, x28, x21, lsl #4",
|
||||
"str q2, [x0, #1040]",
|
||||
"add w22, w20, #0x7 (7)",
|
||||
"and w22, w22, #0x7",
|
||||
"mov w23, #0x8",
|
||||
"sub w20, w23, w20",
|
||||
"ldrb w23, [x28, #1426]",
|
||||
"mov w12, #0xf0f",
|
||||
"lsr w20, w12, w20",
|
||||
"orr w20, w23, w20",
|
||||
"add w21, w20, #0x7 (7)",
|
||||
"and w21, w21, #0x7",
|
||||
"mov w22, #0x8",
|
||||
"sub w20, w22, w20",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"mov w23, #0xf0f",
|
||||
"lsr w20, w23, w20",
|
||||
"orr w20, w22, w20",
|
||||
"strb w20, [x28, #1426]",
|
||||
"ldrb w20, [x28, #1426]",
|
||||
"lsl w21, w21, w22",
|
||||
"mov w22, #0x1",
|
||||
"lsl w21, w22, w21",
|
||||
"bic w20, w20, w21",
|
||||
"strb w20, [x28, #1426]"
|
||||
]
|
||||
@@ -2147,27 +2147,26 @@
|
||||
"ldr x30, [sp], #16",
|
||||
"mov v2.16b, v0.16b",
|
||||
"ldrb w20, [x28, #1019]",
|
||||
"mov w21, #0x1",
|
||||
"sub w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
"strb w20, [x28, #1019]",
|
||||
"add x0, x28, x20, lsl #4",
|
||||
"str q2, [x0, #1040]",
|
||||
"add w22, w20, #0x4 (4)",
|
||||
"and w22, w22, #0x7",
|
||||
"ldrb w23, [x28, #1426]",
|
||||
"lsl w20, w21, w20",
|
||||
"orr w20, w23, w20",
|
||||
"add w21, w20, #0x4 (4)",
|
||||
"and w21, w21, #0x7",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"mov w23, #0x1",
|
||||
"lsl w20, w23, w20",
|
||||
"orr w20, w22, w20",
|
||||
"strb w20, [x28, #1426]",
|
||||
"ldrb w20, [x28, #1426]",
|
||||
"lsl w21, w21, w22",
|
||||
"lsl w21, w23, w21",
|
||||
"bic w20, w20, w21",
|
||||
"strb w20, [x28, #1426]",
|
||||
"ldrb w20, [x28, #1019]",
|
||||
"mov w21, #0x1",
|
||||
"add w22, w20, #0x1 (1)",
|
||||
"and w22, w22, #0x7",
|
||||
"add x0, x28, x22, lsl #4",
|
||||
"add w21, w20, #0x1 (1)",
|
||||
"and w21, w21, #0x7",
|
||||
"add x0, x28, x21, lsl #4",
|
||||
"ldr q2, [x0, #1040]",
|
||||
"add x0, x28, x20, lsl #4",
|
||||
"ldr q3, [x0, #1040]",
|
||||
@@ -2191,9 +2190,10 @@
|
||||
"ldr x30, [sp], #16",
|
||||
"fmov s2, s0",
|
||||
"str s2, [x11]",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"lsl w21, w21, w20",
|
||||
"bic w21, w22, w21",
|
||||
"ldrb w21, [x28, #1426]",
|
||||
"mov w22, #0x1",
|
||||
"lsl w22, w22, w20",
|
||||
"bic w21, w21, w22",
|
||||
"strb w21, [x28, #1426]",
|
||||
"add w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
@@ -2232,15 +2232,15 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldrb w20, [x28, #1019]",
|
||||
"mov w21, #0x1",
|
||||
"add w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
"strb w20, [x28, #1019]",
|
||||
"add w20, w20, #0x7 (7)",
|
||||
"and w20, w20, #0x7",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"lsl w20, w21, w20",
|
||||
"bic w20, w22, w20",
|
||||
"ldrb w21, [x28, #1426]",
|
||||
"mov w22, #0x1",
|
||||
"lsl w20, w22, w20",
|
||||
"bic w20, w21, w20",
|
||||
"strb w20, [x28, #1426]",
|
||||
"ldrb w20, [x28, #1019]",
|
||||
"add x0, x28, x20, lsl #4",
|
||||
@@ -2437,15 +2437,15 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"ldrb w20, [x28, #1019]",
|
||||
"mov w21, #0x1",
|
||||
"add w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
"strb w20, [x28, #1019]",
|
||||
"add w20, w20, #0x7 (7)",
|
||||
"and w20, w20, #0x7",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"lsl w20, w21, w20",
|
||||
"bic w20, w22, w20",
|
||||
"ldrb w21, [x28, #1426]",
|
||||
"mov w22, #0x1",
|
||||
"lsl w20, w22, w20",
|
||||
"bic w20, w21, w20",
|
||||
"strb w20, [x28, #1426]",
|
||||
"ldrb w20, [x28, #1019]",
|
||||
"add x0, x28, x20, lsl #4",
|
||||
|
||||
@@ -17712,12 +17712,12 @@
|
||||
"sub w20, w20, w21",
|
||||
"str w20, [x8, #-4]!",
|
||||
"ldrb w21, [x28, #1019]",
|
||||
"mov w22, #0x1",
|
||||
"add w21, w21, #0x1 (1)",
|
||||
"and w21, w21, #0x7",
|
||||
"ldrb w23, [x28, #1426]",
|
||||
"lsl w21, w22, w21",
|
||||
"bic w21, w23, w21",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"mov w23, #0x1",
|
||||
"lsl w21, w23, w21",
|
||||
"bic w21, w22, w21",
|
||||
"strb w21, [x28, #1426]"
|
||||
]
|
||||
},
|
||||
@@ -20226,7 +20226,7 @@
|
||||
"mov w22, #0x0",
|
||||
"cmp x20, #0x0 (0)",
|
||||
"mov w23, #0x8000",
|
||||
"csel x12, x23, xzr, lt",
|
||||
"csel x12, x23, x22, lt",
|
||||
"cneg x20, x20, mi",
|
||||
"mov w13, #0x3f",
|
||||
"mov x0, #0x3f",
|
||||
@@ -20277,7 +20277,7 @@
|
||||
"sxtw x20, w20",
|
||||
"mrs x21, nzcv",
|
||||
"cmp x20, #0x0 (0)",
|
||||
"csel x23, x23, xzr, lt",
|
||||
"csel x23, x23, x22, lt",
|
||||
"cneg x20, x20, mi",
|
||||
"mov x0, #0x3f",
|
||||
"clz x12, x20",
|
||||
@@ -26201,12 +26201,12 @@
|
||||
"fmov s2, s0",
|
||||
"str s2, [x7]",
|
||||
"ldrb w20, [x28, #1019]",
|
||||
"mov w21, #0x1",
|
||||
"add w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"lsl w20, w21, w20",
|
||||
"bic w20, w22, w20",
|
||||
"ldrb w21, [x28, #1426]",
|
||||
"mov w22, #0x1",
|
||||
"lsl w20, w22, w20",
|
||||
"bic w20, w21, w20",
|
||||
"strb w20, [x28, #1426]"
|
||||
]
|
||||
},
|
||||
@@ -26549,15 +26549,15 @@
|
||||
"sub w20, w20, w21",
|
||||
"str w20, [x8, #-4]!",
|
||||
"ldrb w21, [x28, #1019]",
|
||||
"mov w22, #0x1",
|
||||
"sub w21, w21, #0x1 (1)",
|
||||
"and w21, w21, #0x7",
|
||||
"strb w21, [x28, #1019]",
|
||||
"add x0, x28, x21, lsl #4",
|
||||
"str q2, [x0, #1040]",
|
||||
"ldrb w23, [x28, #1426]",
|
||||
"lsl w21, w22, w21",
|
||||
"orr w21, w23, w21",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"mov w23, #0x1",
|
||||
"lsl w21, w23, w21",
|
||||
"orr w21, w22, w21",
|
||||
"strb w21, [x28, #1426]"
|
||||
]
|
||||
},
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
@@ -450,12 +450,12 @@
|
||||
"umov w20, v2.h[0]",
|
||||
"umov w21, v3.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
@@ -734,7 +734,7 @@
|
||||
"eor w21, w20, #0x20000000",
|
||||
"msr nzcv, x21",
|
||||
"adcs w4, w6, w4",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
@@ -749,7 +749,7 @@
|
||||
"eor w21, w20, #0x20000000",
|
||||
"msr nzcv, x21",
|
||||
"adcs x4, x6, x4",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
@@ -763,7 +763,7 @@
|
||||
"mrs x20, nzcv",
|
||||
"ccmp wzr, #0, #nzcv, vs",
|
||||
"adcs w4, w6, w4",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"bfi w20, w21, #28, #1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
@@ -777,7 +777,7 @@
|
||||
"mrs x20, nzcv",
|
||||
"ccmp wzr, #0, #nzcv, vs",
|
||||
"adcs x4, x6, x4",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"bfi w20, w21, #28, #1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
|
||||
@@ -2640,7 +2640,7 @@
|
||||
"ExpectedInstructionCount": 39,
|
||||
"Comment": "0x9c",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x27, x26",
|
||||
"ubfx w21, w21, #4, #1",
|
||||
"orr x20, x20, x21, lsl #4",
|
||||
@@ -2651,7 +2651,7 @@
|
||||
"ldrsb x21, [x28, #986]",
|
||||
"lsr x21, x21, #63",
|
||||
"orr x20, x20, x21, lsl #10",
|
||||
"cset w21, vs",
|
||||
"cset x21, vs",
|
||||
"orr x20, x20, x21, lsl #11",
|
||||
"ldrb w21, [x28, #988]",
|
||||
"orr x20, x20, x21, lsl #12",
|
||||
@@ -2685,7 +2685,7 @@
|
||||
"ExpectedInstructionCount": 39,
|
||||
"Comment": "0x9c",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x27, x26",
|
||||
"ubfx w21, w21, #4, #1",
|
||||
"orr x20, x20, x21, lsl #4",
|
||||
@@ -2696,7 +2696,7 @@
|
||||
"ldrsb x21, [x28, #986]",
|
||||
"lsr x21, x21, #63",
|
||||
"orr x20, x20, x21, lsl #10",
|
||||
"cset w21, vs",
|
||||
"cset x21, vs",
|
||||
"orr x20, x20, x21, lsl #11",
|
||||
"ldrb w21, [x28, #988]",
|
||||
"orr x20, x20, x21, lsl #12",
|
||||
@@ -2799,7 +2799,7 @@
|
||||
"ExpectedInstructionCount": 14,
|
||||
"Comment": "0x9f",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x27, x26",
|
||||
"ubfx w21, w21, #4, #1",
|
||||
"orr x20, x20, x21, lsl #4",
|
||||
|
||||
@@ -1017,22 +1017,22 @@
|
||||
"Comment": "GROUP2 0xC0 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxtb w20, w4",
|
||||
"mov w21, #0x0",
|
||||
"cset w22, lo",
|
||||
"bfi x21, x20, #55, #8",
|
||||
"bfi x21, x22, #63, #1",
|
||||
"bfi x21, x20, #46, #8",
|
||||
"bfi x21, x22, #54, #1",
|
||||
"bfi x21, x20, #37, #8",
|
||||
"bfi x21, x22, #45, #1",
|
||||
"bfi x21, x20, #28, #8",
|
||||
"bfi x21, x22, #36, #1",
|
||||
"bfi x21, x20, #19, #8",
|
||||
"bfi x21, x22, #27, #1",
|
||||
"bfxil x21, x20, #0, #8",
|
||||
"ror x20, x21, #62",
|
||||
"cset x21, lo",
|
||||
"mov w22, #0x0",
|
||||
"bfi x22, x20, #55, #8",
|
||||
"bfi x22, x21, #63, #1",
|
||||
"bfi x22, x20, #46, #8",
|
||||
"bfi x22, x21, #54, #1",
|
||||
"bfi x22, x20, #37, #8",
|
||||
"bfi x22, x21, #45, #1",
|
||||
"bfi x22, x20, #28, #8",
|
||||
"bfi x22, x21, #36, #1",
|
||||
"bfi x22, x20, #19, #8",
|
||||
"bfi x22, x21, #27, #1",
|
||||
"bfxil x22, x20, #0, #8",
|
||||
"ror x20, x22, #62",
|
||||
"bfxil x4, x20, #0, #8",
|
||||
"ror x20, x21, #61",
|
||||
"ror x20, x22, #61",
|
||||
"eor x20, x20, #0x1",
|
||||
"ubfx x20, x20, #0, #1",
|
||||
"mrs x21, nzcv",
|
||||
@@ -1044,7 +1044,7 @@
|
||||
"ExpectedInstructionCount": 13,
|
||||
"Comment": "GROUP2 0xC0 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"uxtb w21, w4",
|
||||
"bfi x21, x20, #8, #1",
|
||||
"bfi x21, x21, #9, #9",
|
||||
@@ -1186,18 +1186,18 @@
|
||||
"Comment": "GROUP2 0xC1 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxth w20, w4",
|
||||
"mov w21, #0x0",
|
||||
"cset w22, lo",
|
||||
"bfi x21, x20, #47, #16",
|
||||
"bfi x21, x22, #63, #1",
|
||||
"bfi x21, x20, #30, #16",
|
||||
"bfi x21, x22, #46, #1",
|
||||
"bfi x21, x20, #13, #16",
|
||||
"bfi x21, x22, #29, #1",
|
||||
"bfxil x21, x20, #0, #16",
|
||||
"ror x20, x21, #62",
|
||||
"cset x21, lo",
|
||||
"mov w22, #0x0",
|
||||
"bfi x22, x20, #47, #16",
|
||||
"bfi x22, x21, #63, #1",
|
||||
"bfi x22, x20, #30, #16",
|
||||
"bfi x22, x21, #46, #1",
|
||||
"bfi x22, x20, #13, #16",
|
||||
"bfi x22, x21, #29, #1",
|
||||
"bfxil x22, x20, #0, #16",
|
||||
"ror x20, x22, #62",
|
||||
"bfxil x4, x20, #0, #16",
|
||||
"ror x20, x21, #61",
|
||||
"ror x20, x22, #61",
|
||||
"eor x20, x20, #0x1",
|
||||
"ubfx x20, x20, #0, #1",
|
||||
"mrs x21, nzcv",
|
||||
@@ -1210,7 +1210,7 @@
|
||||
"Comment": "GROUP2 0xC1 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"lsl w20, w4, #2",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr w20, w20, w4, lsr #31",
|
||||
"eor x22, x4, #0x40000000",
|
||||
"ubfx x22, x22, #30, #1",
|
||||
@@ -1225,7 +1225,7 @@
|
||||
"Comment": "GROUP2 0xC1 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"lsl x20, x4, #2",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr x20, x20, x4, lsr #63",
|
||||
"eor x22, x4, #0x4000000000000000",
|
||||
"ubfx x22, x22, #62, #1",
|
||||
@@ -1239,7 +1239,7 @@
|
||||
"ExpectedInstructionCount": 12,
|
||||
"Comment": "GROUP2 0xC1 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"uxth w21, w4",
|
||||
"bfi x21, x20, #16, #1",
|
||||
"bfi x21, x21, #17, #17",
|
||||
@@ -1258,7 +1258,7 @@
|
||||
"Comment": "GROUP2 0xC1 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"lsr w20, w4, #2",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr w20, w20, w4, lsl #31",
|
||||
"eor x22, x4, #0x2",
|
||||
"ubfx x22, x22, #1, #1",
|
||||
@@ -1273,7 +1273,7 @@
|
||||
"Comment": "GROUP2 0xC1 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"lsr x20, x4, #2",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr x20, x20, x4, lsl #63",
|
||||
"eor x22, x4, #0x2",
|
||||
"ubfx x22, x22, #1, #1",
|
||||
@@ -1452,7 +1452,7 @@
|
||||
"Comment": "GROUP2 0xd0 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxtb w20, w4",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr w21, w21, w20, lsl #1",
|
||||
"eor x22, x20, #0x80",
|
||||
"ubfx x22, x22, #7, #1",
|
||||
@@ -1470,7 +1470,7 @@
|
||||
"Comment": "GROUP2 0xd0 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxtb w20, w4",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"eor x22, x20, #0x1",
|
||||
"ubfx x22, x22, #0, #1",
|
||||
"mrs x23, nzcv",
|
||||
@@ -1634,7 +1634,7 @@
|
||||
"Comment": "GROUP2 0xd1 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxth w20, w4",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr w21, w21, w20, lsl #1",
|
||||
"eor x22, x20, #0x8000",
|
||||
"ubfx x22, x22, #15, #1",
|
||||
@@ -1652,7 +1652,7 @@
|
||||
"Comment": "GROUP2 0xd1 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov w20, w4",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"orr w4, w21, w20, lsl #1",
|
||||
"eor x21, x20, #0x80000000",
|
||||
"ubfx x21, x21, #31, #1",
|
||||
@@ -1668,7 +1668,7 @@
|
||||
"ExpectedInstructionCount": 11,
|
||||
"Comment": "GROUP2 0xd1 /2",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"orr x20, x20, x4, lsl #1",
|
||||
"eor x21, x4, #0x8000000000000000",
|
||||
"lsr x21, x21, #63",
|
||||
@@ -1685,7 +1685,7 @@
|
||||
"ExpectedInstructionCount": 12,
|
||||
"Comment": "GROUP2 0xd1 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x4, #0x1",
|
||||
"ubfx x21, x21, #0, #1",
|
||||
"mrs x22, nzcv",
|
||||
@@ -1703,7 +1703,7 @@
|
||||
"ExpectedInstructionCount": 10,
|
||||
"Comment": "GROUP2 0xd1 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x4, #0x1",
|
||||
"ubfx x21, x21, #0, #1",
|
||||
"mrs x22, nzcv",
|
||||
@@ -1719,7 +1719,7 @@
|
||||
"ExpectedInstructionCount": 10,
|
||||
"Comment": "GROUP2 0xd1 /3",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, lo",
|
||||
"cset x20, lo",
|
||||
"eor x21, x4, #0x1",
|
||||
"ubfx x21, x21, #0, #1",
|
||||
"mrs x22, nzcv",
|
||||
@@ -1923,25 +1923,25 @@
|
||||
"cbz x20, #+0x78",
|
||||
"and x20, x7, #0x1f",
|
||||
"uxtb w21, w4",
|
||||
"mov w22, #0x0",
|
||||
"cset w23, lo",
|
||||
"bfi x22, x21, #55, #8",
|
||||
"bfi x22, x23, #63, #1",
|
||||
"bfi x22, x21, #46, #8",
|
||||
"bfi x22, x23, #54, #1",
|
||||
"bfi x22, x21, #37, #8",
|
||||
"bfi x22, x23, #45, #1",
|
||||
"bfi x22, x21, #28, #8",
|
||||
"bfi x22, x23, #36, #1",
|
||||
"bfi x22, x21, #19, #8",
|
||||
"bfi x22, x23, #27, #1",
|
||||
"bfxil x22, x21, #0, #8",
|
||||
"cset x22, lo",
|
||||
"mov w23, #0x0",
|
||||
"bfi x23, x21, #55, #8",
|
||||
"bfi x23, x22, #63, #1",
|
||||
"bfi x23, x21, #46, #8",
|
||||
"bfi x23, x22, #54, #1",
|
||||
"bfi x23, x21, #37, #8",
|
||||
"bfi x23, x22, #45, #1",
|
||||
"bfi x23, x21, #28, #8",
|
||||
"bfi x23, x22, #36, #1",
|
||||
"bfi x23, x21, #19, #8",
|
||||
"bfi x23, x22, #27, #1",
|
||||
"bfxil x23, x21, #0, #8",
|
||||
"neg x21, x20",
|
||||
"ror x21, x22, x21",
|
||||
"ror x21, x23, x21",
|
||||
"bfxil x4, x21, #0, #8",
|
||||
"mov w23, #0x3f",
|
||||
"sub x20, x23, x20",
|
||||
"ror x20, x22, x20",
|
||||
"mov w22, #0x3f",
|
||||
"sub x20, x22, x20",
|
||||
"ror x20, x23, x20",
|
||||
"eor x22, x20, #0x1",
|
||||
"ubfx x22, x22, #0, #1",
|
||||
"mrs x23, nzcv",
|
||||
@@ -1959,7 +1959,7 @@
|
||||
"and x20, x7, #0x1f",
|
||||
"cbz x20, #+0x50",
|
||||
"and x20, x7, #0x1f",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"uxtb w22, w4",
|
||||
"bfi x22, x21, #8, #1",
|
||||
"bfi x22, x22, #9, #9",
|
||||
@@ -2153,21 +2153,21 @@
|
||||
"cbz x20, #+0x68",
|
||||
"and x20, x7, #0x1f",
|
||||
"uxth w21, w4",
|
||||
"mov w22, #0x0",
|
||||
"cset w23, lo",
|
||||
"bfi x22, x21, #47, #16",
|
||||
"bfi x22, x23, #63, #1",
|
||||
"bfi x22, x21, #30, #16",
|
||||
"bfi x22, x23, #46, #1",
|
||||
"bfi x22, x21, #13, #16",
|
||||
"bfi x22, x23, #29, #1",
|
||||
"bfxil x22, x21, #0, #16",
|
||||
"cset x22, lo",
|
||||
"mov w23, #0x0",
|
||||
"bfi x23, x21, #47, #16",
|
||||
"bfi x23, x22, #63, #1",
|
||||
"bfi x23, x21, #30, #16",
|
||||
"bfi x23, x22, #46, #1",
|
||||
"bfi x23, x21, #13, #16",
|
||||
"bfi x23, x22, #29, #1",
|
||||
"bfxil x23, x21, #0, #16",
|
||||
"neg x21, x20",
|
||||
"ror x21, x22, x21",
|
||||
"ror x21, x23, x21",
|
||||
"bfxil x4, x21, #0, #16",
|
||||
"mov w23, #0x3f",
|
||||
"sub x20, x23, x20",
|
||||
"ror x20, x22, x20",
|
||||
"mov w22, #0x3f",
|
||||
"sub x20, x22, x20",
|
||||
"ror x20, x23, x20",
|
||||
"eor x22, x20, #0x1",
|
||||
"ubfx x22, x22, #0, #1",
|
||||
"mrs x23, nzcv",
|
||||
@@ -2185,7 +2185,7 @@
|
||||
"and w20, w7, #0x1f",
|
||||
"cbz x20, #+0x4c",
|
||||
"lsl w20, w4, w7",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"neg w22, w7",
|
||||
"lsr w23, w4, w22",
|
||||
"orr w20, w20, w23, lsr #1",
|
||||
@@ -2212,7 +2212,7 @@
|
||||
"and x20, x7, #0x3f",
|
||||
"cbz x20, #+0x48",
|
||||
"lsl x20, x4, x7",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"neg x22, x7",
|
||||
"lsr x23, x4, x22",
|
||||
"orr x20, x20, x23, lsr #1",
|
||||
@@ -2237,7 +2237,7 @@
|
||||
"and x20, x7, #0x1f",
|
||||
"cbz x20, #+0x4c",
|
||||
"and x20, x7, #0x1f",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"uxth w22, w4",
|
||||
"bfi x22, x21, #16, #1",
|
||||
"bfi x22, x22, #17, #17",
|
||||
@@ -2263,7 +2263,7 @@
|
||||
"and w20, w7, #0x1f",
|
||||
"cbz x20, #+0x4c",
|
||||
"lsr w20, w4, w7",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"neg w22, w7",
|
||||
"lsl w23, w4, w22",
|
||||
"orr w20, w20, w23, lsl #1",
|
||||
@@ -2290,7 +2290,7 @@
|
||||
"and x20, x7, #0x3f",
|
||||
"cbz x20, #+0x48",
|
||||
"lsr x20, x4, x7",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"neg x22, x7",
|
||||
"lsl x23, x4, x22",
|
||||
"orr x20, x20, x23, lsl #1",
|
||||
@@ -2856,7 +2856,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"uxtb w27, w4",
|
||||
"mov w20, #0x1",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"lsl w0, w27, #24",
|
||||
"cmn w0, w20, lsl #24",
|
||||
"add w26, w27, #0x1 (1)",
|
||||
@@ -2872,7 +2872,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"uxtb w27, w4",
|
||||
"mov w20, #0x1",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"lsl w0, w27, #24",
|
||||
"cmp w0, w20, lsl #24",
|
||||
"sub w26, w27, #0x1 (1)",
|
||||
@@ -2888,7 +2888,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"uxth w27, w4",
|
||||
"mov w20, #0x1",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"lsl w0, w27, #16",
|
||||
"cmn w0, w20, lsl #16",
|
||||
"add w26, w27, #0x1 (1)",
|
||||
@@ -2902,7 +2902,7 @@
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Comment": "GROUP4 0xfe /0",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"adds w26, w4, #0x1 (1)",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
@@ -2915,7 +2915,7 @@
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Comment": "GROUP4 0xfe /0",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"adds x26, x4, #0x1 (1)",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
@@ -2930,7 +2930,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"uxth w27, w4",
|
||||
"mov w20, #0x1",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"lsl w0, w27, #16",
|
||||
"cmp w0, w20, lsl #16",
|
||||
"sub w26, w27, #0x1 (1)",
|
||||
@@ -2944,7 +2944,7 @@
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Comment": "GROUP4 0xfe /1",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"subs w26, w4, #0x1 (1)",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
@@ -2957,7 +2957,7 @@
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Comment": "GROUP4 0xfe /1",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"subs x26, x4, #0x1 (1)",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
|
||||
@@ -106,7 +106,7 @@
|
||||
"Comment": "0x27",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxtb w20, w4",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"and x22, x20, #0xf",
|
||||
"cmp x22, #0x9 (9)",
|
||||
"cset x22, hi",
|
||||
@@ -135,7 +135,7 @@
|
||||
"Comment": "0x2f",
|
||||
"ExpectedArm64ASM": [
|
||||
"uxtb w20, w4",
|
||||
"cset w21, lo",
|
||||
"cset x21, lo",
|
||||
"and x22, x20, #0xf",
|
||||
"cmp x22, #0x9 (9)",
|
||||
"cset x22, hi",
|
||||
@@ -207,7 +207,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"uxth w27, w4",
|
||||
"mov w20, #0x1",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"lsl w0, w27, #16",
|
||||
"cmn w0, w20, lsl #16",
|
||||
"add w26, w27, #0x1 (1)",
|
||||
@@ -221,7 +221,7 @@
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Comment": "0x40",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"adds w26, w4, #0x1 (1)",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
@@ -236,7 +236,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"uxth w27, w4",
|
||||
"mov w20, #0x1",
|
||||
"cset w21, hs",
|
||||
"cset x21, hs",
|
||||
"lsl w0, w27, #16",
|
||||
"cmp w0, w20, lsl #16",
|
||||
"sub w26, w27, #0x1 (1)",
|
||||
@@ -264,7 +264,7 @@
|
||||
"ExpectedInstructionCount": 7,
|
||||
"Comment": "0x48",
|
||||
"ExpectedArm64ASM": [
|
||||
"cset w20, hs",
|
||||
"cset x20, hs",
|
||||
"subs w26, w4, #0x1 (1)",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
|
||||
@@ -228,8 +228,8 @@
|
||||
"Comment": "0x0f 0x2e",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"csetm x0, eq",
|
||||
"ccmn x26, x0, #nzCv, le"
|
||||
]
|
||||
@@ -239,8 +239,8 @@
|
||||
"Comment": "0x0f 0x2f",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"csetm x0, eq",
|
||||
"ccmn x26, x0, #nzCv, le"
|
||||
]
|
||||
|
||||
@@ -941,9 +941,9 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mrs x20, rndr",
|
||||
"bfxil x4, x20, #0, #16",
|
||||
"cset x20, eq",
|
||||
"mov w26, #0x1",
|
||||
"mov w27, #0x0",
|
||||
"cset w20, eq",
|
||||
"mov w0, w27",
|
||||
"bfi w0, w20, #29, #1",
|
||||
"mov w20, w0",
|
||||
@@ -956,9 +956,9 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mrs x20, rndr",
|
||||
"mov w4, w20",
|
||||
"cset x20, eq",
|
||||
"mov w26, #0x1",
|
||||
"mov w27, #0x0",
|
||||
"cset w20, eq",
|
||||
"mov w0, w27",
|
||||
"bfi w0, w20, #29, #1",
|
||||
"mov w20, w0",
|
||||
@@ -970,9 +970,9 @@
|
||||
"Comment": "GROUP9 0x0F 0xC7 /6",
|
||||
"ExpectedArm64ASM": [
|
||||
"mrs x4, rndr",
|
||||
"cset x20, eq",
|
||||
"mov w26, #0x1",
|
||||
"mov w27, #0x0",
|
||||
"cset w20, eq",
|
||||
"mov w0, w27",
|
||||
"bfi w0, w20, #29, #1",
|
||||
"mov w20, w0",
|
||||
@@ -985,9 +985,9 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mrs x20, rndrrs",
|
||||
"bfxil x4, x20, #0, #16",
|
||||
"cset x20, eq",
|
||||
"mov w26, #0x1",
|
||||
"mov w27, #0x0",
|
||||
"cset w20, eq",
|
||||
"mov w0, w27",
|
||||
"bfi w0, w20, #29, #1",
|
||||
"mov w20, w0",
|
||||
@@ -1000,9 +1000,9 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mrs x20, rndrrs",
|
||||
"mov w4, w20",
|
||||
"cset x20, eq",
|
||||
"mov w26, #0x1",
|
||||
"mov w27, #0x0",
|
||||
"cset w20, eq",
|
||||
"mov w0, w27",
|
||||
"bfi w0, w20, #29, #1",
|
||||
"mov w20, w0",
|
||||
@@ -1014,9 +1014,9 @@
|
||||
"Comment": "GROUP9 0x0F 0xC7 /7",
|
||||
"ExpectedArm64ASM": [
|
||||
"mrs x4, rndrrs",
|
||||
"cset x20, eq",
|
||||
"mov w26, #0x1",
|
||||
"mov w27, #0x0",
|
||||
"cset w20, eq",
|
||||
"mov w0, w27",
|
||||
"bfi w0, w20, #29, #1",
|
||||
"mov w20, w0",
|
||||
|
||||
@@ -157,8 +157,8 @@
|
||||
"Comment": "0x66 0x0f 0x2e",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"csetm x0, eq",
|
||||
"ccmn x26, x0, #nzCv, le"
|
||||
]
|
||||
@@ -168,8 +168,8 @@
|
||||
"Comment": "0x66 0x0f 0x2f",
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"csetm x0, eq",
|
||||
"ccmn x26, x0, #nzCv, le"
|
||||
]
|
||||
|
||||
@@ -3394,8 +3394,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"csetm x0, eq",
|
||||
"ccmn x26, x0, #nzCv, le"
|
||||
]
|
||||
@@ -3407,8 +3407,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"csetm x0, eq",
|
||||
"ccmn x26, x0, #nzCv, le"
|
||||
]
|
||||
@@ -3420,8 +3420,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp s16, s17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"csetm x0, eq",
|
||||
"ccmn x26, x0, #nzCv, le"
|
||||
]
|
||||
@@ -3433,8 +3433,8 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"fcmp d16, d17",
|
||||
"cset x26, vc",
|
||||
"mov w27, #0x0",
|
||||
"cset w26, vc",
|
||||
"csetm x0, eq",
|
||||
"ccmn x26, x0, #nzCv, le"
|
||||
]
|
||||
|
||||
@@ -517,12 +517,12 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
@@ -543,12 +543,12 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
@@ -569,12 +569,12 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
@@ -595,12 +595,12 @@
|
||||
"umov w20, v3.h[0]",
|
||||
"umov w21, v2.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
@@ -675,12 +675,12 @@
|
||||
"umov w20, v2.h[0]",
|
||||
"umov w21, v3.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
@@ -697,12 +697,12 @@
|
||||
"umov w20, v2.h[0]",
|
||||
"umov w21, v3.h[0]",
|
||||
"mov w27, #0x0",
|
||||
"mov w26, #0x1",
|
||||
"cmp x21, #0x0 (0)",
|
||||
"cset x21, ne",
|
||||
"cmp w20, #0x0 (0)",
|
||||
"mrs x20, nzcv",
|
||||
"bfi w20, w21, #29, #1",
|
||||
"mov w26, #0x1",
|
||||
"msr nzcv, x20"
|
||||
]
|
||||
},
|
||||
@@ -4925,7 +4925,7 @@
|
||||
"bic w20, w6, w20",
|
||||
"tst x7, #0xe0",
|
||||
"csel w4, w6, w20, ne",
|
||||
"cset w20, eq",
|
||||
"cset x20, eq",
|
||||
"cmp w4, #0x0 (0)",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
@@ -4943,7 +4943,7 @@
|
||||
"bic x20, x6, x20",
|
||||
"tst x7, #0xc0",
|
||||
"csel x4, x6, x20, ne",
|
||||
"cset w20, eq",
|
||||
"cset x20, eq",
|
||||
"cmp x4, #0x0 (0)",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
|
||||
@@ -680,7 +680,7 @@
|
||||
"neg w20, w6",
|
||||
"and w4, w6, w20",
|
||||
"cmp w4, #0x0 (0)",
|
||||
"cset w20, eq",
|
||||
"cset x20, eq",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
"msr nzcv, x21"
|
||||
@@ -695,7 +695,7 @@
|
||||
"neg x20, x6",
|
||||
"and x4, x6, x20",
|
||||
"cmp x4, #0x0 (0)",
|
||||
"cset w20, eq",
|
||||
"cset x20, eq",
|
||||
"mrs x21, nzcv",
|
||||
"bfi w21, w20, #29, #1",
|
||||
"msr nzcv, x21"
|
||||
|
||||
@@ -175,15 +175,15 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"ld1h {z2.h}, p2/z, [x4]",
|
||||
"ldrb w20, [x28, #1019]",
|
||||
"mov w21, #0x1",
|
||||
"sub w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
"strb w20, [x28, #1019]",
|
||||
"add x0, x28, x20, lsl #4",
|
||||
"str q2, [x0, #1040]",
|
||||
"ldrb w22, [x28, #1426]",
|
||||
"lsl w20, w21, w20",
|
||||
"orr w20, w22, w20",
|
||||
"ldrb w21, [x28, #1426]",
|
||||
"mov w22, #0x1",
|
||||
"lsl w20, w22, w20",
|
||||
"orr w20, w21, w20",
|
||||
"strb w20, [x28, #1426]"
|
||||
]
|
||||
},
|
||||
|
||||
File diff suppressed because it is too large.
Load diff
File diff suppressed because it is too large.
Load diff
Reference in new issue
Block a user