mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 10:00:16 +02:00
Merge pull request #3069 from Sonicadvance1/fix_redundant_load_rclse
IR:RCLSE: Partially reenables the RCLSE pass
This commit is contained in:
9 files changed
+158
-225
No files matched your search
@@ -1601,7 +1601,7 @@ DEF_OP(VBroadcastFromMem) {
|
||||
DEF_OP(Push) {
|
||||
const auto Op = IROp->C<IR::IROp_Push>();
|
||||
const auto ValueSize = Op->ValueSize;
|
||||
const auto Src = GetReg(Op->Value.ID());
|
||||
auto Src = GetReg(Op->Value.ID());
|
||||
const auto AddrSrc = GetReg(Op->Addr.ID());
|
||||
const auto Dst = GetReg(Node);
|
||||
|
||||
@@ -1617,6 +1617,22 @@ DEF_OP(Push) {
|
||||
}
|
||||
}
|
||||
|
||||
if (Src == AddrSrc) {
|
||||
// If the data source is the address source then we need to do some additional work.
|
||||
// This is because it is undefined behaviour to do a writeback on store operation where dest == src.
|
||||
// In the case of writeback where the source is the address there are multiple behaviours.
|
||||
// - SIGILL - Apple Silicon Behaviour
|
||||
// - Stores original value - Cortex behaviour
|
||||
// - Stores value after pre-index adjust adjust - Vixl simulator behaviour.
|
||||
// - Undefined value stored
|
||||
// - Undefined behaviour(!)
|
||||
|
||||
// In this path Src can end up overlapping both AddrSrc and Dst.
|
||||
// Move the data to a temporary and store from there instead.
|
||||
mov(TMP1, Src.X());
|
||||
Src = TMP1;
|
||||
}
|
||||
|
||||
if (NeedsMoveAfterwards) {
|
||||
switch (ValueSize) {
|
||||
case 1: {
|
||||
|
||||
@@ -3664,11 +3664,18 @@ DEF_OP(VUMulH) {
|
||||
// Do predicated to ensure upper-bits get zero as expected
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
|
||||
if (Dst != Vector1) {
|
||||
if (Dst == Vector1) {
|
||||
umulh(SubRegSize, Dst.Z(), Mask, Dst.Z(), Vector2.Z());
|
||||
}
|
||||
else if (Dst == Vector2) {
|
||||
umulh(SubRegSize, Dst.Z(), Mask, Dst.Z(), Vector1.Z());
|
||||
}
|
||||
else {
|
||||
// Destination register doesn't overlap either source.
|
||||
// NOTE: SVE umulh (predicated) is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector1.Z());
|
||||
umulh(SubRegSize, Dst.Z(), Mask, Dst.Z(), Vector2.Z());
|
||||
}
|
||||
umulh(SubRegSize, Dst.Z(), Mask, Dst.Z(), Vector2.Z());
|
||||
}
|
||||
else {
|
||||
umulh(SubRegSize, Dst.Z(), Vector1.Z(), Vector2.Z());
|
||||
@@ -3717,11 +3724,18 @@ DEF_OP(VSMulH) {
|
||||
// Do predicated to ensure upper-bits get zero as expected
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
|
||||
if (Dst != Vector1) {
|
||||
// NOTE: SVE smulh (predicated) is a destructive operation.
|
||||
if (Dst == Vector1) {
|
||||
smulh(SubRegSize, Dst.Z(), Mask, Dst.Z(), Vector2.Z());
|
||||
}
|
||||
else if (Dst == Vector2) {
|
||||
smulh(SubRegSize, Dst.Z(), Mask, Dst.Z(), Vector1.Z());
|
||||
}
|
||||
else {
|
||||
// Destination register doesn't overlap either source.
|
||||
// NOTE: SVE umulh (predicated) is a destructive operation.
|
||||
movprfx(Dst.Z(), Vector1.Z());
|
||||
smulh(SubRegSize, Dst.Z(), Mask, Dst.Z(), Vector2.Z());
|
||||
}
|
||||
smulh(SubRegSize, Dst.Z(), Mask, Dst.Z(), Vector2.Z());
|
||||
}
|
||||
else {
|
||||
smulh(SubRegSize, Dst.Z(), Vector1.Z(), Vector2.Z());
|
||||
|
||||
@@ -3899,19 +3899,7 @@ void OpDispatchBuilder::VectorVariableBlend(OpcodeArgs) {
|
||||
OrderedNode *Dest = LoadSource(FPRClass, Op, Op->Dest, Op->Flags, -1);
|
||||
OrderedNode *Src = LoadSource(FPRClass, Op, Op->Src[0], Op->Flags, -1);
|
||||
|
||||
OrderedNode *Mask{};
|
||||
|
||||
// The mask is hardcoded to be xmm0 in this instruction.
|
||||
// Reuse one of the incoming sources if it happens to be xmm0.
|
||||
if (Op->Dest.Data.GPR.GPR == X86State::REG_XMM_0) {
|
||||
Mask = Dest;
|
||||
}
|
||||
else if (Op->Src[0].IsGPR() && Op->Src[0].Data.GPR.GPR == X86State::REG_XMM_0) {
|
||||
Mask = Src;
|
||||
}
|
||||
else {
|
||||
Mask = LoadXMMRegister(0);
|
||||
}
|
||||
auto Mask = LoadXMMRegister(0);
|
||||
|
||||
// Each element is selected by the high bit of that element size
|
||||
// Dest[ElementIdx] = Xmm0[ElementIndex][HighBit] ? Src : Dest;
|
||||
|
||||
@@ -456,7 +456,10 @@ private:
|
||||
ContextMemberInfo *FindMemberInfo(ContextInfo *ClassifiedInfo, uint32_t Offset, uint8_t Size);
|
||||
ContextMemberInfo *RecordAccess(ContextMemberInfo *Info, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode = nullptr);
|
||||
ContextMemberInfo *RecordAccess(ContextInfo *ClassifiedInfo, FEXCore::IR::RegisterClassType RegClass, uint32_t Offset, uint8_t Size, LastAccessType AccessType, FEXCore::IR::OrderedNode *Node, FEXCore::IR::OrderedNode *StoreNode = nullptr);
|
||||
void CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit);
|
||||
|
||||
// Classify context loads and stores.
|
||||
bool ClassifyContextLoad(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode *CodeNode, FEXCore::IR::NodeIterator BlockEnd);
|
||||
bool ClassifyContextStore(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode *CodeNode, FEXCore::IR::OrderedNode *ValueNode);
|
||||
|
||||
// Block local Passes
|
||||
bool RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit);
|
||||
@@ -493,55 +496,31 @@ ContextMemberInfo *RCLSE::RecordAccess(ContextInfo *ClassifiedInfo, FEXCore::IR:
|
||||
return RecordAccess(Info, RegClass, Offset, Size, AccessType, ValueNode, StoreNode);
|
||||
}
|
||||
|
||||
void RCLSE::CalculateControlFlowInfo(FEXCore::IR::IREmitter *IREmit) {
|
||||
using namespace FEXCore;
|
||||
using namespace FEXCore::IR;
|
||||
bool RCLSE::ClassifyContextLoad(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode *CodeNode, FEXCore::IR::NodeIterator BlockEnd) {
|
||||
auto Info = FindMemberInfo(LocalInfo, Offset, Size);
|
||||
ContextMemberInfo PreviousMemberInfoCopy = *Info;
|
||||
RecordAccess(Info, Class, Offset, Size, ACCESS_READ, CodeNode);
|
||||
|
||||
OffsetToBlockMap.clear();
|
||||
auto CurrentIR = IREmit->ViewIR();
|
||||
|
||||
for (auto [BlockNode, BlockHeader] : CurrentIR.GetBlocks()) {
|
||||
BlockInfo *CurrentBlock = &OffsetToBlockMap.try_emplace(CurrentIR.GetID(BlockNode)).first->second;
|
||||
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
|
||||
switch (IROp->Op) {
|
||||
case IR::OP_CONDJUMP: {
|
||||
auto Op = IROp->CW<IR::IROp_CondJump>();
|
||||
|
||||
OrderedNode *TrueTargetNode = CurrentIR.GetNode(Op->TrueBlock);
|
||||
OrderedNode *FalseTargetNode = CurrentIR.GetNode(Op->FalseBlock);
|
||||
|
||||
CurrentBlock->Successors.emplace_back(TrueTargetNode);
|
||||
CurrentBlock->Successors.emplace_back(FalseTargetNode);
|
||||
|
||||
{
|
||||
auto Block = &OffsetToBlockMap.try_emplace(Op->TrueBlock.ID()).first->second;
|
||||
Block->Predecessors.emplace_back(BlockNode);
|
||||
}
|
||||
|
||||
{
|
||||
auto Block = &OffsetToBlockMap.try_emplace(Op->FalseBlock.ID()).first->second;
|
||||
Block->Predecessors.emplace_back(BlockNode);
|
||||
}
|
||||
|
||||
break;
|
||||
}
|
||||
case IR::OP_JUMP: {
|
||||
auto Op = IROp->CW<IR::IROp_Jump>();
|
||||
OrderedNode *TargetNode = CurrentIR.GetNode(Op->Header.Args[0]);
|
||||
CurrentBlock->Successors.emplace_back(TargetNode);
|
||||
|
||||
{
|
||||
auto Block = OffsetToBlockMap.try_emplace(Op->Header.Args[0].ID()).first;
|
||||
Block->second.Predecessors.emplace_back(BlockNode);
|
||||
}
|
||||
break;
|
||||
}
|
||||
default: break;
|
||||
}
|
||||
}
|
||||
if (IsReadAccess(PreviousMemberInfoCopy.Accessed) &&
|
||||
IsReadAccess(Info->Accessed) &&
|
||||
PreviousMemberInfoCopy.AccessRegClass == Info->AccessRegClass &&
|
||||
PreviousMemberInfoCopy.AccessOffset == Info->AccessOffset &&
|
||||
PreviousMemberInfoCopy.AccessSize == Size) {
|
||||
// Optimize the case of redundant reads of the same exact value.
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, PreviousMemberInfoCopy.ValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Class, Offset, Size, ACCESS_READ, PreviousMemberInfoCopy.ValueNode);
|
||||
return true;
|
||||
}
|
||||
// TODO: Optimize the case of Store->Load.
|
||||
return false;
|
||||
}
|
||||
|
||||
bool RCLSE::ClassifyContextStore(FEXCore::IR::IREmitter *IREmit, ContextInfo *LocalInfo, FEXCore::IR::RegisterClassType Class, uint32_t Offset, uint8_t Size, FEXCore::IR::OrderedNode *CodeNode, FEXCore::IR::OrderedNode *ValueNode) {
|
||||
auto Info = FindMemberInfo(LocalInfo, Offset, Size);
|
||||
Info = RecordAccess(Info, Class, Offset, Size, ACCESS_WRITE, ValueNode, CodeNode);
|
||||
// TODO: Optimize redundant stores.
|
||||
// ContextMemberInfo PreviousMemberInfoCopy = *Info;
|
||||
return false;
|
||||
}
|
||||
|
||||
/**
|
||||
@@ -597,123 +576,19 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
for (auto [CodeNode, IROp] : CurrentIR.GetCode(BlockNode)) {
|
||||
if (IROp->Op == OP_STORECONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreContext>();
|
||||
auto Info = FindMemberInfo(&LocalInfo, Op->Offset, IROp->Size);
|
||||
uint8_t LastClass = Info->AccessRegClass;
|
||||
uint32_t LastOffset = Info->AccessOffset;
|
||||
uint8_t LastSize = Info->AccessSize;
|
||||
LastAccessType LastAccess = Info->Accessed;
|
||||
OrderedNode *LastStoreNode = Info->StoreNode;
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_WRITE, CurrentIR.GetNode(Op->Value), CodeNode);
|
||||
|
||||
if (IsWriteAccess(LastAccess) &&
|
||||
LastClass == Op->Class &&
|
||||
LastOffset == Op->Offset &&
|
||||
LastSize <= IROp->Size) {
|
||||
// Remove the last store because this one overwrites it entirely
|
||||
// Happens when we store in to a location then store again
|
||||
IREmit->Remove(LastStoreNode);
|
||||
|
||||
if (LastSize < IROp->Size) {
|
||||
//fmt::print("RCLSE: Eliminated partial write\n");
|
||||
}
|
||||
Changed = true;
|
||||
}
|
||||
Changed |= ClassifyContextStore(IREmit, &LocalInfo, Op->Class, Op->Offset, IROp->Size, CodeNode, CurrentIR.GetNode(Op->Value));
|
||||
}
|
||||
else if (IROp->Op == OP_STOREREGISTER) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreRegister>();
|
||||
Changed |= ClassifyContextStore(IREmit, &LocalInfo, Op->Class, Op->Offset, IROp->Size, CodeNode, CurrentIR.GetNode(Op->Value));
|
||||
}
|
||||
else if (IROp->Op == OP_LOADREGISTER) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadRegister>();
|
||||
Changed |= ClassifyContextLoad(IREmit, &LocalInfo, Op->Class, Op->Offset, IROp->Size, CodeNode, BlockEnd);
|
||||
}
|
||||
else if (IROp->Op == OP_LOADCONTEXT) {
|
||||
auto Op = IROp->CW<IR::IROp_LoadContext>();
|
||||
auto Info = FindMemberInfo(&LocalInfo, Op->Offset, IROp->Size);
|
||||
RegisterClassType LastClass = Info->AccessRegClass;
|
||||
uint32_t LastOffset = Info->AccessOffset;
|
||||
uint8_t LastSize = Info->AccessSize;
|
||||
LastAccessType LastAccess = Info->Accessed;
|
||||
OrderedNode *LastValueNode = Info->ValueNode;
|
||||
OrderedNode *LastStoreNode = Info->StoreNode;
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, CodeNode);
|
||||
|
||||
if (IsWriteAccess(LastAccess) &&
|
||||
LastClass == Op->Class &&
|
||||
LastOffset == Op->Offset &&
|
||||
IROp->Size <= LastSize) {
|
||||
// If the last store matches this load value then we can replace the loaded value with the previous valid one
|
||||
|
||||
if (LastClass == GPRClass) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
|
||||
uint8_t TruncateSize = IREmit->GetOpSize(LastValueNode);
|
||||
|
||||
// Did store context do an implicit truncation?
|
||||
if (IREmit->GetOpSize(LastStoreNode) < TruncateSize)
|
||||
TruncateSize = IREmit->GetOpSize(LastStoreNode);
|
||||
|
||||
// Or are we doing a partial read
|
||||
if (IROp->Size < TruncateSize)
|
||||
TruncateSize = IROp->Size;
|
||||
|
||||
if (TruncateSize != IREmit->GetOpSize(LastValueNode)) {
|
||||
// We need to insert an explict truncation
|
||||
LastValueNode = IREmit->_Bfe(IR::SizeToOpSize(std::max<uint8_t>(4u, Info->AccessSize)), TruncateSize * 8, 0, LastValueNode);
|
||||
}
|
||||
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else if (LastClass == FPRClass) {
|
||||
if (LastSize == IROp->Size && LastSize == IREmit->GetOpSize(LastValueNode)) {
|
||||
if (IsFullAccess(Info->Accessed)) {
|
||||
// LoadCtx matches StoreCtx and Node Size
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
else {
|
||||
// If this load size is a partial load then it may be expecting a zext of
|
||||
// the vector element
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// zext to size
|
||||
LastValueNode = IREmit->_VMov(IROp->Size, LastValueNode);
|
||||
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
} else if (LastSize >= IROp->Size &&
|
||||
IROp->Size == IREmit->GetOpSize(LastValueNode)) {
|
||||
// LoadCtx is <= StoreCtx and Node is LoadCtx
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else if (LastSize >= IROp->Size &&
|
||||
IROp->Size < IREmit->GetOpSize(LastValueNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// trucate to size
|
||||
LastValueNode = IREmit->_VMov(IROp->Size, LastValueNode);
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else if (LastSize >= IROp->Size &&
|
||||
IROp->Size > IREmit->GetOpSize(LastValueNode)) {
|
||||
IREmit->SetWriteCursor(CodeNode);
|
||||
// zext to size
|
||||
LastValueNode = IREmit->_VMov(IROp->Size, LastValueNode);
|
||||
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
} else {
|
||||
//fmt::print("RCLSE: Not GPR class, missed, {}, lastS: {}, S: {}, Node S: {}\n", LastClass, LastSize, IROp->Size, IREmit->GetOpSize(LastValueNode));
|
||||
}
|
||||
}
|
||||
}
|
||||
else if (IsReadAccess(LastAccess) &&
|
||||
IsReadAccess(Info->Accessed) &&
|
||||
LastClass == Op->Class &&
|
||||
LastOffset == Op->Offset &&
|
||||
LastSize == IROp->Size) {
|
||||
// Did we read and then read again?
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, LastValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Op->Class, Op->Offset, IROp->Size, ACCESS_READ, LastValueNode);
|
||||
Changed = true;
|
||||
}
|
||||
Changed |= ClassifyContextLoad(IREmit, &LocalInfo, Op->Class, Op->Offset, IROp->Size, CodeNode, BlockEnd);
|
||||
}
|
||||
else if (IROp->Op == OP_STOREFLAG) {
|
||||
auto Op = IROp->CW<IR::IROp_StoreFlag>();
|
||||
@@ -803,8 +678,6 @@ bool RCLSE::RedundantStoreLoadElimination(FEXCore::IR::IREmitter *IREmit) {
|
||||
|
||||
bool RCLSE::Run(FEXCore::IR::IREmitter *IREmit) {
|
||||
FEXCORE_PROFILE_SCOPED("PassManager::RCLSE");
|
||||
// XXX: We don't do cross-block optimizations yet
|
||||
//CalculateControlFlowInfo(IREmit);
|
||||
bool Changed = false;
|
||||
|
||||
// Run up to 5 times
|
||||
|
||||
@@ -0,0 +1,22 @@
|
||||
%ifdef CONFIG
|
||||
{
|
||||
"RegData": {
|
||||
"RAX": "0xe0000010",
|
||||
"RSP": "0xe0000008"
|
||||
}
|
||||
}
|
||||
%endif
|
||||
|
||||
; FEX had a bug where a `push rsp` would generate an Arm64 instruction with undefined behaviour.
|
||||
; `push rsp` -> `str x8, [x8, #-8]!`
|
||||
; This instruction has constrained undefined behaviour.
|
||||
; On Cortex it stores the original value.
|
||||
; On Apple Silicon it raises a SIGILL.
|
||||
; It can also store undefined data or have undefined behaviour.
|
||||
; Test to ensure we don't generate undefined behaviour.
|
||||
mov rsp, 0xe0000010
|
||||
push rsp
|
||||
|
||||
mov rax, [rsp]
|
||||
|
||||
hlt
|
||||
@@ -758,16 +758,15 @@
|
||||
],
|
||||
"ExpectedArm64ASM": [
|
||||
"sub sp, sp, #0x120 (288)",
|
||||
"mov v4.16b, v16.16b",
|
||||
"mov w20, v17.s[3]",
|
||||
"mov w21, v17.s[2]",
|
||||
"mov w22, v4.s[3]",
|
||||
"mov w23, v4.s[2]",
|
||||
"mov w22, v16.s[3]",
|
||||
"mov w23, v16.s[2]",
|
||||
"str w23, [sp]",
|
||||
"mov w23, v17.s[1]",
|
||||
"mov w24, v17.s[0]",
|
||||
"mov w25, v4.s[1]",
|
||||
"mov w26, v4.s[0]",
|
||||
"mov w25, v16.s[1]",
|
||||
"mov w26, v16.s[0]",
|
||||
"mov w27, v16.s[0]",
|
||||
"mov w30, v16.s[1]",
|
||||
"str w30, [sp, #32]",
|
||||
@@ -857,6 +856,7 @@
|
||||
"add w21, w21, w23",
|
||||
"ldr w23, [sp, #128]",
|
||||
"add w21, w21, w23",
|
||||
"mov v4.16b, v16.16b",
|
||||
"mov v4.s[3], w20",
|
||||
"mov v4.s[2], w24",
|
||||
"mov v4.s[1], w21",
|
||||
|
||||
@@ -1263,6 +1263,17 @@
|
||||
"str d4, [x28, #752]"
|
||||
]
|
||||
},
|
||||
"packsswb mm0, mm0": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x0f 0x63",
|
||||
"ExpectedArm64ASM": [
|
||||
"ldr d4, [x28, #752]",
|
||||
"zip1 v4.2d, v4.2d, v4.2d",
|
||||
"sqxtn v4.8b, v4.8h",
|
||||
"str d4, [x28, #752]"
|
||||
]
|
||||
},
|
||||
"pcmpgtb mm0, mm1": {
|
||||
"ExpectedInstructionCount": 4,
|
||||
"Optimal": "Yes",
|
||||
@@ -2623,73 +2634,70 @@
|
||||
]
|
||||
},
|
||||
"cmpxchg eax, ebx": {
|
||||
"ExpectedInstructionCount": 20,
|
||||
"ExpectedInstructionCount": 19,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x0f 0xb1",
|
||||
"ExpectedArm64ASM": [
|
||||
"lsr w20, w7, #0",
|
||||
"mov x21, x4",
|
||||
"mov x22, x4",
|
||||
"ubfx x22, x21, #0, #32",
|
||||
"ubfx x23, x21, #0, #32",
|
||||
"ubfx x24, x22, #0, #32",
|
||||
"cmp x23, x24",
|
||||
"csel x25, x24, x23, eq",
|
||||
"cmp x23, x24",
|
||||
"cmp x22, x23",
|
||||
"csel x24, x23, x22, eq",
|
||||
"cmp x22, x23",
|
||||
"csel x20, x20, x21, eq",
|
||||
"cmp x25, x24",
|
||||
"csel x4, x22, x23, eq",
|
||||
"cmp x24, x23",
|
||||
"csel x4, x21, x22, eq",
|
||||
"mov x4, x20",
|
||||
"sub x20, x24, x25",
|
||||
"eor w21, w24, w25",
|
||||
"sub x20, x23, x24",
|
||||
"eor w21, w23, w24",
|
||||
"strb w21, [x28, #708]",
|
||||
"strb w20, [x28, #706]",
|
||||
"cmp w24, w25",
|
||||
"cmp w23, w24",
|
||||
"mrs x20, nzcv",
|
||||
"eor w20, w20, #0x20000000",
|
||||
"str w20, [x28, #728]"
|
||||
]
|
||||
},
|
||||
"cmpxchg [rax], ebx": {
|
||||
"ExpectedInstructionCount": 16,
|
||||
"ExpectedInstructionCount": 15,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x0f 0xb1",
|
||||
"ExpectedArm64ASM": [
|
||||
"lsr w20, w7, #0",
|
||||
"mov x21, x4",
|
||||
"lsr w22, w21, #0",
|
||||
"mov w1, w22",
|
||||
"lsr w21, w4, #0",
|
||||
"mov w1, w21",
|
||||
"casal w1, w20, [x4]",
|
||||
"mov w20, w1",
|
||||
"cmp w20, w22",
|
||||
"csel x4, x21, x20, eq",
|
||||
"sub w21, w22, w20",
|
||||
"eor w23, w22, w20",
|
||||
"cmp w20, w21",
|
||||
"csel x4, x4, x20, eq",
|
||||
"sub w22, w21, w20",
|
||||
"eor w23, w21, w20",
|
||||
"strb w23, [x28, #708]",
|
||||
"strb w21, [x28, #706]",
|
||||
"cmp w22, w20",
|
||||
"strb w22, [x28, #706]",
|
||||
"cmp w21, w20",
|
||||
"mrs x20, nzcv",
|
||||
"eor w20, w20, #0x20000000",
|
||||
"str w20, [x28, #728]"
|
||||
]
|
||||
},
|
||||
"cmpxchg rax, rbx": {
|
||||
"ExpectedInstructionCount": 16,
|
||||
"ExpectedInstructionCount": 15,
|
||||
"Optimal": "No",
|
||||
"Comment": "0x0f 0xb1",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov x20, x4",
|
||||
"mov x21, x4",
|
||||
"cmp x20, x21",
|
||||
"csel x22, x21, x20, eq",
|
||||
"cmp x20, x21",
|
||||
"csel x20, x7, x20, eq",
|
||||
"cmp x20, x20",
|
||||
"csel x21, x20, x20, eq",
|
||||
"cmp x20, x20",
|
||||
"csel x22, x7, x20, eq",
|
||||
"mov x4, x21",
|
||||
"mov x4, x22",
|
||||
"mov x4, x20",
|
||||
"sub x20, x21, x22",
|
||||
"eor x23, x21, x22",
|
||||
"sub x22, x20, x21",
|
||||
"eor x23, x20, x21",
|
||||
"strb w23, [x28, #708]",
|
||||
"strb w20, [x28, #706]",
|
||||
"cmp x21, x22",
|
||||
"strb w22, [x28, #706]",
|
||||
"cmp x20, x21",
|
||||
"mrs x20, nzcv",
|
||||
"eor w20, w20, #0x20000000",
|
||||
"str w20, [x28, #728]"
|
||||
@@ -2702,7 +2710,7 @@
|
||||
"ExpectedArm64ASM": [
|
||||
"mov x20, x4",
|
||||
"mov x1, x20",
|
||||
"casal x1, x7, [x4]",
|
||||
"casal x1, x7, [x20]",
|
||||
"mov x4, x1",
|
||||
"sub x21, x20, x4",
|
||||
"eor x22, x20, x4",
|
||||
|
||||
@@ -377,6 +377,16 @@
|
||||
"sqxtn2 v16.16b, v4.8h"
|
||||
]
|
||||
},
|
||||
"packsswb xmm0, xmm0": {
|
||||
"ExpectedInstructionCount": 3,
|
||||
"Optimal": "Yes",
|
||||
"Comment": "0x66 0x0f 0x63",
|
||||
"ExpectedArm64ASM": [
|
||||
"mov v0.16b, v16.16b",
|
||||
"sqxtn v16.8b, v16.8h",
|
||||
"sqxtn2 v16.16b, v0.8h"
|
||||
]
|
||||
},
|
||||
"pcmpgtb xmm0, xmm1": {
|
||||
"ExpectedInstructionCount": 1,
|
||||
"Optimal": "Yes",
|
||||
|
||||
@@ -8979,7 +8979,7 @@
|
||||
]
|
||||
},
|
||||
"fucompp": {
|
||||
"ExpectedInstructionCount": 73,
|
||||
"ExpectedInstructionCount": 74,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"0xda 11b 0xe9 /5"
|
||||
@@ -9049,9 +9049,10 @@
|
||||
"ldrb w22, [x28, #1010]",
|
||||
"lsl w23, w21, w20",
|
||||
"bic w22, w22, w23",
|
||||
"strb w22, [x28, #1010]",
|
||||
"add w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
"uxtb w22, w22",
|
||||
"ldrb w22, [x28, #1010]",
|
||||
"lsl w21, w21, w20",
|
||||
"bic w21, w22, w21",
|
||||
"strb w21, [x28, #1010]",
|
||||
@@ -19567,7 +19568,7 @@
|
||||
]
|
||||
},
|
||||
"fcompp": {
|
||||
"ExpectedInstructionCount": 73,
|
||||
"ExpectedInstructionCount": 74,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"0xde 11b 0xd9 /3"
|
||||
@@ -19637,9 +19638,10 @@
|
||||
"ldrb w22, [x28, #1010]",
|
||||
"lsl w23, w21, w20",
|
||||
"bic w22, w22, w23",
|
||||
"strb w22, [x28, #1010]",
|
||||
"add w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
"uxtb w22, w22",
|
||||
"ldrb w22, [x28, #1010]",
|
||||
"lsl w21, w21, w20",
|
||||
"bic w21, w22, w21",
|
||||
"strb w21, [x28, #1010]",
|
||||
|
||||
Reference in new issue
Block a user