Merge pull request #3144 from Sonicadvance1/optimize_redundant_store_load

RCLSE: Optimize redundant store->load operations
This commit is contained in:
Ryan Houdek authored and GitHub committed 2023-09-23 17:06:10 -07:00
commit 6dc5c0d3be
3 files changed
+37 -21

No files matched your search

@@ -2544,12 +2544,18 @@ DEF_OP(VUShrSWide) {
else if (HostSupportsSVE128) {
const auto Mask = PRED_TMP_16B.Merging();
auto ShiftRegister = ShiftScalar.Z();
auto ShiftRegister = ShiftScalar;
if (OpSize > 8) {
// SVE wide shifts don't need to duplicate the low bits unless the OpSize is 16-bytes
// Slightly more optimal for 8-byte opsize.
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
ShiftRegister = VTMP1.Z();
ShiftRegister = VTMP1;
}
if (Dst == ShiftRegister) {
// If destination aliases the shift vector then we need to move it temporarily.
mov(VTMP1.Z(), ShiftRegister.Z());
ShiftRegister = VTMP1;
}
if (Dst != Vector) {
@@ -2557,10 +2563,10 @@ DEF_OP(VUShrSWide) {
movprfx(Dst.Z(), Vector.Z());
}
if (ElementSize == 8) {
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister.Z());
}
else {
lsr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
lsr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister.Z());
}
} else {
// uqshl + ushr of 57-bits leaves 7-bits remaining.
@@ -2611,12 +2617,18 @@ DEF_OP(VSShrSWide) {
else if (HostSupportsSVE128) {
const auto Mask = PRED_TMP_16B.Merging();
auto ShiftRegister = ShiftScalar.Z();
auto ShiftRegister = ShiftScalar;
if (OpSize > 8) {
// SVE wide shifts don't need to duplicate the low bits unless the OpSize is 16-bytes
// Slightly more optimal for 8-byte opsize.
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
ShiftRegister = VTMP1.Z();
ShiftRegister = VTMP1;
}
if (Dst == ShiftRegister) {
// If destination aliases the shift vector then we need to move it temporarily.
mov(VTMP1.Z(), ShiftRegister.Z());
ShiftRegister = VTMP1;
}
if (Dst != Vector) {
@@ -2624,10 +2636,10 @@ DEF_OP(VSShrSWide) {
movprfx(Dst.Z(), Vector.Z());
}
if (ElementSize == 8) {
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister.Z());
}
else {
asr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
asr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister.Z());
}
} else {
// uqshl + ushr of 57-bits leaves 7-bits remaining.
@@ -2678,12 +2690,18 @@ DEF_OP(VUShlSWide) {
else if (HostSupportsSVE128) {
const auto Mask = PRED_TMP_16B.Merging();
auto ShiftRegister = ShiftScalar.Z();
auto ShiftRegister = ShiftScalar;
if (OpSize > 8) {
// SVE wide shifts don't need to duplicate the low bits unless the OpSize is 16-bytes
// Slightly more optimal for 8-byte opsize.
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
ShiftRegister = VTMP1.Z();
ShiftRegister = VTMP1;
}
if (Dst == ShiftRegister) {
// If destination aliases the shift vector then we need to move it temporarily.
mov(VTMP1.Z(), ShiftRegister.Z());
ShiftRegister = VTMP1;
}
if (Dst != Vector) {
@@ -2691,10 +2709,10 @@ DEF_OP(VUShlSWide) {
movprfx(Dst.Z(), Vector.Z());
}
if (ElementSize == 8) {
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister.Z());
}
else {
lsl_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
lsl_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister.Z());
}
} else {
// uqshl + ushr of 57-bits leaves 7-bits remaining.
@@ -506,17 +506,17 @@ bool RCLSE::ClassifyContextLoad(FEXCore::IR::IREmitter *IREmit, ContextInfo *Loc
ContextMemberInfo PreviousMemberInfoCopy = *Info;
RecordAccess(Info, Class, Offset, Size, LastAccessType::READ, CodeNode);
if (IsReadAccess(PreviousMemberInfoCopy.Accessed) &&
IsReadAccess(Info->Accessed) &&
PreviousMemberInfoCopy.AccessRegClass == Info->AccessRegClass &&
if (PreviousMemberInfoCopy.AccessRegClass == Info->AccessRegClass &&
PreviousMemberInfoCopy.AccessOffset == Info->AccessOffset &&
PreviousMemberInfoCopy.AccessSize == Size) {
// Optimize the case of redundant reads of the same exact value.
// This optimizes two cases:
// - Previous access was a load, and we have a redundant load of the same value.
// - Previous access was a store, and we are redundantly loading immediately after the store. Eliminating the store.
IREmit->ReplaceAllUsesWithRange(CodeNode, PreviousMemberInfoCopy.ValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
RecordAccess(Info, Class, Offset, Size, LastAccessType::READ, PreviousMemberInfoCopy.ValueNode);
return true;
}
// TODO: Optimize the case of Store->Load.
// TODO: Optimize the case of partial loads.
return false;
}
+2 -4
View File
@@ -9039,7 +9039,7 @@
]
},
"fucompp": {
"ExpectedInstructionCount": 77,
"ExpectedInstructionCount": 76,
"Optimal": "No",
"Comment": [
"0xda 11b 0xe9 /5"
@@ -9115,7 +9115,6 @@
"strb w22, [x28, #1010]",
"add w20, w20, #0x1 (1)",
"and w20, w20, #0x7",
"ldrb w22, [x28, #1010]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
"strb w21, [x28, #1010]",
@@ -19655,7 +19654,7 @@
]
},
"fcompp": {
"ExpectedInstructionCount": 77,
"ExpectedInstructionCount": 76,
"Optimal": "No",
"Comment": [
"0xde 11b 0xd9 /3"
@@ -19731,7 +19730,6 @@
"strb w22, [x28, #1010]",
"add w20, w20, #0x1 (1)",
"and w20, w20, #0x7",
"ldrb w22, [x28, #1010]",
"lsl w21, w21, w20",
"bic w21, w22, w21",
"strb w21, [x28, #1010]",