mirror of
https://github.com/FEX-Emu/FEX.git
synced 2026-10-06 14:00:16 +02:00
Merge pull request #3144 from Sonicadvance1/optimize_redundant_store_load
RCLSE: Optimize redundant store->load operations
This commit is contained in:
3 files changed
+37
-21
No files matched your search
@@ -2544,12 +2544,18 @@ DEF_OP(VUShrSWide) {
|
||||
else if (HostSupportsSVE128) {
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
|
||||
auto ShiftRegister = ShiftScalar.Z();
|
||||
auto ShiftRegister = ShiftScalar;
|
||||
if (OpSize > 8) {
|
||||
// SVE wide shifts don't need to duplicate the low bits unless the OpSize is 16-bytes
|
||||
// Slightly more optimal for 8-byte opsize.
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||
ShiftRegister = VTMP1.Z();
|
||||
ShiftRegister = VTMP1;
|
||||
}
|
||||
|
||||
if (Dst == ShiftRegister) {
|
||||
// If destination aliases the shift vector then we need to move it temporarily.
|
||||
mov(VTMP1.Z(), ShiftRegister.Z());
|
||||
ShiftRegister = VTMP1;
|
||||
}
|
||||
|
||||
if (Dst != Vector) {
|
||||
@@ -2557,10 +2563,10 @@ DEF_OP(VUShrSWide) {
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
if (ElementSize == 8) {
|
||||
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
|
||||
lsr(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister.Z());
|
||||
}
|
||||
else {
|
||||
lsr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
|
||||
lsr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister.Z());
|
||||
}
|
||||
} else {
|
||||
// uqshl + ushr of 57-bits leaves 7-bits remaining.
|
||||
@@ -2611,12 +2617,18 @@ DEF_OP(VSShrSWide) {
|
||||
else if (HostSupportsSVE128) {
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
|
||||
auto ShiftRegister = ShiftScalar.Z();
|
||||
auto ShiftRegister = ShiftScalar;
|
||||
if (OpSize > 8) {
|
||||
// SVE wide shifts don't need to duplicate the low bits unless the OpSize is 16-bytes
|
||||
// Slightly more optimal for 8-byte opsize.
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||
ShiftRegister = VTMP1.Z();
|
||||
ShiftRegister = VTMP1;
|
||||
}
|
||||
|
||||
if (Dst == ShiftRegister) {
|
||||
// If destination aliases the shift vector then we need to move it temporarily.
|
||||
mov(VTMP1.Z(), ShiftRegister.Z());
|
||||
ShiftRegister = VTMP1;
|
||||
}
|
||||
|
||||
if (Dst != Vector) {
|
||||
@@ -2624,10 +2636,10 @@ DEF_OP(VSShrSWide) {
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
if (ElementSize == 8) {
|
||||
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
|
||||
asr(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister.Z());
|
||||
}
|
||||
else {
|
||||
asr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
|
||||
asr_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister.Z());
|
||||
}
|
||||
} else {
|
||||
// uqshl + ushr of 57-bits leaves 7-bits remaining.
|
||||
@@ -2678,12 +2690,18 @@ DEF_OP(VUShlSWide) {
|
||||
else if (HostSupportsSVE128) {
|
||||
const auto Mask = PRED_TMP_16B.Merging();
|
||||
|
||||
auto ShiftRegister = ShiftScalar.Z();
|
||||
auto ShiftRegister = ShiftScalar;
|
||||
if (OpSize > 8) {
|
||||
// SVE wide shifts don't need to duplicate the low bits unless the OpSize is 16-bytes
|
||||
// Slightly more optimal for 8-byte opsize.
|
||||
dup(ARMEmitter::SubRegSize::i64Bit, VTMP1.Z(), ShiftScalar.Z(), 0);
|
||||
ShiftRegister = VTMP1.Z();
|
||||
ShiftRegister = VTMP1;
|
||||
}
|
||||
|
||||
if (Dst == ShiftRegister) {
|
||||
// If destination aliases the shift vector then we need to move it temporarily.
|
||||
mov(VTMP1.Z(), ShiftRegister.Z());
|
||||
ShiftRegister = VTMP1;
|
||||
}
|
||||
|
||||
if (Dst != Vector) {
|
||||
@@ -2691,10 +2709,10 @@ DEF_OP(VUShlSWide) {
|
||||
movprfx(Dst.Z(), Vector.Z());
|
||||
}
|
||||
if (ElementSize == 8) {
|
||||
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
|
||||
lsl(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister.Z());
|
||||
}
|
||||
else {
|
||||
lsl_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister);
|
||||
lsl_wide(SubRegSize, Dst.Z(), Mask, Dst.Z(), ShiftRegister.Z());
|
||||
}
|
||||
} else {
|
||||
// uqshl + ushr of 57-bits leaves 7-bits remaining.
|
||||
|
||||
@@ -506,17 +506,17 @@ bool RCLSE::ClassifyContextLoad(FEXCore::IR::IREmitter *IREmit, ContextInfo *Loc
|
||||
ContextMemberInfo PreviousMemberInfoCopy = *Info;
|
||||
RecordAccess(Info, Class, Offset, Size, LastAccessType::READ, CodeNode);
|
||||
|
||||
if (IsReadAccess(PreviousMemberInfoCopy.Accessed) &&
|
||||
IsReadAccess(Info->Accessed) &&
|
||||
PreviousMemberInfoCopy.AccessRegClass == Info->AccessRegClass &&
|
||||
if (PreviousMemberInfoCopy.AccessRegClass == Info->AccessRegClass &&
|
||||
PreviousMemberInfoCopy.AccessOffset == Info->AccessOffset &&
|
||||
PreviousMemberInfoCopy.AccessSize == Size) {
|
||||
// Optimize the case of redundant reads of the same exact value.
|
||||
// This optimizes two cases:
|
||||
// - Previous access was a load, and we have a redundant load of the same value.
|
||||
// - Previous access was a store, and we are redundantly loading immediately after the store. Eliminating the store.
|
||||
IREmit->ReplaceAllUsesWithRange(CodeNode, PreviousMemberInfoCopy.ValueNode, IREmit->GetIterator(IREmit->WrapNode(CodeNode)), BlockEnd);
|
||||
RecordAccess(Info, Class, Offset, Size, LastAccessType::READ, PreviousMemberInfoCopy.ValueNode);
|
||||
return true;
|
||||
}
|
||||
// TODO: Optimize the case of Store->Load.
|
||||
// TODO: Optimize the case of partial loads.
|
||||
return false;
|
||||
}
|
||||
|
||||
|
||||
@@ -9039,7 +9039,7 @@
|
||||
]
|
||||
},
|
||||
"fucompp": {
|
||||
"ExpectedInstructionCount": 77,
|
||||
"ExpectedInstructionCount": 76,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"0xda 11b 0xe9 /5"
|
||||
@@ -9115,7 +9115,6 @@
|
||||
"strb w22, [x28, #1010]",
|
||||
"add w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
"ldrb w22, [x28, #1010]",
|
||||
"lsl w21, w21, w20",
|
||||
"bic w21, w22, w21",
|
||||
"strb w21, [x28, #1010]",
|
||||
@@ -19655,7 +19654,7 @@
|
||||
]
|
||||
},
|
||||
"fcompp": {
|
||||
"ExpectedInstructionCount": 77,
|
||||
"ExpectedInstructionCount": 76,
|
||||
"Optimal": "No",
|
||||
"Comment": [
|
||||
"0xde 11b 0xd9 /3"
|
||||
@@ -19731,7 +19730,6 @@
|
||||
"strb w22, [x28, #1010]",
|
||||
"add w20, w20, #0x1 (1)",
|
||||
"and w20, w20, #0x7",
|
||||
"ldrb w22, [x28, #1010]",
|
||||
"lsl w21, w21, w20",
|
||||
"bic w21, w22, w21",
|
||||
"strb w21, [x28, #1010]",
|
||||
|
||||
Reference in new issue
Block a user