Merge pull request #3070 from Sonicadvance1/update_rcr_opsize

OpcodeDispatcher: Update 32/64-bit RCR for operating size
This commit is contained in:
Mai authored and GitHub committed 2023-09-11 15:33:34 -04:00
commit d029394c27
2 files changed
+22 -26

No files matched your search

@@ -2601,6 +2601,7 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
RCRSmallerOp(Op);
return;
}
const auto OpSize = OpSizeFromSrc(Op);
// Calculate flags early.
CalculateDeferredFlags();
@@ -2610,11 +2611,11 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
auto CF = GetRFLAG(FEXCore::X86State::RFLAG_CF_LOC);
// Res = Src >> Shift
OrderedNode *Res = _Lshr(OpSizeFromSrc(Op), Dest, Src);
OrderedNode *Res = _Lshr(OpSize, Dest, Src);
// Res |= (Src << (Size - Shift + 1));
OrderedNode *SrcShl = _Sub(OpSizeFromSrc(Op), _Constant(Size, Size + 1), Src);
auto TmpHigher = _Lshl(OpSizeFromSrc(Op), Dest, SrcShl);
OrderedNode *SrcShl = _Sub(OpSize, _Constant(Size, Size + 1), Src);
auto TmpHigher = _Lshl(OpSize, Dest, SrcShl);
auto One = _Constant(Size, 1);
auto Zero = _Constant(Size, 0);
@@ -2623,25 +2624,23 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
Src, One,
TmpHigher, Zero);
// TODO: Can use OpSizeFromSrc(Op)
Res = _Or(IR::SizeToOpSize(std::max<uint8_t>(4u, std::max(GetOpSize(Res), GetOpSize(CompareResult)))), Res, CompareResult);
Res = _Or(OpSize, Res, CompareResult);
// If Shift != 0 then we can inject the CF
OrderedNode *CFShl = _Sub(OpSizeFromSrc(Op), _Constant(Size, Size), Src);
OrderedNode *CFShl = _Sub(OpSize, _Constant(Size, Size), Src);
auto TmpCF = _Lshl(OpSize::i64Bit, CF, CFShl);
CompareResult = _Select(FEXCore::IR::COND_UGE,
Src, One,
TmpCF, Zero);
// TODO: Can use OpSizeFromSrc(Op)
Res = _Or(IR::SizeToOpSize(std::max<uint8_t>(4u, std::max(GetOpSize(Res), GetOpSize(CompareResult)))), Res, CompareResult);
Res = _Or(OpSize, Res, CompareResult);
StoreResult(GPRClass, Op, Res, -1);
// CF only changes if we actually shifted
// Our new CF will be bit (Shift - 1) of the source
auto NewCF = _Bfe(OpSizeFromSrc(Op), 1, 0, _Lshr(OpSizeFromSrc(Op), Dest, _Sub(OpSizeFromSrc(Op), Src, One)));
auto NewCF = _Bfe(OpSize, 1, 0, _Lshr(OpSize, Dest, _Sub(OpSize, Src, One)));
CompareResult = _Select(FEXCore::IR::COND_UGE,
Src, One,
NewCF, CF);
@@ -2651,9 +2650,8 @@ void OpDispatchBuilder::RCROp(OpcodeArgs) {
// OF is the top two MSBs XOR'd together
// Only when Shift == 1, it is undefined otherwise
// Only changed if shift isn't zero
const auto ResSize = IR::SizeToOpSize(std::max<uint8_t>(4u, GetOpSize(Res)));
auto OF = GetRFLAG(FEXCore::X86State::RFLAG_OF_LOC);
auto NewOF = _Xor(ResSize, _Bfe(ResSize, 1, Size - 1, Res), _Bfe(ResSize, 1, Size - 2, Res));
auto NewOF = _Xor(OpSize, _Bfe(OpSize, 1, Size - 1, Res), _Bfe(OpSize, 1, Size - 2, Res));
CompareResult = _Select(FEXCore::IR::COND_EQ,
Src, _Constant(0),
OF, NewOF);
+13 -15
View File
@@ -1234,7 +1234,7 @@
]
},
"rcr eax, 2": {
"ExpectedInstructionCount": 27,
"ExpectedInstructionCount": 26,
"Optimal": "No",
"Comment": "GROUP2 0xC1 /3",
"ExpectedArm64ASM": [
@@ -1251,8 +1251,7 @@
"lsl x25, x23, #30",
"cmp w20, #0x1 (1)",
"csel x25, x25, x26, hs",
"orr x24, x24, x25",
"lsr w4, w24, #0",
"orr w4, w24, w25",
"lsr w21, w21, #1",
"ubfx w21, w21, #0, #1",
"cmp w20, #0x1 (1)",
@@ -1260,9 +1259,9 @@
"mov w0, w22",
"bfi w0, w20, #29, #1",
"mov w20, w0",
"ubfx x21, x24, #31, #1",
"ubfx x22, x24, #30, #1",
"eor x21, x21, x22",
"lsr w21, w4, #31",
"ubfx w22, w4, #30, #1",
"eor w21, w21, w22",
"bfi w20, w21, #28, #1",
"str w20, [x28, #728]"
]
@@ -2498,7 +2497,7 @@
]
},
"rcr eax, cl": {
"ExpectedInstructionCount": 37,
"ExpectedInstructionCount": 36,
"Optimal": "No",
"Comment": "GROUP2 0xd3 /3",
"ExpectedArm64ASM": [
@@ -2519,10 +2518,9 @@
"lsl x25, x23, x25",
"cmp w20, #0x1 (1)",
"csel x25, x25, x26, hs",
"orr x24, x24, x25",
"lsr w4, w24, #0",
"sub w25, w20, #0x1 (1)",
"lsr w21, w21, w25",
"orr w4, w24, w25",
"sub w24, w20, #0x1 (1)",
"lsr w21, w21, w24",
"ubfx w21, w21, #0, #1",
"cmp w20, #0x1 (1)",
"csel w21, w21, w23, hs",
@@ -2530,11 +2528,11 @@
"bfi w0, w21, #29, #1",
"mov w21, w0",
"ubfx w22, w21, #28, #1",
"ubfx x23, x24, #31, #1",
"ubfx x24, x24, #30, #1",
"eor x23, x23, x24",
"lsr w23, w4, #31",
"ubfx w24, w4, #30, #1",
"eor w23, w23, w24",
"cmp x20, #0x0 (0)",
"csel x20, x22, x23, eq",
"csel w20, w22, w23, eq",
"mov w0, w21",
"bfi w0, w20, #28, #1",
"mov w20, w0",