Merge pull request #3119 from alyssarosenzweig/opt/x87-sel

Make x87 FCMOV slightly less terrible
This commit is contained in:
Ryan Houdek authored and GitHub committed 2023-09-19 10:34:29 -07:00
commit 65d558b2c4
6 files changed
+322 -517

No files matched your search

@@ -1285,13 +1285,19 @@ DEF_OP(Select) {
bool is_const_true = IsInlineConstant(Op->TrueVal, &const_true);
bool is_const_false = IsInlineConstant(Op->FalseVal, &const_false);
uint64_t all_ones = OpSize == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
ARMEmitter::Register Dst = GetReg(Node);
if (is_const_true || is_const_false) {
if (is_const_false != true || is_const_true != true || const_true != 1 || const_false != 0) {
if (is_const_false != true || is_const_true != true || !(const_true == 1 || const_true == all_ones) || const_false != 0) {
LOGMAN_MSG_A_FMT("Select: Unsupported compare inline parameters");
}
cset(EmitSize, Dst, cc);
if (const_true == all_ones)
csetm(EmitSize, Dst, cc);
else
cset(EmitSize, Dst, cc);
} else {
csel(EmitSize, Dst, GetReg(Op->TrueVal.ID()), GetReg(Op->FalseVal.ID()), cc);
}
@@ -1133,11 +1133,13 @@ OrderedNode *OpDispatchBuilder::SelectCCExplicitSize(uint8_t OP, IR::OpSize Resu
break;
}
case 0xA: { // JP - Jump if PF == 1
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ}, LoadPF(), ZeroConst, TrueValue, FalseValue);
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ},
LoadPFInverted(), ZeroConst, TrueValue, FalseValue);
break;
}
case 0xB: { // JNP - Jump if PF == 0
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_EQ}, LoadPF(), ZeroConst, TrueValue, FalseValue);
SrcCond = _Select(ResultSize, OpSize::i32Bit, CondClassType{COND_NEQ},
LoadPFInverted(), ZeroConst, TrueValue, FalseValue);
break;
}
case 0xC: { // SF <> OF
@@ -1286,73 +1286,45 @@ void OpDispatchBuilder::X87FXAM(OpcodeArgs) {
}
void OpDispatchBuilder::X87FCMOV(OpcodeArgs) {
enum CompareType {
COMPARE_ZERO,
COMPARE_NOTZERO,
};
uint32_t FLAGMask{};
CompareType Type = COMPARE_ZERO;
OrderedNode *SrcCond;
auto ZeroConst = _Constant(0);
auto OneConst = _Constant(1);
CalculateDeferredFlags();
uint16_t Opcode = Op->OP & 0b1111'1111'1000;
uint8_t CC = 0;
switch (Opcode) {
case 0x3'C0:
FLAGMask = 1 << FEXCore::X86State::RFLAG_CF_LOC;
Type = COMPARE_ZERO;
CC = 0x3; // JNC
break;
case 0x2'C0:
FLAGMask = 1 << FEXCore::X86State::RFLAG_CF_LOC;
Type = COMPARE_NOTZERO;
CC = 0x2; // JC
break;
case 0x2'C8:
FLAGMask = 1 << FEXCore::X86State::RFLAG_ZF_LOC;
Type = COMPARE_NOTZERO;
CC = 0x4; // JE
break;
case 0x3'C8:
FLAGMask = 1 << FEXCore::X86State::RFLAG_ZF_LOC;
Type = COMPARE_ZERO;
CC = 0x5; // JNE
break;
case 0x2'D0:
FLAGMask = (1 << FEXCore::X86State::RFLAG_ZF_LOC) | (1 << FEXCore::X86State::RFLAG_CF_LOC);
Type = COMPARE_NOTZERO;
CC = 0x6; // JNA
break;
case 0x3'D0:
FLAGMask = (1 << FEXCore::X86State::RFLAG_ZF_LOC) | (1 << FEXCore::X86State::RFLAG_CF_LOC);
Type = COMPARE_ZERO;
CC = 0x7; // JA
break;
case 0x2'D8:
FLAGMask = 1 << FEXCore::X86State::RFLAG_PF_LOC;
Type = COMPARE_NOTZERO;
CC = 0xA; // JP
break;
case 0x3'D8:
FLAGMask = 1 << FEXCore::X86State::RFLAG_PF_LOC;
Type = COMPARE_ZERO;
CC = 0xB; // JNP
break;
default:
LOGMAN_MSG_A_FMT("Unhandled FCMOV op: 0x{:x}", Opcode);
break;
}
auto RFLAG = GetPackedRFLAG(FLAGMask);
switch (Type) {
case COMPARE_ZERO: {
SrcCond = _Select(FEXCore::IR::COND_EQ,
RFLAG, ZeroConst, OneConst, ZeroConst);
break;
}
case COMPARE_NOTZERO: {
SrcCond = _Select(FEXCore::IR::COND_EQ,
RFLAG, ZeroConst, ZeroConst, OneConst);
break;
}
}
SrcCond = _Sbfe(OpSize::i64Bit, 1, 0, SrcCond);
auto ZeroConst = _Constant(0);
auto AllOneConst = _Constant(0xffff'ffff'ffff'ffffull);
OrderedNode *SrcCond = SelectCCExplicitSize(CC, OpSize::i64Bit, AllOneConst, ZeroConst);
OrderedNode *VecCond = _VDupFromGPR(16, 8, SrcCond);
auto top = GetX87Top();
@@ -1110,11 +1110,18 @@ bool ConstProp::ConstantInlining(IREmitter *IREmit, const IRListView& CurrentIR)
}
}
uint64_t AllOnes = IROp->Size == 8 ? 0xffff'ffff'ffff'ffffull : 0xffff'ffffull;
#ifdef JIT_ARM64
bool SupportsAllOnes = true;
#else
bool SupportsAllOnes = false;
#endif
uint64_t Constant2{};
uint64_t Constant3{};
if (IREmit->IsValueConstant(Op->Header.Args[2], &Constant2) &&
IREmit->IsValueConstant(Op->Header.Args[3], &Constant3) &&
Constant2 == 1 &&
(Constant2 == 1 || (SupportsAllOnes && Constant2 == AllOnes)) &&
Constant3 == 0)
{
IREmit->SetWriteCursor(CurrentIR.GetNode(Op->Header.Args[2]));
+48 -54
View File
@@ -617,22 +617,6 @@
]
},
"cmovpe ax, bx": {
"ExpectedInstructionCount": 9,
"Optimal": "No",
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"cmp w20, #0x0 (0)",
"csel w20, w7, w4, ne",
"bfxil x4, x20, #0, #16"
]
},
"cmovpe eax, ebx": {
"ExpectedInstructionCount": 8,
"Optimal": "No",
"Comment": "0x0f 0x4a",
@@ -642,58 +626,40 @@
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"cmp w20, #0x0 (0)",
"csel w4, w7, w4, ne"
]
},
"cmovpe rax, rbx": {
"ExpectedInstructionCount": 8,
"Optimal": "No",
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"cmp w20, #0x0 (0)",
"csel x4, x7, x4, ne"
]
},
"cmovnp ax, bx": {
"ExpectedInstructionCount": 9,
"Optimal": "No",
"Comment": "0x0f 0x4b",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"cmp w20, #0x0 (0)",
"csel w20, w7, w4, eq",
"bfxil x4, x20, #0, #16"
]
},
"cmovnp eax, ebx": {
"ExpectedInstructionCount": 8,
"cmovpe eax, ebx": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "0x0f 0x4b",
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"cmp w20, #0x0 (0)",
"csel w4, w7, w4, eq"
]
},
"cmovnp rax, rbx": {
"cmovpe rax, rbx": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "0x0f 0x4a",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"cmp w20, #0x0 (0)",
"csel x4, x7, x4, eq"
]
},
"cmovnp ax, bx": {
"ExpectedInstructionCount": 8,
"Optimal": "No",
"Comment": "0x0f 0x4b",
@@ -703,9 +669,37 @@
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"eor w20, w20, #0x1",
"cmp w20, #0x0 (0)",
"csel x4, x7, x4, eq"
"csel w20, w7, w4, ne",
"bfxil x4, x20, #0, #16"
]
},
"cmovnp eax, ebx": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "0x0f 0x4b",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"cmp w20, #0x0 (0)",
"csel w4, w7, w4, ne"
]
},
"cmovnp rax, rbx": {
"ExpectedInstructionCount": 7,
"Optimal": "No",
"Comment": "0x0f 0x4b",
"ExpectedArm64ASM": [
"ldrb w20, [x28, #706]",
"fmov s2, w20",
"cnt v2.16b, v2.16b",
"umov w20, v2.b[0]",
"and x20, x20, #0x1",
"cmp w20, #0x0 (0)",
"csel x4, x7, x4, ne"
]
},
"cmovl ax, bx": {
File diff suppressed because it is too large. Load diff